模型推理存在间隔时CPU占用率异常升高问题求助
问题:推理间隔导致CPU占用异常升高
我用PyTorch构建图像分类模型并转成ONNX后,发现推理间隔存在时CPU占用率异常飙升,不管用Torch还是ONNX推理都有这个问题。
测试代码及结果
无间隔推理
%%time %%timeit -n 50 _ = [sd.infer_scenes([test_img]) for _ in range(5)]
每次循环耗时7.96 ms ± 340 µs(7次运行,每次50循环),CPU总耗时2.09 s,墙钟时间2.79 s
间隔0.1秒推理
%%time %%timeit -n 50 time.sleep(0.1) _ = [sd.infer_scenes([test_img]) for _ in range(5)]
每次循环耗时113 ms ± 1.79 ms,CPU总耗时7分24秒,墙钟时间39.5 s
间隔1e-5秒推理
%%time %%timeit -n 50 time.sleep(0.00001) _ = [sd.infer_scenes([test_img]) for _ in range(5)]
每次循环耗时15.6 ms ± 277 µs,CPU总耗时3.39 s,墙钟时间5.45 s
间隔0秒推理
%%time %%timeit -n 50 time.sleep(0) _ = [sd.infer_scenes([test_img]) for _ in range(5)]
每次循环耗时8.02 ms ± 391 µs,CPU总耗时2.05 s,墙钟时间2.81 s
SceneDetector.py 代码
import cv2 import time import torch from torch.autograd import Variable import torch.nn as nn import torch.nn.functional as F from torchvision.transforms import transforms # Define a convolution neural network class SceneNetwork(nn.Module): def __init__(self): super(SceneNetwork, self).__init__() self.channels = 1 self.shape = self.channels*228*228 kernel_size = 1 self.conv1 = nn.Conv2d(in_channels=1, out_channels=self.channels, kernel_size=kernel_size, stride=1, padding=1) self.bn1 = nn.BatchNorm2d(self.channels) self.conv2 = nn.Conv2d(in_channels=self.channels, out_channels=self.channels, kernel_size=kernel_size, stride=1, padding=1) self.bn2 = nn.BatchNorm2d(self.channels) self.fc1 = nn.Linear(self.shape, 6) def forward(self, input): output = F.relu(self.bn1(self.conv1(input))) output = F.relu(self.bn2(self.conv2(output))) output = output.view(-1, self.shape) output = self.fc1(output) return output import numpy as np import onnxruntime as rt def crop_bottom_right(image): # bottom_right height, width = image.shape[0:2] w_pct = 0.3 h_pct = w_pct*width/height left = int(width*(1-w_pct)) up = int(height*(1-h_pct)) return image[up: height, left: width] class SceneDetector: classes = ['others', 'inventory', 'minimap_l1', 'minimap_l2', 'map_l1', 'map_l2plus'] transformations = transforms.Compose([ transforms.Lambda(lambda img: crop_bottom_right(img)), transforms.Lambda(lambda img: cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)), transforms.Lambda(lambda img: cv2.Canny(image=img, threshold1=50, threshold2=100)), transforms.Lambda(lambda img: cv2.resize(img, (448, 448))), transforms.ToPILImage(), transforms.Resize((224, 224)), transforms.ToTensor(), transforms.Normalize((0.5), (0.5)) ]) def __init__(self, model_path, model = None, is_onnx = False): self.is_onnx = is_onnx if is_onnx: if not model: #self.model = rt.InferenceSession( # model_path, providers=rt.get_available_providers()) self.model = rt.InferenceSession( model_path, providers=['CUDAExecutionProvider']) else: self.model = model self.input_name = self.model.get_inputs()[0].name else: self.device = torch.device("cuda:0" if torch.cuda.is_available() else "cpu") if not model: self.model = SceneNetwork() self.model.load_state_dict(torch.load(model_path)) else: self.model = model self.model.to(self.device) self.model.eval() def infer_scenes(self, images): start = time.time() if self.is_onnx: images = [ self.transformations(x).numpy() for x in images ] outputs = self.model.run(None, {self.input_name: images})[0] #predicted = np.argmax(outputs, axis=1) _, predicted = torch.max(torch.tensor(outputs), 1) else: images = torch.stack([ self.transformations(x) for x in images ]) images = Variable(images.to(self.device)) start = time.time() outputs = self.model(images) _, predicted = torch.max(outputs, 1) print(f'Infer time: {time.time()-start}') predicted_class = [ self.classes[x] for x in predicted ] return predicted_class
已尝试的解决方案
- 更换延迟方式:试过
threading.Event.wait、asyncio的await,结果和time.sleep一致 - 子进程推理:用
multiprocessing.Pipe在子进程中运行模型,问题依旧
from SceneDetector import SceneDetector from multiprocessing import Pipe, Process from threading import Event import numpy as np import time import cv2 def SceneWorker(pipe_conn, model_path, is_onnx = False): sd = SceneDetector(model_path, is_onnx = is_onnx) while True: data = pipe_conn.recv() if data is None: break results = sd.infer_scenes(data) pipe_conn.send(results) #time.sleep(0) if __name__ == '__main__': model_path = 'models/scene_model.onnx' scene_conn, scene_child_conn = Pipe() scene_process = Process(target=SceneWorker, args=(scene_child_conn, model_path, True,)) scene_process.start() #delay_event = Event() test_img = cv2.imread('TestImages/1440P_1000M_8X.jpg') for _ in range(50): scene_conn.send([test_img for _ in range(5)]) results = scene_conn.recv()[0] print(results) time.sleep(0.1) #delay_event.wait(0.0) scene_conn.send(None) scene_process.join()
- ONNX异步推理:实现了
run_async回调方式,未解决问题
def infer_callback(self, results, user_data, err) -> None: self.results = results self.infer_completed.set() def async_infer_scenes(self, images): start = time.time() images = [ self.transformations(x).numpy() for x in images ] self.model.run_async(None, {self.input_name: images}, self.infer_callback, None) self.infer_completed.wait() self.infer_completed.clear() outputs = self.results[0] #predicted = np.argmax(outputs, axis=1) _, predicted = torch.max(torch.tensor(outputs), 1) print(f'Infer time: {time.time()-start}') predicted_class = [ self.classes[x] for x in predicted ] return predicted_class
请求帮助
寻求解决该CPU占用异常升高问题的方案。
内容的提问来源于stack exchange,提问作者Fawkes Pan
相关产品推荐
相关产品推荐

