TensorFlow Lite卷积层量化:Pose模型边缘推理性能问询
模型量化与边缘设备推理速度问题
我正在尝试对模型进行量化,以提升其在Coral Edge或I.MX8这类边缘设备上的推理速度。根据I.MX机器学习指南文档建议,我选择TensorFlow Lite框架用于I.MX8 NPU和Coral TPU的推理。
我的研究方向为姿态估计,已成功将Openpose-Lite的PyTorch模型通过OpenVino转换为TensorFlow Lite格式,并完成边缘设备要求的全整数量化(Openpose-Lite模型可通过临时链接获取,7天有效)。
将量化后的模型与已优化的Posenet模型在两款边缘设备上的推理速度对比,结果如下:
| 加速器 | Posenet | Openpose Lite |
|---|---|---|
| I.MX8 NPU | 9 ms | 160 ms |
| Coral Edge TPU | 9 ms | 34.5 ms |
尽管Openpose-Lite的网络规模和卷积操作数是Posenet的2-3倍,推理时间理应更长,但I.MX8 NPU上的推理速度差距过大。通过Netron对比两个网络发现,优化后的Posenet卷积后无可见激活层,且采用逐层量化,而Openpose-Lite采用逐通道量化。
核心问题
- 为何Posenet在I.MX8 NPU上的推理速度远快于Openpose-Lite?
卷积层量化代码示例
from tensorflow_model_optimization.python.core.quantization.keras import quantizers import tensorflow_model_optimization as tfmot import tensorflow as tf import numpy as np import os class Default8BitQuantizeConfig(tfmot.quantization.keras.QuantizeConfig): """QuantizeConfig for non recurrent Keras layers.""" def __init__(self, weight_attrs, activation_attrs, quantize_output): self.weight_attrs = weight_attrs self.activation_attrs = activation_attrs self.quantize_output = quantize_output # TODO(pulkitb): For some layers such as Conv2D, per_axis should be True. # Add mapping for which layers support per_axis. self.weight_quantizer = quantizers.LastValueQuantizer( num_bits=8, per_axis=False, symmetric=True, narrow_range=True) self.activation_quantizer = quantizers.MovingAverageQuantizer( num_bits=8, per_axis=False, symmetric=False, narrow_range=False) def get_weights_and_quantizers(self, layer): return [(getattr(layer, weight_attr), self.weight_quantizer) for weight_attr in self.weight_attrs] def get_activations_and_quantizers(self, layer): return [(getattr(layer, activation_attr), self.activation_quantizer) for activation_attr in self.activation_attrs] def set_quantize_weights(self, layer, quantize_weights): if len(self.weight_attrs) != len(quantize_weights): raise ValueError( '`set_quantize_weights` called on layer {} with {} ' 'weight parameters, but layer expects {} values.'.format( layer.name, len(quantize_weights), len(self.weight_attrs))) for weight_attr, weight in zip(self.weight_attrs, quantize_weights): current_weight = getattr(layer, weight_attr) if current_weight.shape != weight.shape: raise ValueError('Existing layer weight shape {} is incompatible with' 'provided weight shape {}'.format( current_weight.shape, weight.shape)) setattr(layer, weight_attr, weight) def set_quantize_activations(self, layer, quantize_activations): if len(self.activation_attrs) != len(quantize_activations): raise ValueError( '`set_quantize_activations` called on layer {} with {} ' 'activation parameters, but layer expects {} values.'.format( layer.name, len(quantize_activations), len(self.activation_attrs))) for activation_attr, activation in \ zip(self.activation_attrs, quantize_activations): setattr(layer, activation_attr, activation) def get_output_quantizers(self, layer): if self.quantize_output: return [self.activation_quantizer] return [] @classmethod def from_config(cls, config): """Instantiates a `Default8BitQuantizeConfig` from its config. Args: config: Output of `get_config()`. Returns: A `Default8BitQuantizeConfig` instance. """ return cls(**config) def get_config(self): # TODO(pulkitb): Add weight and activation quantizer to config. # Currently it's created internally, but ideally the quantizers should be # part of the constructor and passed in from the registry. return { 'weight_attrs': self.weight_attrs, 'activation_attrs': self.activation_attrs, 'quantize_output': self.quantize_output } def __eq__(self, other): if not isinstance(other, Default8BitQuantizeConfig): return False return (self.weight_attrs == other.weight_attrs and self.activation_attrs == other.activation_attrs and self.weight_quantizer == other.weight_quantizer and self.activation_quantizer == other.activation_quantizer and self.quantize_output == other.quantize_output) def __ne__(self, other): return not self.__eq__(other) class Default8BitConvWeightsQuantizer(quantizers.LastValueQuantizer): """Quantizer for handling weights in Conv2D/DepthwiseConv2D layers.""" def __init__(self): """Construct LastValueQuantizer with params specific for TFLite Convs.""" super(Default8BitConvWeightsQuantizer, self).__init__( num_bits=8, per_axis=False, symmetric=True, narrow_range=True) def build(self, tensor_shape, name, layer): min_weight = layer.add_weight( name + '_min', shape=None, initializer=tf.keras.initializers.Constant(-6.0), trainable=False) max_weight = layer.add_weight( name + '_max', shape=None, initializer=tf.keras.initializers.Constant(6.0), trainable=False) return {'min_var': min_weight, 'max_var': max_weight} class CustomDefault8BitConvQuantizeConfig(Default8BitQuantizeConfig): """QuantizeConfig for Conv2D/DepthwiseConv2D layers.""" def __init__(self, weight_attrs, activation_attrs, quantize_output): super(CustomDefault8BitConvQuantizeConfig, self).__init__(weight_attrs, activation_attrs, quantize_output) self.weight_quantizer = Default8BitConvWeightsQuantizer() def setup_model(): quantize_annotate_model = tfmot.quantization.keras.quantize_annotate_model quantize_annotate_layer = tfmot.quantization.keras.quantize_annotate_layer model = quantize_annotate_model(tf.keras.Sequential([ quantize_annotate_layer(tf.keras.layers.Conv2D(64, kernel_size = (3, 3),input_shape=(28, 28, 1), padding = 'same', activation='relu')), quantize_annotate_layer(tf.keras.layers.Conv2D(32, kernel_size = (3, 3), padding = 'same', activation='relu'),quantize_config=CustomDefault8BitConvQuantizeConfig(['kernel'], ['activation'],True)), tf.keras.layers.Dropout(0.5), tf.keras.layers.Conv2D(16, kernel_size = (3, 3), padding = 'same', activation='relu'), tf.keras.layers.Dropout(0.25), tf.keras.layers.Flatten(), tf.keras.layers.Dense(10) ])) quantize_scope = tfmot.quantization.keras.quantize_scope with quantize_scope( {'CustomDefault8BitConvQuantizeConfig': CustomDefault8BitConvQuantizeConfig}): # Use `quantize_apply` to actually make the model quantization aware. quant_aware_model = tfmot.quantization.keras.quantize_apply(model) return quant_aware_model
内容的提问来源于stack exchange,提问作者Daniel Klauser
相关产品推荐
相关产品推荐

