TensorRT INT8训练后量化推理结果异常技术求助
问题:TensorRT INT8量化后ArcFace特征向量与FP16结果差异极大
原本该咨询NVIDIA官方,但因其支持服务口碑不佳,特向熟悉TensorRT C++ API的开发者求助。我无法在此提供完整最小复现示例,但已将可复现项目上传至GitHub仓库的int8分支,该项目用于帮助开发者学习TensorRT API,仅3个文件且文档完善。
目前我可正常运行FP32和FP16精度的ArcFace人脸识别模型推理,但切换至INT8量化时出现异常:INT8推理生成的特征向量与FP16结果差异极大。
我实现了继承自nvinfer1::IInt8EntropyCalibrator2的Int8EntropyCalibrator2类,用于读取校准数据并提供给TensorRT,类定义代码如下:
// Class used for int8 calibration class Int8EntropyCalibrator2 : public nvinfer1::IInt8EntropyCalibrator2 { public: Int8EntropyCalibrator2(int32_t batchSize, int32_t inputW, int32_t inputH, const std::string& calibDataDirPath, const std::string& calibTableName, const std::string& inputBlobName, const std::array<float, 3>& subVals = {0.f, 0.f, 0.f},const std::array<float, 3>& divVals = {1.f, 1.f, 1.f}, bool normalize = true, bool readCache = true); virtual ~Int8EntropyCalibrator2(); // Abstract base class methods which must be implemented int32_t getBatchSize () const noexcept override; bool getBatch (void *bindings[], char const *names[], int32_t nbBindings) noexcept override; void const * readCalibrationCache (std::size_t &length) noexcept override; void writeCalibrationCache (void const *ptr, std::size_t length) noexcept override; private: const int32_t m_batchSize; const int32_t m_inputW; const int32_t m_inputH; int32_t m_imgIdx; std::vector<std::string> m_imgPaths; size_t m_inputCount; const std::string m_calibTableName; const std::string m_inputBlobName; const std::array<float, 3> m_subVals; const std::array<float, 3> m_divVals; const bool m_normalize; const bool m_readCache; void* m_deviceInput; std::vector<char> m_calibCache; };
该类的实现代码如下:
Int8EntropyCalibrator2::Int8EntropyCalibrator2(int32_t batchSize, int32_t inputW, int32_t inputH, const std::string &calibDataDirPath, const std::string &calibTableName, const std::string &inputBlobName, const std::array<float, 3>& subVals, const std::array<float, 3>& divVals, bool normalize, bool readCache) : m_batchSize(batchSize) , m_inputW(inputW) , m_inputH(inputH) , m_imgIdx(0) , m_calibTableName(calibTableName) , m_inputBlobName(inputBlobName) , m_subVals(subVals) , m_divVals(divVals) , m_normalize(normalize) , m_readCache(readCache) { // Allocate GPU memory to hold the entire batch m_inputCount = 3 * inputW * inputH * batchSize; checkCudaErrorCode(cudaMalloc(&m_deviceInput, m_inputCount * sizeof(float))); // Read the name of all the files in the specified directory. if (!doesFileExist(calibDataDirPath)) { throw std::runtime_error("Error, directory at provided path does not exist: " + calibDataDirPath); } m_imgPaths = getFilesInDirectory(calibDataDirPath); if (m_imgPaths.size() < static_cast<size_t>(batchSize)) { throw std::runtime_error("There are fewer calibration images than the specified batch size!"); } // Randomize the calibration data auto rd = std::random_device {}; auto rng = std::default_random_engine { rd() }; std::shuffle(std::begin(m_imgPaths), std::end(m_imgPaths), rng); } int32_t Int8EntropyCalibrator2::getBatchSize() const noexcept { // Return the batch size return m_batchSize; } bool Int8EntropyCalibrator2::getBatch(void **bindings, const char **names, int32_t nbBindings) noexcept { // This method will read a batch of images into GPU memory, and place the pointer to the GPU memory in the bindings variable. if (m_imgIdx + m_batchSize > static_cast<int>(m_imgPaths.size())) { // There are not enough images left to satisfy an entire batch return false; } // Read the calibration images into memory for the current batch std::vector<cv::cuda::GpuMat> inputImgs; for (int i = m_imgIdx; i < m_imgIdx + m_batchSize; i++) { std::cout << "Reading image " << i << ": " << m_imgPaths[i] << std::endl; auto cpuImg = cv::imread(m_imgPaths[i]); if (cpuImg.empty()){ std::cout << "Fatal error: Unable to read image at path: " << m_imgPaths[i] << std::endl; return false; } cv::cuda::GpuMat gpuImg; gpuImg.upload(cpuImg); cv::cuda::cvtColor(gpuImg, gpuImg, cv::COLOR_BGR2RGB); // TODO: Define any preprocessing code here, such as resizing // In this example, we will assume the calibration images are already of the correct size inputImgs.emplace_back(std::move(gpuImg)); } // Convert the batch from NHWC to NCHW // ALso apply normalization, scaling, and mean subtraction auto mfloat = Engine::blobFromGpuMats(inputImgs, m_subVals, m_divVals, m_normalize); auto *dataPointer = mfloat.ptr<void>(); // Copy the GPU buffer to member variable so that it persists checkCudaErrorCode(cudaMemcpyAsync(m_deviceInput, dataPointer, m_inputCount * sizeof(float), cudaMemcpyDeviceToDevice)); m_imgIdx+= m_batchSize; if (std::string(names[0]) != m_inputBlobName) { std::cout << "Error: Incorrect input name provided!" << std::endl; return false; } bindings[0] = m_deviceInput; return true; } void const *Int8EntropyCalibrator2::readCalibrationCache(size_t &length) noexcept { std::cout << "Searching for calibration cache: " << m_calibTableName << std::endl; m_calibCache.clear(); std::ifstream input(m_calibTableName, std::ios::binary); input >> std::noskipws; if (m_readCache && input.good()) { std::cout << "Reading calibration cache: " << m_calibTableName << std::endl; std::copy(std::istream_iterator<char>(input), std::istream_iterator<char>(), std::back_inserter(m_calibCache)); } length = m_calibCache.size(); return length ? m_calibCache.data() : nullptr; } void Int8EntropyCalibrator2::writeCalibrationCache(const void *ptr, std::size_t length) noexcept { std::cout << "Writing calib cache: " << m_calibTableName << " Size: " << length << " bytes" << std::endl; std::ofstream output(m_calibTableName, std::ios::binary); output.write(reinterpret_cast<const char*>(ptr), length); } Int8EntropyCalibrator2::~Int8EntropyCalibrator2() { checkCudaErrorCode(cudaFree(m_deviceInput)); };
完整实现可查看仓库对应分支的代码。
复现步骤
- 克隆仓库并切换至int8分支,安装依赖后编译
- 按照README的「Sanity Check」部分获取ArcFace模型
- 运行FP16模式,得到特征向量:
-0.050293 -0.0993042 0.181152 0.144531 0.222656 0.217529 -0.290283 -0.0638428 0.234375 -0.176636 ...
- 将main.cpp中精度设置从
Precision::FP16改为Precision::INT8 - 下载并解压校准数据
- 将校准数据路径传入
Engine::build方法 - 重新编译运行,得到INT8特征向量:
-0.175003 -0.00527599 -0.128431 -0.147636 0.278055 0.0584708 -0.083089 -0.0100119 -0.185134 0.0172769 ...
可见INT8特征向量与FP16结果差异极大,请问问题出在哪里?相关INT8校准的文档和示例较少,恳请指点。
内容的提问来源于stack exchange,提问作者cyrusbehr
相关产品推荐
相关产品推荐

