InputStream未读取指定文件全部字节排查(CRC32归档校验)
在频繁断电的不稳定系统中,使用CheckedOutputStream/CheckedInputStream计算LZ4压缩TAR归档的CRC32校验和以保证数据完整性,但部分归档出现校验失败:校验时读取字节数比写入时少4字节,导致CRC不匹配。该问题仅在压缩特定文件组合(如SQLite数据库文件)时触发,若在归档开头添加额外文件则校验正常。
归档创建代码
/** * Create a lz4 tar archive of the specified files * * @param filesToCompress files to compress * @param zipFile destination archive file * @return long array of {crc32value, totalBytesWritten} * @throws IOException if io error occurs */ public static long[] makeTarArchive(File[] filesToCompress, File zipFile) throws IOException { //Create LZ4 archive final CRC32 cksum = new CRC32(); CountingOutputStream countingOutputStreamRef = null; try (OutputStream out = Files.newOutputStream(zipFile.toPath(), StandardOpenOption.CREATE); // Used to calculate checksum as bytes are written to the root OutputStream CheckedOutputStream checkedOutputStream = new CheckedOutputStream(out, cksum); // Used to count all bytes written to the root OutputStream (for troubleshooting) CountingOutputStream countingOutputStream = new CountingOutputStream(checkedOutputStream); BufferedOutputStream bufferedOutputStream = new BufferedOutputStream(countingOutputStream); LZ4FrameOutputStream lz4FrameOutputStream = new LZ4FrameOutputStream(bufferedOutputStream); TarArchiveOutputStream zipOut = new TarArchiveOutputStream(lz4FrameOutputStream)) { // Store reference for use after try-with-resources scope closes countingOutputStreamRef = countingOutputStream; for (File file : filesToCompress) { final TarArchiveEntry tarArchiveEntry = new TarArchiveEntry(file, file.getName()); tarArchiveEntry.setSize(file.length()); //Specify the size of this file to be archived zipOut.putArchiveEntry(tarArchiveEntry); //Allocate a new archive entry //Write file bytes to allocated archive entry space try (InputStream in = Files.newInputStream(file.toPath())) { byte[] buf = new byte[4096]; int n; while ((n = in.read(buf)) != -1) { zipOut.write(buf, 0, n); } } zipOut.closeArchiveEntry(); //Close entry. This method MUST be called for all file entries that contain data. } } return new long[] {cksum.getValue(), countingOutputStreamRef.getByteCount()}; }
CRC校验代码
/** * Calculates the crc32 value of a compressed archive, and the total number of bytes read during crc calculation * * @param archive the archive to check * @return long array of {crc32value, totalBytesRead} */ @SuppressWarnings("java:S3626") // 'Continue' is present for code readability private static long[] calculateArchiveCRC(final Path archive) { final CRC32 crc32 = new CRC32(); CountingInputStream countingInputStreamRef; try (InputStream fi = new FileInputStream(archive.toFile()); CheckedInputStream checkedInputStream = new CheckedInputStream(fi, crc32); CountingInputStream countingInputStream = new CountingInputStream(checkedInputStream); BufferedInputStream bi = new BufferedInputStream(countingInputStream); LZ4FrameInputStream lz4i = new LZ4FrameInputStream(bi); TarArchiveInputStream ti = new TarArchiveInputStream(lz4i)) { // Store reference for use after try-with-resources scope closes countingInputStreamRef = countingInputStream; while (ti.getNextTarEntry() != null) { continue; // Nothing to do - getNextTarEntry() reads all bytes in the current entry } } catch (IOException ioException) { LOG.error("Error checking CRC32 value for archive {}", archive, ioException); return new long[] {0L, 0L}; } return new long[] {crc32.getValue(), countingInputStreamRef.getByteCount()}; }
已尝试在校验方法中用逐字节读取替代continue,结果未改变。有可复现问题的归档:写入字节数301613791(与磁盘文件大小一致),校验时仅读取301613787字节,少4字节导致CRC不匹配。
核心问题根源
校验逻辑未读取完整压缩文件字节
创建时,CheckedOutputStream统计的是整个LZ4压缩文件的所有字节(包括LZ4帧头、压缩数据、帧尾4字节校验和);但校验时,依赖TarArchiveInputStream.getNextTarEntry()读取内容,当读完最后一个有效条目后,流读取就会停止,导致LZ4帧尾的4字节校验和未被读取,最终计数少4字节、CRC不匹配。特定文件组合触发的原因
只有当TAR归档的解压后内容结束时,LZ4FrameInputStream恰好还有未读取的压缩字节(即帧尾4字节)时才会触发问题。若归档开头添加额外文件,TAR结构的字节对齐会改变,使得TarArchiveInputStream读取时能间接触发LZ4FrameInputStream读取完整压缩字节,从而规避问题。
解决方法
根据你的需求,有两种修正方向:
方向1:验证压缩文件本身的完整性(推荐,与原创建逻辑匹配)
直接读取整个压缩文件的所有字节计算CRC,无需解析TAR结构,确保与创建时的统计范围完全一致:
private static long[] calculateArchiveCRC(final Path archive) { final CRC32 crc32 = new CRC32(); long totalBytesRead = 0; try (InputStream fi = Files.newInputStream(archive); CheckedInputStream checkedInputStream = new CheckedInputStream(fi, crc32); CountingInputStream countingInputStream = new CountingInputStream(checkedInputStream)) { byte[] buf = new byte[4096]; while (countingInputStream.read(buf) != -1) { // 仅需读取所有字节,无需处理内容 } totalBytesRead = countingInputStream.getByteCount(); } catch (IOException ioException) { LOG.error("Error checking CRC32 value for archive {}", archive, ioException); return new long[]{0L, 0L}; } return new long[]{crc32.getValue(), totalBytesRead}; }
方向2:验证解压后TAR归档的完整性
若需要确认TAR内容本身无损坏,需调整创建和校验逻辑,统一针对解压后的TAR内容计算CRC(包括TAR结束块):
调整后的创建代码
public static long[] makeTarArchive(File[] filesToCompress, File zipFile) throws IOException { final CRC32 cksum = new CRC32(); try (OutputStream out = Files.newOutputStream(zipFile.toPath(), StandardOpenOption.CREATE); BufferedOutputStream bufferedOutputStream = new BufferedOutputStream(out); LZ4FrameOutputStream lz4FrameOutputStream = new LZ4FrameOutputStream(bufferedOutputStream); // 将CheckedOutputStream移至TarArchiveOutputStream上游,统计解压后的TAR内容 CheckedOutputStream checkedOutputStream = new CheckedOutputStream(lz4FrameOutputStream, cksum); TarArchiveOutputStream zipOut = new TarArchiveOutputStream(checkedOutputStream); CountingOutputStream countingOutputStream = new CountingOutputStream(checkedOutputStream)) { for (File file : filesToCompress) { final TarArchiveEntry tarArchiveEntry = new TarArchiveEntry(file, file.getName()); tarArchiveEntry.setSize(file.length()); zipOut.putArchiveEntry(tarArchiveEntry); try (InputStream in = Files.newInputStream(file.toPath())) { byte[] buf = new byte[4096]; int n; while ((n = in.read(buf)) != -1) { zipOut.write(buf, 0, n); } } zipOut.closeArchiveEntry(); } zipOut.close(); // 确保写入TAR结束块 return new long[]{cksum.getValue(), countingOutputStream.getByteCount()}; } }
对应的校验代码
private static long[] calculateArchiveCRC(final Path archive) { final CRC32 crc32 = new CRC32(); long totalBytesRead = 0; try (InputStream fi = Files.newInputStream(archive); BufferedInputStream bi = new BufferedInputStream(fi); LZ4FrameInputStream lz4i = new LZ4FrameInputStream(bi); CheckedInputStream checkedInputStream = new CheckedInputStream(lz4i, crc32); CountingInputStream countingInputStream = new CountingInputStream(checkedInputStream); TarArchiveInputStream ti = new TarArchiveInputStream(countingInputStream)) { byte[] buf = new byte[4096]; TarArchiveEntry entry; while ((entry = ti.getNextTarEntry()) != null) { // 读取条目所有内容,确保TAR结束块被读取 while (ti.read(buf) != -1) { // 无需处理内容 } } totalBytesRead = countingInputStream.getByteCount(); } catch (IOException ioException) { LOG.error("Error checking CRC32 value for archive {}", archive, ioException); return new long[]{0L, 0L}; } return new long[]{crc32.getValue(), totalBytesRead}; }
内容的提问来源于stack exchange,提问作者Jake Henry

