纯文本PDF压缩方案求助:现有代码仅支持含图片PDF压缩
纯文本PDF压缩优化方案求助
问题描述
我需要针对仅含文本、无图片的PDF文件的压缩方案。目前使用基于iText的Android代码进行PDF压缩,但该代码仅针对包含图片的PDF有效,对纯文本PDF几乎无压缩效果(仅能压缩100-200KB)。然而应用商店中的同类APP均能实现纯文本PDF的压缩,特此求助优化方案。
当前代码
(context as AppCompatActivity).lifecycleScope.launch(Dispatchers.IO) { try { val reader = PdfReader(inputPath, password.toByteArray()) //pdfOptimize(reader) compressReader(reader) saveReader(reader) reader.close() onPDFCompletion(outputPath) } catch (e: IOException) { Log.d("PDFCompressionActivityTEST", "execute: ${e.message}") onPDFFailed(e.message) } catch (e: DocumentException) { onPDFFailed(e.message) } catch (e: Exception) { onPDFFailed(e.message) } } @Throws(IOException::class) private fun compressReader(reader: PdfReader) { val n = reader.xrefSize var `object`: PdfObject? var stream: PRStream for (i in 0 until n) { `object` = reader.getPdfObject(i) if (`object` == null || !`object`.isStream) continue stream = `object` as PRStream compressStream(stream) } reader.removeUnusedObjects() } @Throws(IOException::class) private fun compressStream(stream: PRStream) { val pdfSubType = stream[PdfName.SUBTYPE] println(stream.type()) if (pdfSubType != null && pdfSubType.toString() == PdfName.IMAGE.toString()) { val image = PdfImageObject(stream) val imageBytes = image.imageAsBytes val bmp: Bitmap = BitmapFactory.decodeByteArray(imageBytes, 0, imageBytes.size) ?: return val width = bmp.width val height = bmp.height val outBitmap = Bitmap.createBitmap(width, height, Bitmap.Config.ARGB_8888) val outCanvas = Canvas(outBitmap) outCanvas.drawBitmap(bmp, 0f, 0f, null) val imgBytes = ByteArrayOutputStream() outBitmap.compress(Bitmap.CompressFormat.JPEG, quality, imgBytes) stream.clear() stream.setData(imgBytes.toByteArray(), false, PRStream.BEST_COMPRESSION) stream.put(PdfName.TYPE, PdfName.XOBJECT) stream.put(PdfName.SUBTYPE, PdfName.IMAGE) stream.put(PdfName.FILTER, PdfName.DCTDECODE) stream.put(PdfName.WIDTH, PdfNumber(width)) stream.put(PdfName.HEIGHT, PdfNumber(height)) stream.put(PdfName.BITSPERCOMPONENT, PdfNumber(8)) stream.put(PdfName.COLORSPACE, PdfName.DEVICERGB) } } @Throws(DocumentException::class, IOException::class) private fun saveReader(reader: PdfReader) { val stamper = PdfStamper(reader, FileOutputStream(outputPath)) stamper.setFullCompression() stamper.close() }
优化方案
1. 优化内容流压缩
当前代码仅处理图片流,纯文本PDF的核心体积来自内容流,需新增文本流的压缩逻辑:
修改compressStream方法,对非图片流使用最高压缩级别重新处理:
@Throws(IOException::class) private fun compressStream(stream: PRStream) { val pdfSubType = stream[PdfName.SUBTYPE] if (pdfSubType != null && pdfSubType.toString() == PdfName.IMAGE.toString()) { // 保留原图片压缩逻辑 val image = PdfImageObject(stream) val imageBytes = image.imageAsBytes val bmp: Bitmap = BitmapFactory.decodeByteArray(imageBytes, 0, imageBytes.size) ?: return val width = bmp.width val height = bmp.height val outBitmap = Bitmap.createBitmap(width, height, Bitmap.Config.ARGB_8888) val outCanvas = Canvas(outBitmap) outCanvas.drawBitmap(bmp, 0f, 0f, null) val imgBytes = ByteArrayOutputStream() outBitmap.compress(Bitmap.CompressFormat.JPEG, quality, imgBytes) stream.clear() stream.setData(imgBytes.toByteArray(), false, PRStream.BEST_COMPRESSION) stream.put(PdfName.TYPE, PdfName.XOBJECT) stream.put(PdfName.SUBTYPE, PdfName.IMAGE) stream.put(PdfName.FILTER, PdfName.DCTDECODE) stream.put(PdfName.WIDTH, PdfNumber(width)) stream.put(PdfName.HEIGHT, PdfNumber(height)) stream.put(PdfName.BITSPERCOMPONENT, PdfNumber(8)) stream.put(PdfName.COLORSPACE, PdfName.DEVICERGB) } else { // 对文本内容流重新压缩 val streamData = PdfReader.getStreamBytes(stream) stream.setData(streamData, true, PRStream.BEST_COMPRESSION) } }
2. 字体优化(纯文本PDF压缩核心)
纯文本PDF体积大的主要原因是嵌入字体,可通过以下方式优化:
- 移除未使用的字体子集:遍历文档资源中的字体,删除未被页面引用的字体对象
- 替换嵌入字体为系统字体(需注意版权和显示兼容性):
private fun optimizeFonts(reader: PdfReader) { val catalog = reader.catalog val resources = catalog.getAsDict(PdfName.RESOURCES) ?: return val fonts = resources.getAsDict(PdfName.FONT) ?: return val fontKeys = fonts.keys.iterator() while (fontKeys.hasNext()) { val key = fontKeys.next() val fontDict = fonts.getAsDict(key) ?: continue val baseFont = fontDict.getAsString(PdfName.BASEFONT) ?: continue // 针对常见嵌入字体(如Arial),移除嵌入文件,改为引用系统字体 if (baseFont.startsWith("/Arial") || baseFont.startsWith("/TimesNewRoman")) { fontDict.remove(PdfName.FONTFILE) fontDict.remove(PdfName.FONTFILE2) fontDict.remove(PdfName.FONTFILE3) fontDict.put(PdfName.EMBEDDED, PdfBoolean.FALSE) } } }
调用时在compressReader后添加optimizeFonts(reader)即可。
3. 强化对象清理与合并
在compressReader方法中补充更多清理操作:
@Throws(IOException::class) private fun compressReader(reader: PdfReader) { val n = reader.xrefSize var `object`: PdfObject? var stream: PRStream for (i in 0 until n) { `object` = reader.getPdfObject(i) if (`object` == null || !`object`.isStream) continue stream = `object` as PRStream compressStream(stream) } reader.removeUnusedObjects() // 合并重复命名目标 reader.consolidateNamedDestinations() // 清理冗余的页面资源 reader.removeRedundantObjects() }
4. 禁用不必要的PDF元素
如果不需要保留PDF的交互功能(如表单、注释),可以在保存时移除:
@Throws(DocumentException::class, IOException::class) private fun saveReader(reader: PdfReader) { val stamper = PdfStamper(reader, FileOutputStream(outputPath)) // 移除注释 stamper.removeAnnotations() // 移除表单域(若无需交互) stamper.getAcroFields().removeFields() stamper.setFullCompression() stamper.close() }
内容的提问来源于stack exchange,提问作者Kaunain
相关产品推荐
相关产品推荐

