You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C# .NET Core中PDF批量文本替换异常:仅替换首个字典项的解决方法

问题描述

我用iTextSharp实现PDF文本替换,基础功能正常,但代码只能替换字典中的首个键值对,无法完成所有文本替换。以下是我的实现代码,请指出问题并提供解决方案。

实现代码

PDFEdit类代码

public void ReplaceTextInPDF(string sourceFile, string descFile, Dictionary<string, string> textReplacements)
{
    ReplaceText(textReplacements, descFile, sourceFile);
}

private void ReplaceText(Dictionary<string, string> textReplacements, string outputFilePath, string inputFilePath)
{
    try
    {
        using (Stream inputPdfStream = new FileStream(inputFilePath, FileMode.Open, FileAccess.Read, FileShare.Read))
        using (Stream outputPdfStream = new FileStream(outputFilePath, FileMode.Create, FileAccess.Write, FileShare.ReadWrite))
        {
            PdfReader reader = new PdfReader(inputPdfStream);
            PdfStamper stamper = new PdfStamper(reader, outputPdfStream);

            foreach (var kvp in textReplacements)
            {
                string searchText = kvp.Key;
                string replaceText = kvp.Value;

                for (int i = 1; i <= reader.NumberOfPages; i++)
                {
                    var tt = new MyLocationTextExtractionStrategy(searchText);
                    var ex = PdfTextExtractor.GetTextFromPage(reader, i, tt);
                    foreach (var p in tt.myPoints)
                    {
                        iTextSharp.text.Image image = CreateTransparentImage(p.Rect.Width, p.Rect.Height);
                        image.SetAbsolutePosition(p.Rect.Left, (p.Rect.Top - 8));
                        stamper.GetOverContent(i).AddImage(image, true);

                        PdfContentByte cb = stamper.GetOverContent(i);
                        BaseFont bf = BaseFont.CreateFont(BaseFont.HELVETICA, BaseFont.CP1252, BaseFont.NOT_EMBEDDED);
                        cb.SetColorFill(BaseColor.BLACK);
                        cb.SetFontAndSize(bf, 7);
                        cb.BeginText();
                        cb.ShowTextAligned(1, replaceText, p.Rect.Left + 10, p.Rect.Top - 6, 0);
                        cb.EndText();
                    }
                }
            }

            stamper.Close();
        }
    }
    catch (Exception ex)
    {
        // Handle the exception
    }
}

private iTextSharp.text.Image CreateTransparentImage(float width, float height)
{
    Bitmap transparentBitmap = new Bitmap((int)width, (int)height);
    transparentBitmap.MakeTransparent();
    iTextSharp.text.Image image = iTextSharp.text.Image.GetInstance(transparentBitmap, new BaseColor(255, 255, 255));
    return image;
}

public class RectAndText
{
    public iTextSharp.text.Rectangle Rect;
    public String Text;
    public RectAndText(iTextSharp.text.Rectangle rect, String text)
    {
        this.Rect = rect;
        this.Text = text;
    }
}

public class MyLocationTextExtractionStrategy : LocationTextExtractionStrategy
{
    public List<RectAndText> myPoints = new List<RectAndText>();
    public String TextToSearchFor { get; set; }
    public System.Globalization.CompareOptions CompareOptions { get; set; }

    public MyLocationTextExtractionStrategy(String textToSearchFor, System.Globalization.CompareOptions compareOptions = System.Globalization.CompareOptions.None)
    {
        this.TextToSearchFor = textToSearchFor;
        this.CompareOptions = compareOptions;
    }

    public override void RenderText(TextRenderInfo renderInfo)
    {
        base.RenderText(renderInfo);
        var startPosition = System.Globalization.CultureInfo.CurrentCulture.CompareInfo.IndexOf(renderInfo.GetText(), this.TextToSearchFor, this.CompareOptions);
        if (startPosition < 0)
        {
            return;
        }

        var chars = renderInfo.GetCharacterRenderInfos().Skip(startPosition).Take(this.TextToSearchFor.Length).ToList();
        var firstChar = chars.First();
        var lastChar = chars.Last();

        var bottomLeft = firstChar.GetDescentLine().GetStartPoint();
        var topRight = lastChar.GetAscentLine().GetEndPoint();
        var rect = new iTextSharp.text.Rectangle(bottomLeft[Vector.I1], bottomLeft[Vector.I2], topRight[Vector.I1], topRight[Vector.I2]);
        this.myPoints.Add(new RectAndText(rect, this.TextToSearchFor));
    }
}

Main方法代码

static void Main(string[] args)
{
    string sourceFile = "D:\\Form.pdf";
    string destFile = "D:\\output.pdf";

    PDFEdit pdfObj = new PDFEdit();
    Dictionary<string, string> textReplacements = new Dictionary<string, string>
    {
        { "Phone", "Phone" },
        { "Carrier Legal Name", "Name" }
    };

    pdfObj.ReplaceTextInPDF(sourceFile, destFile, textReplacements);
}

问题分析

你的代码仅能匹配单个文本chunk内的内容,但PDF中的文本通常按单词拆分存储(每个单词是一个独立的text chunk)。当搜索多词短语(比如Carrier Legal Name)时,单个chunk中没有完整的目标字符串,导致RenderText方法无法匹配到结果,自然不会执行替换操作。而首个键值对Phone是单个单词,能在单个chunk中找到,所以可以正常替换。

另外,CreateTransparentImage方法存在隐患:Bitmap转iTextSharp Image时,透明区域可能未正确处理,导致覆盖原文本效果不佳,但这不是循环不生效的核心原因。

解决方案

1. 修改文本提取策略,支持跨chunk匹配短语

重写提取策略,先累积页面所有文本chunk,再整体匹配目标短语并定位边界框:

public class PhraseLocationTextExtractionStrategy : LocationTextExtractionStrategy
{
    public List<RectAndText> myPoints = new List<RectAndText>();
    private readonly string _targetPhrase;
    private readonly List<TextChunk> _allChunks = new List<TextChunk>();

    public PhraseLocationTextExtractionStrategy(string targetPhrase)
    {
        _targetPhrase = targetPhrase;
    }

    public override void RenderText(TextRenderInfo renderInfo)
    {
        base.RenderText(renderInfo);
        // 存储每个文本chunk的内容和位置信息
        var chunk = new TextChunk(renderInfo.GetText(), renderInfo.GetDescentLine().GetStartPoint(), renderInfo.GetAscentLine().GetEndPoint());
        _allChunks.Add(chunk);
    }

    // 处理完所有chunk后,查找短语位置
    public void FindPhraseLocations()
    {
        // 拼接所有chunk文本(根据PDF实际格式调整空格处理逻辑)
        string fullText = string.Join(" ", _allChunks.Select(c => c.Text));
        int index = fullText.IndexOf(_targetPhrase, StringComparison.OrdinalIgnoreCase);
        if (index < 0) return;

        // 定位短语对应的起始和结束chunk
        int currentLength = 0;
        int startChunkIndex = -1;
        int endChunkIndex = -1;
        int phraseLength = _targetPhrase.Length;

        for (int i = 0; i < _allChunks.Count; i++)
        {
            string chunkText = _allChunks[i].Text;
            if (startChunkIndex == -1 && currentLength + chunkText.Length >= index)
            {
                startChunkIndex = i;
            }
            if (currentLength + chunkText.Length >= index + phraseLength)
            {
                endChunkIndex = i;
                break;
            }
            currentLength += chunkText.Length + 1; // +1对应chunk间的空格
        }

        if (startChunkIndex == -1 || endChunkIndex == -1) return;

        // 生成短语的整体边界框
        var startChunk = _allChunks[startChunkIndex];
        var endChunk = _allChunks[endChunkIndex];
        float left = startChunk.BottomLeft[Vector.I1];
        float bottom = startChunk.BottomLeft[Vector.I2];
        float right = endChunk.TopRight[Vector.I1];
        float top = endChunk.TopRight[Vector.I2];

        var rect = new iTextSharp.text.Rectangle(left, bottom, right, top);
        myPoints.Add(new RectAndText(rect, _targetPhrase));
    }

    // 辅助类存储文本chunk信息
    private class TextChunk
    {
        public string Text { get; }
        public Vector BottomLeft { get; }
        public Vector TopRight { get; }

        public TextChunk(string text, Vector bottomLeft, Vector topRight)
        {
            Text = text;
            BottomLeft = bottomLeft;
            TopRight = topRight;
        }
    }
}

2. 调整ReplaceText方法的调用逻辑

修改页面处理流程,先提取所有chunk再查找短语位置,并优化原文本覆盖方式:

private void ReplaceText(Dictionary<string, string> textReplacements, string outputFilePath, string inputFilePath)
{
    try
    {
        using (Stream inputPdfStream = new FileStream(inputFilePath, FileMode.Open, FileAccess.Read, FileShare.Read))
        using (Stream outputPdfStream = new FileStream(outputFilePath, FileMode.Create, FileAccess.Write, FileShare.ReadWrite))
        {
            PdfReader reader = new PdfReader(inputPdfStream);
            PdfStamper stamper = new PdfStamper(reader, outputPdfStream);

            foreach (var kvp in textReplacements)
            {
                string searchText = kvp.Key;
                string replaceText = kvp.Value;

                for (int i = 1; i <= reader.NumberOfPages; i++)
                {
                    var tt = new PhraseLocationTextExtractionStrategy(searchText);
                    // 提取页面所有文本chunk
                    PdfTextExtractor.GetTextFromPage(reader, i, tt);
                    // 查找短语位置
                    tt.FindPhraseLocations();

                    foreach (var p in tt.myPoints)
                    {
                        PdfContentByte cb = stamper.GetOverContent(i);
                        // 用白色矩形覆盖原文本,效果更可靠
                        cb.SetColorFill(BaseColor.WHITE);
                        cb.Rectangle(p.Rect.Left, p.Rect.Bottom, p.Rect.Width, p.Rect.Height);
                        cb.Fill();

                        // 绘制新文本,调整位置确保对齐
                        BaseFont bf = BaseFont.CreateFont(BaseFont.HELVETICA, BaseFont.CP1252, BaseFont.NOT_EMBEDDED);
                        cb.SetColorFill(BaseColor.BLACK);
                        cb.SetFontAndSize(bf, 7);
                        cb.BeginText();
                        cb.ShowTextAligned(Element.ALIGN_LEFT, replaceText, p.Rect.Left + 1, p.Rect.Bottom + 1, 0);
                        cb.EndText();
                    }
                }
            }

            stamper.Close();
        }
    }
    catch (Exception ex)
    {
        // 添加具体异常处理,便于排查问题
        Console.WriteLine($"替换失败:{ex.Message}");
    }
}

3. 其他优化点

  • 用白色矩形代替透明图片覆盖原文本,避免透明处理失效的问题;
  • 调整文本绘制的位置参数,确保新文本与原文本对齐;
  • 完善异常处理逻辑,添加日志或控制台输出。

内容的提问来源于stack exchange,提问作者Robin Singh

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.18 22:51:59