You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

.NET Core Web API提取PDF数据时遇已释放对象错误求助

问题:.NET Core Web API PDF提取接口报错“Cannot access a disposed object. Object name: 'ReferenceReadStream'”

我基于.NET Core Web API开发了PDF数据提取接口,在SwaggerUI测试时触发错误:Cannot access a disposed object. Object name: 'ReferenceReadStream',即使上传之前可用的文件也无法正常提取数据,相关代码如下:

using Aspose.Pdf;
using Aspose.Pdf.Text;
using Microsoft.AspNetCore.Http;
using Microsoft.AspNetCore.Mvc;
using System;
using System.Data;
using System.IO;
using System.Linq;

namespace TryingOutAPI.Controllers
{
    [Route("api/[controller]")]
    [ApiController]
    public class ValuesController : ControllerBase
    {
        [HttpPost]
        public IActionResult ProcessPdfTables(IFormFile pdfFile)
        {
            try
            {
                if (pdfFile == null || pdfFile.Length == 0)
                {
                    return BadRequest("No PDF file uploaded.");
                }

                // Load the PDF document from the uploaded file
                Aspose.Pdf.Document pdfDocument;
                using (var stream = pdfFile.OpenReadStream())
                {
                    pdfDocument = new Aspose.Pdf.Document(stream);
                }

                // Extract the pages with the tables
                DataTable[] tables = ExtractTablesFromPdf(pdfDocument, new int[] { 2, 3 });

                // Access the first table from the list of extracted tables
                DataTable table1 = tables[0];
                // Access the second table from the list of extracted tables
                DataTable table2 = tables[1];

                // Specify the correct column names
                string[] columnsToExtract = { "Peak Name", "RT", "Area", "% Area", "RT Ratio", "Height" };

                // Select the desired columns from table 1
                DataTable table1Subset = SelectColumnsFromTable(table1, columnsToExtract);
                table1Subset = RemoveRowsWithNullValues(table1Subset);

                // Select the desired columns from table 2
                DataTable table2Subset = SelectColumnsFromTable(table2, columnsToExtract);
                table2Subset = RemoveRowsWithNullValues(table2Subset);

                // Return the subsets of tables without null rows as JSON
                return Ok(new { Table1 = table1Subset, Table2 = table2Subset });
            }
            catch (Exception ex)
            {
                // Handle any exceptions and return an error response
                return StatusCode(StatusCodes.Status500InternalServerError, ex.Message);
            }
        }

        private DataTable[] ExtractTablesFromPdf(Aspose.Pdf.Document pdfDocument, int[] pages)
        {
            DataTable[] tables = new DataTable[pages.Length];

            for (int i = 0; i < pages.Length; i++)
            {
                int pageNumber = pages[i];
                Page pdfPage = pdfDocument.Pages[pageNumber];

                // Extract text from the page
                TextAbsorber textAbsorber = new TextAbsorber();
                pdfPage.Accept(textAbsorber);
                string pageContent = textAbsorber.Text;

                tables[i] = ConvertTextToDataTable(pageContent);
            }

            return tables;
        }
        private DataTable SelectColumnsFromTable(DataTable table, string[] columnsToExtract)
        {
            DataTable subset = new DataTable();

            foreach (string column in columnsToExtract)
            {
                DataColumn existingColumn = table.Columns.Cast<DataColumn>()
                    .FirstOrDefault(c => c.ColumnName == column);
                if (existingColumn != null)
                {
                    subset.Columns.Add(existingColumn.ColumnName);
                }
            }

            foreach (DataRow row in table.Rows)
            {
                DataRow newRow = subset.NewRow();
                foreach (DataColumn column in subset.Columns)
                {
                    newRow[column.ColumnName] = row[column.ColumnName];
                }
                subset.Rows.Add(newRow);
            }

            return subset;
        }
        private DataTable RemoveRowsWithNullValues(DataTable table)
        {
            DataTable filteredTable = table.Clone();

            foreach (DataRow row in table.Rows)
            {
                bool hasNullValues = row.ItemArray.Any(x => x is DBNull || string.IsNullOrWhiteSpace(x.ToString()));
                if (!hasNullValues)
                {
                    filteredTable.ImportRow(row);
                }
            }

            return filteredTable;
        }
        private DataTable ConvertTextToDataTable(string text)
        {
            DataTable dataTable = new DataTable();

            // Split the text into lines
            string[] lines = text.Split('\n');

            // Extract column names from the first line
            string[] columnNames = lines[0].Split('\t');

            // Add columns to the DataTable
            foreach (string columnName in columnNames)
            {
                dataTable.Columns.Add(columnName.Trim());
            }

            // Extract data rows from subsequent lines
            for (int i = 1; i < lines.Length; i++)
            {
                string[] rowValues = lines[i].Split('\t');

                // Create a new DataRow
                DataRow dataRow = dataTable.NewRow();

                // Set values for each column in the row
                for (int j = 0; j < columnNames.Length; j++)
                {
                    dataRow[j] = rowValues[j].Trim();
                }

                // Add the row to the DataTable
                dataTable.Rows.Add(dataRow);
            }

            return dataTable;
        }

    }
}

错误原因

问题出在PDF文档的加载逻辑:using块包裹了pdfFile.OpenReadStream(),当using块执行完毕后,上传文件的流会被自动释放。但Aspose.Pdf.Document默认采用延迟加载机制,不会在初始化时就把整个PDF内容读取到内存,后续调用pdfDocument.Pages[pageNumber]和TextAbsorber提取文本时,会尝试访问已经被释放的原始流,从而触发“无法访问已释放对象”的错误。

修复方案

将上传的PDF文件流复制到内存流中,再用内存流初始化Aspose.Pdf.Document。内存流会驻留在内存中,直到文档处理完成,避免流提前被释放的问题。

修正后的代码

重点修改ProcessPdfTables方法中加载PDF的部分:

using Aspose.Pdf;
using Aspose.Pdf.Text;
using Microsoft.AspNetCore.Http;
using Microsoft.AspNetCore.Mvc;
using System;
using System.Data;
using System.IO;
using System.Linq;

namespace TryingOutAPI.Controllers
{
    [Route("api/[controller]")]
    [ApiController]
    public class ValuesController : ControllerBase
    {
        [HttpPost]
        public IActionResult ProcessPdfTables(IFormFile pdfFile)
        {
            try
            {
                if (pdfFile == null || pdfFile.Length == 0)
                {
                    return BadRequest("No PDF file uploaded.");
                }

                // 修复:将上传文件流复制到内存流,避免原始流被提前释放
                Aspose.Pdf.Document pdfDocument;
                using (var memoryStream = new MemoryStream())
                {
                    pdfFile.CopyTo(memoryStream);
                    memoryStream.Position = 0; // 重置流指针到起始位置
                    pdfDocument = new Aspose.Pdf.Document(memoryStream);
                }

                // Extract the pages with the tables
                DataTable[] tables = ExtractTablesFromPdf(pdfDocument, new int[] { 2, 3 });

                // Access the first table from the list of extracted tables
                DataTable table1 = tables[0];
                // Access the second table from the list of extracted tables
                DataTable table2 = tables[1];

                // Specify the correct column names
                string[] columnsToExtract = { "Peak Name", "RT", "Area", "% Area", "RT Ratio", "Height" };

                // Select the desired columns from table 1
                DataTable table1Subset = SelectColumnsFromTable(table1, columnsToExtract);
                table1Subset = RemoveRowsWithNullValues(table1Subset);

                // Select the desired columns from table 2
                DataTable table2Subset = SelectColumnsFromTable(table2, columnsToExtract);
                table2Subset = RemoveRowsWithNullValues(table2Subset);

                // Return the subsets of tables without null rows as JSON
                return Ok(new { Table1 = table1Subset, Table2 = table2Subset });
            }
            catch (Exception ex)
            {
                // Handle any exceptions and return an error response
                return StatusCode(StatusCodes.Status500InternalServerError, ex.Message);
            }
        }

        private DataTable[] ExtractTablesFromPdf(Aspose.Pdf.Document pdfDocument, int[] pages)
        {
            DataTable[] tables = new DataTable[pages.Length];

            for (int i = 0; i < pages.Length; i++)
            {
                int pageNumber = pages[i];
                Page pdfPage = pdfDocument.Pages[pageNumber];

                // Extract text from the page
                TextAbsorber textAbsorber = new TextAbsorber();
                pdfPage.Accept(textAbsorber);
                string pageContent = textAbsorber.Text;

                tables[i] = ConvertTextToDataTable(pageContent);
            }

            return tables;
        }
        private DataTable SelectColumnsFromTable(DataTable table, string[] columnsToExtract)
        {
            DataTable subset = new DataTable();

            foreach (string column in columnsToExtract)
            {
                DataColumn existingColumn = table.Columns.Cast<DataColumn>()
                    .FirstOrDefault(c => c.ColumnName == column);
                if (existingColumn != null)
                {
                    subset.Columns.Add(existingColumn.ColumnName);
                }
            }

            foreach (DataRow row in table.Rows)
            {
                DataRow newRow = subset.NewRow();
                foreach (DataColumn column in subset.Columns)
                {
                    newRow[column.ColumnName] = row[column.ColumnName];
                }
                subset.Rows.Add(newRow);
            }

            return subset;
        }
        private DataTable RemoveRowsWithNullValues(DataTable table)
        {
            DataTable filteredTable = table.Clone();

            foreach (DataRow row in table.Rows)
            {
                bool hasNullValues = row.ItemArray.Any(x => x is DBNull || string.IsNullOrWhiteSpace(x.ToString()));
                if (!hasNullValues)
                {
                    filteredTable.ImportRow(row);
                }
            }

            return filteredTable;
        }
        private DataTable ConvertTextToDataTable(string text)
        {
            DataTable dataTable = new DataTable();

            // Split the text into lines
            string[] lines = text.Split('\n');

            // Extract column names from the first line
            string[] columnNames = lines[0].Split('\t');

            // Add columns to the DataTable
            foreach (string columnName in columnNames)
            {
                dataTable.Columns.Add(columnName.Trim());
            }

            // Extract data rows from subsequent lines
            for (int i = 1; i < lines.Length; i++)
            {
                string[] rowValues = lines[i].Split('\t');

                // Create a new DataRow
                DataRow dataRow = dataTable.NewRow();

                // Set values for each column in the row
                for (int j = 0; j < columnNames.Length; j++)
                {
                    dataRow[j] = rowValues[j].Trim();
                }

                // Add the row to the DataTable
                dataTable.Rows.Add(dataRow);
            }

            return dataTable;
        }

    }
}

额外说明

  • 内存流的方式确保了PDF内容完全加载到内存,后续操作不再依赖原始上传文件流;
  • 注意重置内存流的Position到0,否则Aspose.Pdf会从流的末尾开始读取,导致加载失败;
  • 如果处理超大PDF文件,内存流可能会占用较多内存,此时可以考虑其他持久化方式(如临时文件),但对于常规大小的PDF,内存流是最简便的解决方案。

内容的提问来源于stack exchange,提问作者FluffyFox332

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.17 08:47:01