.NET Core Web API提取PDF数据时遇已释放对象错误求助
问题:.NET Core Web API PDF提取接口报错“Cannot access a disposed object. Object name: 'ReferenceReadStream'”
我基于.NET Core Web API开发了PDF数据提取接口,在SwaggerUI测试时触发错误:Cannot access a disposed object. Object name: 'ReferenceReadStream',即使上传之前可用的文件也无法正常提取数据,相关代码如下:
using Aspose.Pdf; using Aspose.Pdf.Text; using Microsoft.AspNetCore.Http; using Microsoft.AspNetCore.Mvc; using System; using System.Data; using System.IO; using System.Linq; namespace TryingOutAPI.Controllers { [Route("api/[controller]")] [ApiController] public class ValuesController : ControllerBase { [HttpPost] public IActionResult ProcessPdfTables(IFormFile pdfFile) { try { if (pdfFile == null || pdfFile.Length == 0) { return BadRequest("No PDF file uploaded."); } // Load the PDF document from the uploaded file Aspose.Pdf.Document pdfDocument; using (var stream = pdfFile.OpenReadStream()) { pdfDocument = new Aspose.Pdf.Document(stream); } // Extract the pages with the tables DataTable[] tables = ExtractTablesFromPdf(pdfDocument, new int[] { 2, 3 }); // Access the first table from the list of extracted tables DataTable table1 = tables[0]; // Access the second table from the list of extracted tables DataTable table2 = tables[1]; // Specify the correct column names string[] columnsToExtract = { "Peak Name", "RT", "Area", "% Area", "RT Ratio", "Height" }; // Select the desired columns from table 1 DataTable table1Subset = SelectColumnsFromTable(table1, columnsToExtract); table1Subset = RemoveRowsWithNullValues(table1Subset); // Select the desired columns from table 2 DataTable table2Subset = SelectColumnsFromTable(table2, columnsToExtract); table2Subset = RemoveRowsWithNullValues(table2Subset); // Return the subsets of tables without null rows as JSON return Ok(new { Table1 = table1Subset, Table2 = table2Subset }); } catch (Exception ex) { // Handle any exceptions and return an error response return StatusCode(StatusCodes.Status500InternalServerError, ex.Message); } } private DataTable[] ExtractTablesFromPdf(Aspose.Pdf.Document pdfDocument, int[] pages) { DataTable[] tables = new DataTable[pages.Length]; for (int i = 0; i < pages.Length; i++) { int pageNumber = pages[i]; Page pdfPage = pdfDocument.Pages[pageNumber]; // Extract text from the page TextAbsorber textAbsorber = new TextAbsorber(); pdfPage.Accept(textAbsorber); string pageContent = textAbsorber.Text; tables[i] = ConvertTextToDataTable(pageContent); } return tables; } private DataTable SelectColumnsFromTable(DataTable table, string[] columnsToExtract) { DataTable subset = new DataTable(); foreach (string column in columnsToExtract) { DataColumn existingColumn = table.Columns.Cast<DataColumn>() .FirstOrDefault(c => c.ColumnName == column); if (existingColumn != null) { subset.Columns.Add(existingColumn.ColumnName); } } foreach (DataRow row in table.Rows) { DataRow newRow = subset.NewRow(); foreach (DataColumn column in subset.Columns) { newRow[column.ColumnName] = row[column.ColumnName]; } subset.Rows.Add(newRow); } return subset; } private DataTable RemoveRowsWithNullValues(DataTable table) { DataTable filteredTable = table.Clone(); foreach (DataRow row in table.Rows) { bool hasNullValues = row.ItemArray.Any(x => x is DBNull || string.IsNullOrWhiteSpace(x.ToString())); if (!hasNullValues) { filteredTable.ImportRow(row); } } return filteredTable; } private DataTable ConvertTextToDataTable(string text) { DataTable dataTable = new DataTable(); // Split the text into lines string[] lines = text.Split('\n'); // Extract column names from the first line string[] columnNames = lines[0].Split('\t'); // Add columns to the DataTable foreach (string columnName in columnNames) { dataTable.Columns.Add(columnName.Trim()); } // Extract data rows from subsequent lines for (int i = 1; i < lines.Length; i++) { string[] rowValues = lines[i].Split('\t'); // Create a new DataRow DataRow dataRow = dataTable.NewRow(); // Set values for each column in the row for (int j = 0; j < columnNames.Length; j++) { dataRow[j] = rowValues[j].Trim(); } // Add the row to the DataTable dataTable.Rows.Add(dataRow); } return dataTable; } } }
错误原因
问题出在PDF文档的加载逻辑:using块包裹了pdfFile.OpenReadStream(),当using块执行完毕后,上传文件的流会被自动释放。但Aspose.Pdf.Document默认采用延迟加载机制,不会在初始化时就把整个PDF内容读取到内存,后续调用pdfDocument.Pages[pageNumber]和TextAbsorber提取文本时,会尝试访问已经被释放的原始流,从而触发“无法访问已释放对象”的错误。
修复方案
将上传的PDF文件流复制到内存流中,再用内存流初始化Aspose.Pdf.Document。内存流会驻留在内存中,直到文档处理完成,避免流提前被释放的问题。
修正后的代码
重点修改ProcessPdfTables方法中加载PDF的部分:
using Aspose.Pdf; using Aspose.Pdf.Text; using Microsoft.AspNetCore.Http; using Microsoft.AspNetCore.Mvc; using System; using System.Data; using System.IO; using System.Linq; namespace TryingOutAPI.Controllers { [Route("api/[controller]")] [ApiController] public class ValuesController : ControllerBase { [HttpPost] public IActionResult ProcessPdfTables(IFormFile pdfFile) { try { if (pdfFile == null || pdfFile.Length == 0) { return BadRequest("No PDF file uploaded."); } // 修复:将上传文件流复制到内存流,避免原始流被提前释放 Aspose.Pdf.Document pdfDocument; using (var memoryStream = new MemoryStream()) { pdfFile.CopyTo(memoryStream); memoryStream.Position = 0; // 重置流指针到起始位置 pdfDocument = new Aspose.Pdf.Document(memoryStream); } // Extract the pages with the tables DataTable[] tables = ExtractTablesFromPdf(pdfDocument, new int[] { 2, 3 }); // Access the first table from the list of extracted tables DataTable table1 = tables[0]; // Access the second table from the list of extracted tables DataTable table2 = tables[1]; // Specify the correct column names string[] columnsToExtract = { "Peak Name", "RT", "Area", "% Area", "RT Ratio", "Height" }; // Select the desired columns from table 1 DataTable table1Subset = SelectColumnsFromTable(table1, columnsToExtract); table1Subset = RemoveRowsWithNullValues(table1Subset); // Select the desired columns from table 2 DataTable table2Subset = SelectColumnsFromTable(table2, columnsToExtract); table2Subset = RemoveRowsWithNullValues(table2Subset); // Return the subsets of tables without null rows as JSON return Ok(new { Table1 = table1Subset, Table2 = table2Subset }); } catch (Exception ex) { // Handle any exceptions and return an error response return StatusCode(StatusCodes.Status500InternalServerError, ex.Message); } } private DataTable[] ExtractTablesFromPdf(Aspose.Pdf.Document pdfDocument, int[] pages) { DataTable[] tables = new DataTable[pages.Length]; for (int i = 0; i < pages.Length; i++) { int pageNumber = pages[i]; Page pdfPage = pdfDocument.Pages[pageNumber]; // Extract text from the page TextAbsorber textAbsorber = new TextAbsorber(); pdfPage.Accept(textAbsorber); string pageContent = textAbsorber.Text; tables[i] = ConvertTextToDataTable(pageContent); } return tables; } private DataTable SelectColumnsFromTable(DataTable table, string[] columnsToExtract) { DataTable subset = new DataTable(); foreach (string column in columnsToExtract) { DataColumn existingColumn = table.Columns.Cast<DataColumn>() .FirstOrDefault(c => c.ColumnName == column); if (existingColumn != null) { subset.Columns.Add(existingColumn.ColumnName); } } foreach (DataRow row in table.Rows) { DataRow newRow = subset.NewRow(); foreach (DataColumn column in subset.Columns) { newRow[column.ColumnName] = row[column.ColumnName]; } subset.Rows.Add(newRow); } return subset; } private DataTable RemoveRowsWithNullValues(DataTable table) { DataTable filteredTable = table.Clone(); foreach (DataRow row in table.Rows) { bool hasNullValues = row.ItemArray.Any(x => x is DBNull || string.IsNullOrWhiteSpace(x.ToString())); if (!hasNullValues) { filteredTable.ImportRow(row); } } return filteredTable; } private DataTable ConvertTextToDataTable(string text) { DataTable dataTable = new DataTable(); // Split the text into lines string[] lines = text.Split('\n'); // Extract column names from the first line string[] columnNames = lines[0].Split('\t'); // Add columns to the DataTable foreach (string columnName in columnNames) { dataTable.Columns.Add(columnName.Trim()); } // Extract data rows from subsequent lines for (int i = 1; i < lines.Length; i++) { string[] rowValues = lines[i].Split('\t'); // Create a new DataRow DataRow dataRow = dataTable.NewRow(); // Set values for each column in the row for (int j = 0; j < columnNames.Length; j++) { dataRow[j] = rowValues[j].Trim(); } // Add the row to the DataTable dataTable.Rows.Add(dataRow); } return dataTable; } } }
额外说明
- 内存流的方式确保了PDF内容完全加载到内存,后续操作不再依赖原始上传文件流;
- 注意重置内存流的
Position到0,否则Aspose.Pdf会从流的末尾开始读取,导致加载失败; - 如果处理超大PDF文件,内存流可能会占用较多内存,此时可以考虑其他持久化方式(如临时文件),但对于常规大小的PDF,内存流是最简便的解决方案。
内容的提问来源于stack exchange,提问作者FluffyFox332
相关产品推荐
相关产品推荐

