You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

DocumentExtractionSkill未提取文档,仍显示Base64文本问题排查

问题描述

配置Azure Cognitive Search的索引器与技能集后,使用DocumentExtractionSkill未能正常提取文档内容,输出仍保留Base64文本而非预期的明文“hello”,即便采用微软官方示例也出现该问题。


技能集配置

{
  "@odata.context": "https://<resource>.search.windows.net/$metadata#skillsets/$entity",
  "@odata.etag": "\"0x8DBF5E21D1A188B\"",
  "name": "document-extraction-skill",
  "description": "",
  "skills": [
    {
      "@odata.type": "#Microsoft.Skills.Util.DocumentExtractionSkill",
      "name": "#1",
      "description": null,
      "context": "/document",
      "parsingMode": "default",
      "dataToExtract": "contentAndMetadata",
      "inputs": [
        {
          "name": "file_data",
          "source": "/document/file_data"
        }
      ],
      "outputs": [
        {
          "name": "content",
          "targetName": "extracted_content"
        },
        {
          "name": "normalized_images",
          "targetName": "extracted_normalized_images"
        }
      ],
      "configuration": {
        "imageAction": "generateNormalizedImages",
        "normalizedImageMaxWidth@odata.type": "#Int64",
        "normalizedImageMaxWidth": 2000,
        "normalizedImageMaxHeight@odata.type": "#Int64",
        "normalizedImageMaxHeight": 2000
      }
    },
    {
      "@odata.type": "#Microsoft.Skills.Util.ShaperSkill",
      "name": "#3",
      "description": null,
      "context": "/document",
      "inputs": [
        {
          "name": "content",
          "source": "/document/extracted_content"
        },
        {
          "name": "normalized_images",
          "source": "/document/extracted_normalized_images"
        }
      ],
      "outputs": [
        {
          "name": "output",
          "targetName": "object_projection"
        }
      ]
    }
  ],
  "cognitiveServices": {
    "@odata.type": "#Microsoft.Azure.Search.DefaultCognitiveServices",
    "description": null
  },
  "knowledgeStore": {
    "storageConnectionString": "<connectionstring>",
    "identity": null,
    "projections": [
      {
        "tables": [],
        "objects": [
          {
            "storageContainer": "search-experiment-data-container",
            "referenceKeyName": null,
            "generatedKeyName": "key",
            "source": "/document/object_projection",
            "sourceContext": null,
            "inputs": []
          }
        ],
        "files": []
      }
    ],
    "parameters": {
      "synthesizeGeneratedKeyName": true
    }
  },
  "indexProjections": null,
  "encryptionKey": null
}

索引器配置

{
  "@odata.context": "https://<resource>.search.windows.net/$metadata#indexers/$entity",
  "@odata.etag": "\"0x8DBF5E2A0476504\"",
  "name": "scanned-pdf-indexer",
  "description": null,
  "dataSourceName": "scanned-pdf-source",
  "skillsetName": "document-extraction-skill",
  "targetIndexName": "dummy-index-knowledge-store",
  "disabled": null,
  "schedule": null,
  "parameters": {
    "batchSize": null,
    "maxFailedItems": null,
    "maxFailedItemsPerBatch": null,
    "base64EncodeKeys": null,
    "configuration": {
      "indexedFileNameExtensions": ".json",
      "allowSkillsetToReadFileData": true,
      "imageAction": "generateNormalizedImages"
    }
  },
  "fieldMappings": [
    {
      "sourceFieldName": "metadata_storage_name",
      "targetFieldName": "id",
      "mappingFunction": {
        "name": "base64Encode",
        "parameters": null
      }
    }
  ],
  "outputFieldMappings": [
    {
      "sourceFieldName": "/document/extracted_content",
      "targetFieldName": "content"
    }
  ],
  "cache": null,
  "encryptionKey": null
}

输入数据

{
  "values": [
    {
      "recordId": "1",
      "data":
      {
        "file_data": {
          "$type": "file",
          "data": "aGVsbG8="
        }
      }
    }
  ]
}

错误输出

{"content":"{\r\n  \"values\": [\r\n    {\r\n      \"recordId\": \"1\",\r\n      \"data\":\r\n      {\r\n        \"file_data\": {\r\n          \"$type\": \"file\",\r\n          \"data\": \"aGVsbG8=\"\r\n        }\r\n      }\r\n    }\r\n  ]\r\n}\n","normalized_images":[]}

问题原因
  1. 索引器文件类型限制错误:索引器配置中indexedFileNameExtensions设为.json,限定仅处理JSON格式文件。输入内容为纯文本(Base64解码后是"hello"),无匹配的文件扩展名,导致DocumentExtractionSkill无法识别文件类型,未执行内容提取逻辑,直接返回原始输入结构。
  2. 输入缺少文件类型标识:输入的file_data未包含fileName字段来指定文件类型(比如"fileName": "test.txt"),技能无法推断内容格式,进一步导致提取失败。

内容的提问来源于stack exchange,提问作者Dan Hunex

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.04 12:14:54