如何为Azure Cognitive Search分块子项正确映射元数据?
问题描述
我正在使用自定义技能对图像进行OCR识别与数据分块,同时从CSV文件中提取额外上下文数据,尝试将其作为附加元数据添加到每个搜索结果中。目前父项可显示元数据,但子项元数据缺失,请问是否有方法正确映射这些元数据?
当前映射配置
{ "outputFieldMappings": [ { "sourceFieldName": "/document/metadata/special_code", "targetFieldName": "metadata_special_code" }, { "sourceFieldName": "/document/metadata/document_type", "targetFieldName": "metadata_document_type" }, { "sourceFieldName": "/document/metadata/location", "targetFieldName": "metadata_location" } ] }
搜索结果示例(子项元数据缺失)
{ "@search.score": 0.01515151560306549, "@search.rerankerScore": 0.8941482305526733, "@search.captions": [ { "text": "sample file.pdf.", "highlights": "<em>sample</em> file.pdf." }], "chunk_id":"<parent id>", "parent_id": null, "chunk": null, "title": "sample file.pdf", "metadata_special_code": "12345678", "metadata_document_type": "pdf", "metadata_location": "test-store/sample file.pdf" }, { "@search.score": 0.032786883413791656, "@search.rerankerScore": 0.9278492331504822, "@search.captions": [ { "text": "sample file.pdf. <text here>" }], "chunk_id":"<chunk id>", "parent_id":"<parent id>", "chunk": "<text here>", "title": "sample file.pdf", "metadata_special_code": null, "metadata_document_type": null, "metadata_location": null }
补充配置详情
索引定义
{ "@odata.context": "https://search.windows.net/$metadata#indexes/$entity", "@odata.etag": "", "name": "test", "defaultScoringProfile": null, "fields": [ { "name": "chunk_id", "type": "Edm.String", "searchable": true, "filterable": true, "retrievable": true, "sortable": true, "facetable": true, "key": true, "indexAnalyzer": null, "searchAnalyzer": null, "analyzer": "keyword", "normalizer": null, "dimensions": null, "vectorSearchProfile": null, "synonymMaps": [] }, { "name": "parent_id", "type": "Edm.String", "searchable": true, "filterable": true, "retrievable": true, "sortable": true, "facetable": true, "key": false, "indexAnalyzer": null, "searchAnalyzer": null, "analyzer": null, "normalizer": null, "dimensions": null, "vectorSearchProfile": null, "synonymMaps": [] }, { "name": "chunk", "type": "Edm.String", "searchable": true, "filterable": false, "retrievable": true, "sortable": false, "facetable": false, "key": false, "indexAnalyzer": null, "searchAnalyzer": null, "analyzer": null, "normalizer": null, "dimensions": null, "vectorSearchProfile": null, "synonymMaps": [] }, { "name": "title", "type": "Edm.String", "searchable": true, "filterable": true, "retrievable": true, "sortable": false, "facetable": false, "key": false, "indexAnalyzer": null, "searchAnalyzer": null, "analyzer": null, "normalizer": null, "dimensions": null, "vectorSearchProfile": null, "synonymMaps": [] }, { "name": "vector", "type": "Collection(Edm.Single)", "searchable": true, "filterable": false, "retrievable": true, "sortable": false, "facetable": false, "key": false, "indexAnalyzer": null, "searchAnalyzer": null, "analyzer": null, "normalizer": null, "dimensions": 1536, "vectorSearchProfile": "full-skill-test-profile", "synonymMaps": [] }, { "name": "metadata_cutomer_code", "type": "Edm.String", "searchable": true, "filterable": true, "retrievable": true, "sortable": true, "facetable": false, "key": false, "indexAnalyzer": null, "searchAnalyzer": null, "analyzer": null, "normalizer": null, "dimensions": null, "vectorSearchProfile": null, "synonymMaps": [] }, { "name": "metadata_document_type", "type": "Edm.String", "searchable": true, "filterable": true, "retrievable": true, "sortable": true, "facetable": false, "key": false, "indexAnalyzer": null, "searchAnalyzer": null, "analyzer": "standard.lucene", "normalizer": null, "dimensions": null, "vectorSearchProfile": null, "synonymMaps": [] }, { "name": "metadata_content", "type": "Edm.String", "searchable": true, "filterable": true, "retrievable": true, "sortable": true, "facetable": false, "key": false, "indexAnalyzer": null, "searchAnalyzer": null, "analyzer": "standard.lucene", "normalizer": null, "dimensions": null, "vectorSearchProfile": null, "synonymMaps": [] }, { "name": "metadata_customer_code", "type": "Edm.String", "searchable": true, "filterable": true, "retrievable": true, "sortable": true, "facetable": false, "key": false, "indexAnalyzer": null, "searchAnalyzer": null, "analyzer": "standard.lucene", "normalizer": null, "dimensions": null, "vectorSearchProfile": null, "synonymMaps": [] } ], "scoringProfiles": [], "corsOptions": null, "suggesters": [], "analyzers": [], "normalizers": [], "tokenizers": [], "tokenFilters": [], "charFilters": [], "encryptionKey": null, "similarity": { "@odata.type": "#Microsoft.Azure.Search.BM25Similarity", "k1": null, "b": null }, "semantic": { "defaultConfiguration": "full-skill-test-semantic-configuration", "configurations": [ { "name": "full-skill-test-semantic-configuration", "prioritizedFields": { "titleField": { "fieldName": "title" }, "prioritizedContentFields": [ { "fieldName": "chunk" } ], "prioritizedKeywordsFields": [] } } ] }, "vectorSearch": { "algorithms": [ { "name": "full-skill-test-algorithm", "kind": "hnsw", "hnswParameters": { "metric": "cosine", "m": 4, "efConstruction": 400, "efSearch": 500 }, "exhaustiveKnnParameters": null } ], "profiles": [ { "name": "full-skill-test-profile", "algorithm": "full-skill-test-algorithm", "vectorizer": "full-skill-test-vectorizer" } ], "vectorizers": [ { "name": "full-skill-test-vectorizer", "kind": "azureOpenAI", "azureOpenAIParameters": { "resourceUri": "https://openai.azure.com", "deploymentId": "text-embedding-ada-002", "apiKey": "<redacted>", "authIdentity": null }, "customWebApiParameters": null } ] } }
索引器定义
{ "@odata.context": "https://search.windows.net/$metadata#indexers/$entity", "@odata.etag": "", "name": "indexer", "description": null, "dataSourceName": "datasource", "skillsetName": "skillset", "targetIndexName": "index", "disabled": null, "schedule": null, "parameters": { "batchSize": null, "maxFailedItems": null, "maxFailedItemsPerBatch": null, "base64EncodeKeys": null, "configuration": { "dataToExtract": "contentAndMetadata", "parsingMode": "default", "imageAction": "generateNormalizedImagePerPage", "allowSkillsetToReadFileData": true } }, "fieldMappings": [ { "sourceFieldName": "metadata_storage_name", "targetFieldName": "title", "mappingFunction": null } ], "outputFieldMappings": [ { "sourceFieldName": "/document/ref_metadata/special_code", "targetFieldName": "metadata_special_code" }, { "sourceFieldName": "/document/ref_metadata/document_type", "targetFieldName": "metadata_document_type" }, { "sourceFieldName": "/document/ref_metadata/location", "targetFieldName": "metadata_location" } ], "cache": null, "encryptionKey": null }
技能集
{ "@odata.type": "#Microsoft.Skills.Custom.WebApiSkill", "name": "#2", "description": "", "context": "/document", "uri": "https://functionapp.azurewebsites.net/api/MetadataOutput?code=<code>", "httpMethod": "POST", "timeout": "PT3M50S", "batchSize": 1, "degreeOfParallelism": 1, "inputs": [ { "name": "document", "source": "/document/metadata_storage_name" } ], "outputs": [ { "name": "ref_metadata", "targetName": "output_metadata" } ], "httpHeaders": {} }
解决方案
核心问题原因
当前元数据仅附加在父文档(/document)层级,数据分块后生成的子项(通常是/document/pages/*或/document/chunks/*)未继承父文档元数据,导致索引器无法将元数据映射到子项。
具体解决步骤
1. 添加ShaperSkill传递元数据到子块
在技能集中加入ShaperSkill,将父文档的元数据与分块后的子内容合并,确保每个子块携带父级元数据:
{ "@odata.type": "#Microsoft.Skills.Util.ShaperSkill", "name": "shaper-metadata-to-chunks", "context": "/document/chunks/*", "inputs": [ { "name": "chunk", "source": "/document/chunks/*" }, { "name": "special_code", "source": "/document/ref_metadata/special_code" }, { "name": "document_type", "source": "/document/ref_metadata/document_type" }, { "name": "location", "source": "/document/ref_metadata/location" } ], "outputs": [ { "name": "output", "targetName": "chunk_with_metadata" } ] }
2. 更新索引器输出字段映射
修改索引器的outputFieldMappings,新增子项元数据的映射规则:
"outputFieldMappings": [ // 父项映射保持不变 { "sourceFieldName": "/document/ref_metadata/special_code", "targetFieldName": "metadata_special_code" }, { "sourceFieldName": "/document/ref_metadata/document_type", "targetFieldName": "metadata_document_type" }, { "sourceFieldName": "/document/ref_metadata/location", "targetFieldName": "metadata_location" }, // 新增子项元数据映射 { "sourceFieldName": "/document/chunks/*/chunk_with_metadata/special_code", "targetFieldName": "metadata_special_code" }, { "sourceFieldName": "/document/chunks/*/chunk_with_metadata/document_type", "targetFieldName": "metadata_document_type" }, { "sourceFieldName": "/document/chunks/*/chunk_with_metadata/location", "targetFieldName": "metadata_location" } ]
3. 补充索引缺失字段
从当前索引定义看,缺少metadata_special_code和metadata_location字段,需添加到索引的fields数组中:
{ "name": "metadata_special_code", "type": "Edm.String", "searchable": true, "filterable": true, "retrievable": true, "sortable": true, "facetable": false }, { "name": "metadata_location", "type": "Edm.String", "searchable": true, "filterable": true, "retrievable": true, "sortable": true, "facetable": false }
4. 重置并重新运行索引器
- 重置索引器清除现有数据:
az search indexer reset --name indexer --resource-group your-resource-group --service-name your-search-service
- 重新运行索引器:
az search indexer run --name indexer --resource-group your-resource-group --service-name your-search-service
替代方案:分块技能直接关联元数据
如果使用内置SplitSkill,可将技能context设为/document,后续通过ShaperSkill关联父元数据,步骤同上:
{ "@odata.type": "#Microsoft.Skills.Text.SplitSkill", "name": "split-skill", "context": "/document", "textSplitMode": "pages", "maximumPageLength": 1000, "inputs": [ { "name": "text", "source": "/document/content" }, { "name": "languageCode", "source": "/document/language" } ], "outputs": [ { "name": "textItems", "targetName": "chunks" } ] }
内容的提问来源于stack exchange,提问作者user23846015
相关产品推荐
相关产品推荐

