You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

OpenSearch Transform数组字段拆分生成新事件问题求助

OpenSearch Transform数组字段拆分问题解决方案

问题场景

执行OpenSearch Transform操作时,源数据中的数组字段(如serviceIdentifiers)被拆分成独立事件,每个数组元素对应一条转换后的记录,不符合预期。

源事件示例

{
  "_index": "collated_txn_health_2022.05",
  "_type": "_doc",
  "_id": "LAUpboIBh6CUatILrsN3",
  "_score": 1,
  "_source": {
    "timeInGMT": 0,
    "kpiId": 0,
    "compInstanceIdentifier": "d0352b7d-0484-4714-bbc8-eb67cbb7be70",
    "agentIdentifier": "ComponentAgent-171",
    "kpiIdentifier": "PACKETS_DROPPED",
    "categoryIdentifier": "Network Utilization",
    "applicationIdentifier": null,
    "serviceIdentifiers": [
      "Supervisor_Controller Service",
      "Event_Detector Service",
      "UI_Service",
      "Redis",
      "CC_Service"
    ],
    "clusterIdentifiers": [
      "a5c57ef5-4018-41b8-b727-27c8f8376c0e"
    ],
    "collectionInterval": 60,
    "value": "0.0",
    "kpiType": "Core",
    "groupAttribute": "ALL",
    "groupIdentifier": null,
    "watcherValue": null,
    "errorCode": null,
    "clusterOperation": null,
    "aggLevelInMins": 1,
    "error": false,
    "kpiGroup": false,
    "discovery": false,
    "maintenanceExcluded": false,
    "@timestamp": "2022-05-01T01:32:00.000Z"
  }
}

当前Transform任务配置

curl -u admin:admin -XPUT "http://XXX.XXX.XX.XXX9201/_plugins/_transform/my-array-job-2" -H 'Content-type: application/json' -d'
{
    "transform": {
        "schedule": {
            "interval": {
                "start_time": 1659705000000,
                "period": 1,
                "unit": "Minutes"
            }
        },
        "metadata_id": null,
        "updated_at": 1659456180000,
        "enabled": true,
        "enabled_at": 1659457620000,
        "description": "",
        "source_index": "collated_txn_health_2022.05",
        "data_selection_query": {
            "match_all": {
                "boost": 1
            }
          },
        "target_index": "transform_collated_txn_health_2022.05",
        "page_size": 1000,
        "groups": [
            {
                "date_histogram": {
                    "fixed_interval": "1m",
                    "source_field": "@timestamp",
                    "target_field": "@timestamp",
                    "timezone": "Asia/Calcutta"
                }
            },
            {
                "terms": {
                    "source_field": "clusterIdentifiers",
                    "target_field": "clusterIdentifiers"
                }
            },
            {
                "terms": {
                    "source_field": "serviceIdentifiers",
                    "target_field": "serviceIdentifiers"
                }
            },
            {
                "terms": {
                    "source_field": "compInstanceIdentifier",
                    "target_field": "compInstanceIdentifier"
                }
            },
            {
                "terms": {
                    "source_field": "agentIdentifier",
                    "target_field": "agentIdentifier"
                }
            }
        ],
        "aggregations": {
            "count_@timestamp": {
                "value_count": {
                    "field": "@timestamp"
                }
            }
        }
    }
}'

转换后异常事件示例

{
  "_index": "transform_heal_collated_txn_health_2022.05",
  "_type": "_doc",
  "_id": "ybK0McQ9NZrt9xdo9iWKbA",
  "_score": 1,
  "_source": {
    "transform._id": "my-array-job-2",
    "transform._doc_count": 2,
    "@timestamp": 1651365120000,
    "clusterIdentifiers": "a5c57ef5-4018-41b8-b727-27c8f8376c0e",
    "serviceIdentifiers": "Redis",
    "compInstanceIdentifier": "a5c57ef5-4018-41b8-b727-27c8f8376c0e",
    "agentIdentifier": "ComponentAgent-170",
    "count_@timestamp": 2
  }
},
{
  "_index": "transform_heal_collated_txn_health_2022.05",
  "_type": "_doc",
  "_id": "Wf-4KwnFaYuw9bL-V-9WEQ",
  "_score": 1,
  "_source": {
    "transform._id": "my-array-job-2",
    "transform._doc_count": 2,
    "@timestamp": 1651365120000,
    "clusterIdentifiers": "a5c57ef5-4018-41b8-b727-27c8f8376c0e",
    "serviceIdentifiers": "Redis_Server Service",
    "compInstanceIdentifier": "a5c57ef5-4018-41b8-b727-27c8f8376c0e",
    "agentIdentifier": "ComponentAgent-170",
    "count_@timestamp": 2
  }
}

问题根源

你在groups配置中对数组字段(serviceIdentifiers、clusterIdentifiers)使用了terms聚合——OpenSearch的terms聚合默认会展开数组,将数组中的每个元素作为独立的分组维度,因此每个数组元素都会触发生成一条单独的转换后事件。

解决方案

如果需要保留数组的完整性,不拆分事件,可通过以下两种方式修改Transform配置:

方案1:将数组字段从分组移至聚合,保留原始数组

把数组字段从groups中移除,改用top_hits聚合获取原始数组内容,同时保留其他分组维度:

curl -u admin:admin -XPUT "http://XXX.XXX.XX.XXX9201/_plugins/_transform/my-array-job-2" -H 'Content-type: application/json' -d'
{
    "transform": {
        "schedule": {
            "interval": {
                "start_time": 1659705000000,
                "period": 1,
                "unit": "Minutes"
            }
        },
        "metadata_id": null,
        "updated_at": 1659456180000,
        "enabled": true,
        "enabled_at": 1659457620000,
        "description": "",
        "source_index": "collated_txn_health_2022.05",
        "data_selection_query": {
            "match_all": {
                "boost": 1
            }
          },
        "target_index": "transform_collated_txn_health_2022.05",
        "page_size": 1000,
        "groups": [
            {
                "date_histogram": {
                    "fixed_interval": "1m",
                    "source_field": "@timestamp",
                    "target_field": "@timestamp",
                    "timezone": "Asia/Calcutta"
                }
            },
            {
                "terms": {
                    "source_field": "compInstanceIdentifier",
                    "target_field": "compInstanceIdentifier"
                }
            },
            {
                "terms": {
                    "source_field": "agentIdentifier",
                    "target_field": "agentIdentifier"
                }
            }
        ],
        "aggregations": {
            "count_@timestamp": {
                "value_count": {
                    "field": "@timestamp"
                }
            },
            "serviceIdentifiers": {
                "top_hits": {
                    "size": 1,
                    "_source": {
                        "includes": ["serviceIdentifiers"]
                    }
                }
            },
            "clusterIdentifiers": {
                "top_hits": {
                    "size": 1,
                    "_source": {
                        "includes": ["clusterIdentifiers"]
                    }
                }
            }
        }
    }
}'

方案2:使用terms聚合的collect_mode参数(适用于需聚合数组元素但不拆分事件)

如果需要对数组元素进行统计但不想拆分事件,可在terms聚合中设置collect_mode: collect,将数组作为整体分组:

curl -u admin:admin -XPUT "http://XXX.XXX.XX.XXX9201/_plugins/_transform/my-array-job-2" -H 'Content-type: application/json' -d'
{
    "transform": {
        "schedule": {
            "interval": {
                "start_time": 1659705000000,
                "period": 1,
                "unit": "Minutes"
            }
        },
        "metadata_id": null,
        "updated_at": 1659456180000,
        "enabled": true,
        "enabled_at": 1659457620000,
        "description": "",
        "source_index": "collated_txn_health_2022.05",
        "data_selection_query": {
            "match_all": {
                "boost": 1
            }
          },
        "target_index": "transform_collated_txn_health_2022.05",
        "page_size": 1000,
        "groups": [
            {
                "date_histogram": {
                    "fixed_interval": "1m",
                    "source_field": "@timestamp",
                    "target_field": "@timestamp",
                    "timezone": "Asia/Calcutta"
                }
            },
            {
                "terms": {
                    "source_field": "compInstanceIdentifier",
                    "target_field": "compInstanceIdentifier"
                }
            },
            {
                "terms": {
                    "source_field": "agentIdentifier",
                    "target_field": "agentIdentifier"
                }
            }
        ],
        "aggregations": {
            "count_@timestamp": {
                "value_count": {
                    "field": "@timestamp"
                }
            },
            "serviceIdentifiers_terms": {
                "terms": {
                    "field": "serviceIdentifiers",
                    "collect_mode": "collect"
                }
            },
            "clusterIdentifiers_terms": {
                "terms": {
                    "field": "clusterIdentifiers",
                    "collect_mode": "collect"
                }
            }
        }
    }
}'

说明

  • 方案1适合需要完整保留原始数组结构的场景,转换后会在聚合结果中包含原始数组。
  • 方案2适合需要统计数组元素出现频次但不想拆分事件的场景,会返回数组元素的聚合统计结果。

内容的提问来源于stack exchange,提问作者Akshay Kulkarni

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.23 12:27:16