You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何通过Logstash Split插件将数组对象放入Elasticsearch文档根层级

问题:如何将Logstash Split拆分后的数组对象直接放入Elasticsearch文档根层级?

我需要使用Logstash的Split过滤器插件,将拆分后的数组对象直接放入Elasticsearch文档的根层级,避免嵌套在"test"键下。以下是相关信息:

输入源(REST API返回数据)

{
  "rid": "dc755b8d-14bf-4211-a820-9aab01e5475b",
  "rval": {
    "totalCount": 3,
    "data": [
      {
        "id": 1,
        "name": "Object1",
        "category": "Object1",
        "sequence": 1
      },
      {
        "id": 2,
        "name": "Object2",
        "sequence": 2
      },
      {
        "id": 3,
        "name": "Obect3",
        "sequence": 3
      }
    ]
  }
}

当前Logstash配置

input {
  http_poller {
    urls => {
      test => {
        method => post
        url => "http://localhost/object"
        body => '{"category": "test"}'
        headers => {
          "Content-Type" => "application/json"
        }
      }
    }
    request_timeout => 60
    schedule => { cron => "* * * * * UTC" }
    codec => "json"
  }
}

filter {
  split {
    field => "[rval][data]"
    target => "test"
  }
  mutate {
    remove_field => [ "rid", "rval", "event", "success", "dateTime", "@version", "@timestamp" ]
  }
}

output {
  elasticsearch {
    hosts => ["localhost:9200"]
    index => "test"
    document_id => "%{[test][id]}"
  }
  stdout {
    codec => rubydebug
  }
}

现有输出

{
  "took": 7,
  "timed_out": false,
  "_shards": {
    "total": 1,
    "successful": 1,
    "skipped": 0,
    "failed": 0
  },
  "hits": {
    "total": {
      "value": 3,
      "relation": "eq"
    },
    "max_score": 1.0,
    "hits": [
      {
        "_index": "test",
        "_id": "1",
        "_score": 1.0,
        "_source": {
          "test": {
            "id": 1,
            "name": "Object1",
            "category": "Object1",
            "sequence": 1
          }
        }
      },
      {
        "_index": "test",
        "_id": "2",
        "_score": 1.0,
        "_source": {
          "test": {
            "id": 2,
            "name": "Object2",
            "category": "Object2",
            "sequence": 2
          }
        }
      },
      {
        "_index": "test",
        "_id": "3",
        "_score": 1.0,
        "_source": {
          "test": {
            "id": 3,
            "name": "Object3",
            "category": "Object3",
            "sequence": 3
          }
        }
      }
    ]
  }
}

预期输出

{
  "took": 7,
  "timed_out": false,
  "_shards": {
    "total": 1,
    "successful": 1,
    "skipped": 0,
    "failed": 0
  },
  "hits": {
    "total": {
      "value": 3,
      "relation": "eq"
    },
    "max_score": 1.0,
    "hits": [
      {
        "_index": "test",
        "_id": "1",
        "_score": 1.0,
        "_source": {
          "id": 1,
          "name": "Object1",
          "category": "Object1",
          "sequence": 1
        }
      },
      {
        "_index": "test",
        "_id": "2",
        "_score": 1.0,
        "_source": {
          "id": 2,
          "name": "Object2",
          "category": "Object2",
          "sequence": 2
        }
      },
      {
        "_index": "test",
        "_id": "3",
        "_score": 1.0,
        "_source": {
          "id": 3,
          "name": "Object3",
          "category": "Object3",
          "sequence": 3
        }
      }
    ]
  }
}

遇到的问题

  • 当前输出中拆分后的对象嵌套在test键下,不符合预期
  • 若移除split配置中的target => "test",拆分后rval.data的完整数组会保留在文档中,无法得到独立的对象文档

解决方案

修改Logstash的filter段,在split之后使用ruby过滤器将test字段的内容直接替换整个event,同时调整output中的document_id引用路径:

input {
  http_poller {
    urls => {
      test => {
        method => post
        url => "http://localhost/object"
        body => '{"category": "test"}'
        headers => {
          "Content-Type" => "application/json"
        }
      }
    }
    request_timeout => 60
    schedule => { cron => "* * * * * UTC" }
    codec => "json"
  }
}

filter {
  split {
    field => "[rval][data]"
    target => "test"
  }
  # 将test字段的内容替换整个event,直接得到根层级的对象
  ruby {
    code => "event.replace(event.get('test'))"
  }
  # 移除不需要的元数据字段
  mutate {
    remove_field => [ "@version", "@timestamp" ]
  }
}

output {
  elasticsearch {
    hosts => ["localhost:9200"]
    index => "test"
    # 现在直接引用根层级的id字段
    document_id => "%{id}"
  }
  stdout {
    codec => rubydebug
  }
}

说明

  1. split过滤器将rval.data数组拆分为单个对象并存入test字段
  2. ruby过滤器通过event.replace(event.get('test'))将整个事件替换为拆分后的对象,直接实现根层级结构
  3. 调整document_id为%{id},因为此时id已在根层级
  4. mutate remove_field仅清除不需要的元数据字段,原有rid、rval等字段已被event.replace清除

内容的提问来源于stack exchange,提问作者Archanfel

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.31 20:25:17