You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Elasticsearch查询结果解析报错:string indices must be integers排查

解决Elasticsearch查询结果字段提取报错问题

问题描述

需求

从Elasticsearch返回的日志中筛选_index和@timestamp字段,本地模拟JSON数据时,以下代码可正常获取目标值:

index = (data['hits']['hits'][0]['_index'])
timestamp = (data['hits']['hits'][0]['_source']['@timestamp'])

输出结果:

winlogbeat-dc*
2022-11-16T04:19:13.622Z

报错情况

直接调用Elasticsearch服务器结果时,抛出错误:

Traceback (most recent call last):
  File "c:\Users\user\Desktop\PYTHON\tiny2.py", line 96, in <module>
    query()
  File "c:\Users\user\Desktop\PYTHON\tiny2.py", line 77, in query
    index = (final_data['hits']['hits'][0]['_index'])
TypeError: string indices must be integers

尝试调整后又出现新报错:

raise TypeError(f'the JSON object must be str, bytes or bytearray, '
TypeError: the JSON object must be str, bytes or bytearray, not ObjectApiResponse

完整代码

import os
import ast
import csv
import json
from elasticsearch import Elasticsearch
from datetime import datetime,timedelta
import datetime

ELASTIC_USERNAME = 'elastic'
ELASTIC_PASSWORD = "abc123"
PORT= str('9200')
HOST = str('10.20.20.131')
CERT = os.path.join(os.path.dirname(__file__),"cert.crt")

initial_time = datetime.datetime.now()
past_time = datetime.datetime.now() - (timedelta(minutes=15))

def query():
    try: #connection to Elastic server
        es = Elasticsearch(
            "https://10.20.20.131:9200",
            ca_certs = CERT,
            verify_certs=False,
            basic_auth = (ELASTIC_USERNAME, ELASTIC_PASSWORD)
        )
    except ConnectionRefusedError as error:
        print("[-] Connection error")
    else: #DSL Elastic query of Domain Controler logs
        query_res = es.search(
            index="winlogbeat-dc*",
            body={
                "size": 3,
                "sort": [
                    {
                        "timestamp": {
                            "order": "desc",
                            "unmapped_type": "boolean"
                        }
                    }
                ],
                "_source": [
                    "agent.hostname",
                    "@timestamp"
                ],
                "query": {
                    "bool": {
                    "must": [],
                    "filter": [
                        {
                        "range": {
                            "@timestamp": {
                            "format": "strict_date_optional_time",
                            "gte": f'{initial_time}',
                            "lte": f'{past_time}'
                            }
                        }
                        }
                    ],
                    "should": [],
                    "must_not": []
                    }
                }
                }
            )
    
    if query_res:
        parse_to_json =json.loads(query_res)
        final_data = json.dumps(str(parse_to_json))
   
        index = ast.literal_eval(final_data)['hits']['hits'][0]['_index']
        timestamp = ast.literal_eval(final_data)['hits']['hits'][0]['_source']['@timestamp']

        columns = ['Index','Last Updated']
        rows = [[f'{index}',f'{timestamp}']]

        with open("final_data.csv", 'w') as csv_file:
            write_to_csv = csv.writer(csv_file)
            write_to_csv.writerow(columns)
            write_to_csv.writerows(rows)
            print("CSV file created!")

    else:
        print("Log not found")
query()

问题根源与修复步骤

1. 核心错误:对Elasticsearch返回对象的误解

es.search()返回的不是JSON字符串,而是ObjectApiResponse对象,它本身就是可直接访问的字典结构,不需要用json.loads()解析。之前的代码错误地把它当作字符串处理,导致后续操作全部出错。

2. 关键修复点

(1)移除多余的JSON解析/序列化操作

将错误的解析代码:

parse_to_json =json.loads(query_res)
final_data = json.dumps(str(parse_to_json))

index = ast.literal_eval(final_data)['hits']['hits'][0]['_index']
timestamp = ast.literal_eval(final_data)['hits']['hits'][0]['_source']['@timestamp']

替换为直接访问字典属性:

index = query_res['hits']['hits'][0]['_index']
timestamp = query_res['hits']['hits'][0]['_source']['@timestamp']

(2)修复时间范围查询的格式与逻辑

  • 直接将datetime对象转为字符串不符合Elasticsearch的日期格式要求,需生成ISO格式字符串:
    past_time = (datetime.now() - timedelta(minutes=15)).isoformat()
    initial_time = datetime.now().isoformat()
    
  • 时间范围逻辑写反:gte应对应过去时间,lte对应当前时间,否则无法查询到数据:
    "range": {
        "@timestamp": {
            "format": "strict_date_optional_time",
            "gte": past_time,
            "lte": initial_time
        }
    }
    

(3)修正排序字段

排序字段使用了"timestamp",但实际数据中的字段是"@timestamp",需调整为:

"sort": [
    {
        "@timestamp": {
            "order": "desc",
            "unmapped_type": "boolean"
        }
    }
]

3. 完整修复后的代码

import os
import csv
from elasticsearch import Elasticsearch
from datetime import datetime,timedelta

ELASTIC_USERNAME = 'elastic'
ELASTIC_PASSWORD = "abc123"
HOST = '10.20.20.131'
CERT = os.path.join(os.path.dirname(__file__),"cert.crt")

# 生成符合ES要求的ISO格式时间字符串
past_time = (datetime.now() - timedelta(minutes=15)).isoformat()
initial_time = datetime.now().isoformat()

def query():
    try:
        es = Elasticsearch(
            f"https://{HOST}:9200",
            ca_certs=CERT,
            verify_certs=False,
            basic_auth=(ELASTIC_USERNAME, ELASTIC_PASSWORD)
        )
        # 验证连接是否成功
        if not es.ping():
            print("[-] 无法连接到Elasticsearch服务器")
            return
    except Exception as error:
        print(f"[-] 连接错误: {error}")
        return
    else:
        query_res = es.search(
            index="winlogbeat-dc*",
            body={
                "size": 3,
                "sort": [
                    {
                        "@timestamp": {
                            "order": "desc",
                            "unmapped_type": "boolean"
                        }
                    }
                ],
                "_source": [
                    "agent.hostname",
                    "@timestamp"
                ],
                "query": {
                    "bool": {
                        "must": [],
                        "filter": [
                            {
                                "range": {
                                    "@timestamp": {
                                        "format": "strict_date_optional_time",
                                        "gte": past_time,
                                        "lte": initial_time
                                    }
                                }
                            }
                        ],
                        "should": [],
                        "must_not": []
                    }
                }
            }
        )
    
    if query_res and query_res['hits']['hits']:
        # 直接从返回的字典中提取字段
        index = query_res['hits']['hits'][0]['_index']
        timestamp = query_res['hits']['hits'][0]['_source']['@timestamp']

        columns = ['Index','Last Updated']
        rows = [[index, timestamp]]

        with open("final_data.csv", 'w', newline='') as csv_file:
            write_to_csv = csv.writer(csv_file)
            write_to_csv.writerow(columns)
            write_to_csv.writerows(rows)
            print("CSV文件已创建!")
    else:
        print("未找到日志")

query()

内容的提问来源于stack exchange,提问作者PythonNoob

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.08 02:25:24