You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

请求协助编写Python程序,批量查询DBpedia与Wikidata人物信息

Python批量查询DBpedia与Wikidata人物信息方案

嘿,我来帮你搞定这个批量查询人物信息的需求!下面是一个实用的Python实现方案,能针对你的姓名列表,分别从DBpedia和Wikidata获取结构化的人物数据,包含常见的核心信息比如出生日期、职业、简介等。

所需依赖

首先确保你安装了requests库(用来发送HTTP请求),执行下面的命令安装:

pip install requests

完整代码实现

import requests
import json

def fetch_dbpedia_info(person_name):
    """从DBpedia获取人物信息"""
    sparql_query = f"""
    SELECT ?person ?birthDate ?occupation ?abstract
    WHERE {{
        ?person rdf:type dbo:Person ;
                foaf:name "{person_name}"@en ;
                dbo:birthDate ?birthDate ;
                dbo:occupation ?occupation ;
                dbo:abstract ?abstract .
        FILTER(lang(?abstract) = 'en')
    }}
    LIMIT 1
    """
    url = "http://dbpedia.org/sparql"
    params = {
        "query": sparql_query,
        "format": "json"
    }
    try:
        response = requests.get(url, params=params)
        response.raise_for_status()
        data = response.json()
        if data["results"]["bindings"]:
            result = data["results"]["bindings"][0]
            return {
                "source": "DBpedia",
                "name": person_name,
                "birth_date": result.get("birthDate", {}).get("value"),
                "occupation": result.get("occupation", {}).get("value"),
                "abstract": result.get("abstract", {}).get("value")
            }
        else:
            return {"source": "DBpedia", "name": person_name, "message": "未找到匹配的人物信息"}
    except Exception as e:
        return {"source": "DBpedia", "name": person_name, "error": str(e)}

def fetch_wikidata_info(person_name):
    """从Wikidata获取人物信息"""
    sparql_query = f"""
    SELECT ?person ?birthDate ?occupation ?description
    WHERE {{
        ?person wdt:P31 wd:Q5 ; # 实体是人类
                rdfs:label "{person_name}"@en ;
                wdt:P569 ?birthDate . # 出生日期
        OPTIONAL {{ ?person wdt:P106 ?occupation . }} # 职业
        OPTIONAL {{ ?person schema:description ?description FILTER(lang(?description) = 'en') }}
    }}
    LIMIT 1
    """
    url = "https://query.wikidata.org/sparql"
    params = {
        "query": sparql_query,
        "format": "json"
    }
    try:
        response = requests.get(url, params=params)
        response.raise_for_status()
        data = response.json()
        if data["results"]["bindings"]:
            result = data["results"]["bindings"][0]
            occupation_label = None
            if "occupation" in result:
                # 额外请求获取职业的可读标签
                occ_uri = result["occupation"]["value"]
                occ_id = occ_uri.split("/")[-1]
                occ_label_query = f"""
                SELECT ?label WHERE {{ wd:{occ_id} rdfs:label ?label FILTER(lang(?label) = 'en') }} LIMIT 1
                """
                occ_response = requests.get(url, params={"query": occ_label_query, "format": "json"})
                occ_data = occ_response.json()
                if occ_data["results"]["bindings"]:
                    occupation_label = occ_data["results"]["bindings"][0]["label"]["value"]
            return {
                "source": "Wikidata",
                "name": person_name,
                "birth_date": result.get("birthDate", {}).get("value"),
                "occupation": occupation_label or result.get("occupation", {}).get("value"),
                "description": result.get("description", {}).get("value")
            }
        else:
            return {"source": "Wikidata", "name": person_name, "message": "未找到匹配的人物信息"}
    except Exception as e:
        return {"source": "Wikidata", "name": person_name, "error": str(e)}

def batch_fetch_person_info(person_list):
    """批量处理人物列表,返回所有信息"""
    all_results = []
    for name in person_list:
        dbpedia_result = fetch_dbpedia_info(name)
        wikidata_result = fetch_wikidata_info(name)
        all_results.append({
            "name": name,
            "dbpedia": dbpedia_result,
            "wikidata": wikidata_result
        })
    return all_results

# 使用示例
if __name__ == "__main__":
    people = ["Albert Einstein", "Marie Curie", "Leonardo da Vinci"]
    results = batch_fetch_person_info(people)
    for res in results:
        print(f"\n=== {res['name']} 的查询结果 ===")
        print("DBpedia信息:")
        print(json.dumps(res['dbpedia'], indent=2, ensure_ascii=False))
        print("\nWikidata信息:")
        print(json.dumps(res['wikidata'], indent=2, ensure_ascii=False))

代码说明

  • fetch_dbpedia_info函数:通过DBpedia的SPARQL接口查询人物核心信息,包括出生日期、职业和英文简介。如果没有匹配结果会返回提示,请求出错则返回错误详情。
  • fetch_wikidata_info函数:调用Wikidata的SPARQL接口,特别处理了职业字段——因为Wikidata返回的是URI,所以额外发起请求获取可读的职业标签,让结果更直观。
  • batch_fetch_person_info函数:遍历你的姓名列表,逐个调用上面两个查询函数,把结果整合后返回。

自定义提示

  • 如果需要获取更多字段(比如出生地、教育背景),你可以修改SPARQL查询语句,添加对应的属性(DBpedia属性前缀是dbo:,Wikidata属性是wdt:,比如出生地在DBpedia是dbo:birthPlace,Wikidata是wdt:P19)。
  • 注意姓名拼写要准确(优先用英文姓名,因为DBpedia和Wikidata的英文数据更完善),如果是中文姓名,可以调整查询语句中的语言标签(把@en改成@zh)。

内容的提问来源于stack exchange,提问作者Daniel

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.19 09:58:38