请求协助编写Python程序,批量查询DBpedia与Wikidata人物信息
Python批量查询DBpedia与Wikidata人物信息方案
嘿,我来帮你搞定这个批量查询人物信息的需求!下面是一个实用的Python实现方案,能针对你的姓名列表,分别从DBpedia和Wikidata获取结构化的人物数据,包含常见的核心信息比如出生日期、职业、简介等。
所需依赖
首先确保你安装了requests库(用来发送HTTP请求),执行下面的命令安装:
pip install requests
完整代码实现
import requests import json def fetch_dbpedia_info(person_name): """从DBpedia获取人物信息""" sparql_query = f""" SELECT ?person ?birthDate ?occupation ?abstract WHERE {{ ?person rdf:type dbo:Person ; foaf:name "{person_name}"@en ; dbo:birthDate ?birthDate ; dbo:occupation ?occupation ; dbo:abstract ?abstract . FILTER(lang(?abstract) = 'en') }} LIMIT 1 """ url = "http://dbpedia.org/sparql" params = { "query": sparql_query, "format": "json" } try: response = requests.get(url, params=params) response.raise_for_status() data = response.json() if data["results"]["bindings"]: result = data["results"]["bindings"][0] return { "source": "DBpedia", "name": person_name, "birth_date": result.get("birthDate", {}).get("value"), "occupation": result.get("occupation", {}).get("value"), "abstract": result.get("abstract", {}).get("value") } else: return {"source": "DBpedia", "name": person_name, "message": "未找到匹配的人物信息"} except Exception as e: return {"source": "DBpedia", "name": person_name, "error": str(e)} def fetch_wikidata_info(person_name): """从Wikidata获取人物信息""" sparql_query = f""" SELECT ?person ?birthDate ?occupation ?description WHERE {{ ?person wdt:P31 wd:Q5 ; # 实体是人类 rdfs:label "{person_name}"@en ; wdt:P569 ?birthDate . # 出生日期 OPTIONAL {{ ?person wdt:P106 ?occupation . }} # 职业 OPTIONAL {{ ?person schema:description ?description FILTER(lang(?description) = 'en') }} }} LIMIT 1 """ url = "https://query.wikidata.org/sparql" params = { "query": sparql_query, "format": "json" } try: response = requests.get(url, params=params) response.raise_for_status() data = response.json() if data["results"]["bindings"]: result = data["results"]["bindings"][0] occupation_label = None if "occupation" in result: # 额外请求获取职业的可读标签 occ_uri = result["occupation"]["value"] occ_id = occ_uri.split("/")[-1] occ_label_query = f""" SELECT ?label WHERE {{ wd:{occ_id} rdfs:label ?label FILTER(lang(?label) = 'en') }} LIMIT 1 """ occ_response = requests.get(url, params={"query": occ_label_query, "format": "json"}) occ_data = occ_response.json() if occ_data["results"]["bindings"]: occupation_label = occ_data["results"]["bindings"][0]["label"]["value"] return { "source": "Wikidata", "name": person_name, "birth_date": result.get("birthDate", {}).get("value"), "occupation": occupation_label or result.get("occupation", {}).get("value"), "description": result.get("description", {}).get("value") } else: return {"source": "Wikidata", "name": person_name, "message": "未找到匹配的人物信息"} except Exception as e: return {"source": "Wikidata", "name": person_name, "error": str(e)} def batch_fetch_person_info(person_list): """批量处理人物列表,返回所有信息""" all_results = [] for name in person_list: dbpedia_result = fetch_dbpedia_info(name) wikidata_result = fetch_wikidata_info(name) all_results.append({ "name": name, "dbpedia": dbpedia_result, "wikidata": wikidata_result }) return all_results # 使用示例 if __name__ == "__main__": people = ["Albert Einstein", "Marie Curie", "Leonardo da Vinci"] results = batch_fetch_person_info(people) for res in results: print(f"\n=== {res['name']} 的查询结果 ===") print("DBpedia信息:") print(json.dumps(res['dbpedia'], indent=2, ensure_ascii=False)) print("\nWikidata信息:") print(json.dumps(res['wikidata'], indent=2, ensure_ascii=False))
代码说明
fetch_dbpedia_info函数:通过DBpedia的SPARQL接口查询人物核心信息,包括出生日期、职业和英文简介。如果没有匹配结果会返回提示,请求出错则返回错误详情。fetch_wikidata_info函数:调用Wikidata的SPARQL接口,特别处理了职业字段——因为Wikidata返回的是URI,所以额外发起请求获取可读的职业标签,让结果更直观。batch_fetch_person_info函数:遍历你的姓名列表,逐个调用上面两个查询函数,把结果整合后返回。
自定义提示
- 如果需要获取更多字段(比如出生地、教育背景),你可以修改SPARQL查询语句,添加对应的属性(DBpedia属性前缀是
dbo:,Wikidata属性是wdt:,比如出生地在DBpedia是dbo:birthPlace,Wikidata是wdt:P19)。 - 注意姓名拼写要准确(优先用英文姓名,因为DBpedia和Wikidata的英文数据更完善),如果是中文姓名,可以调整查询语句中的语言标签(把
@en改成@zh)。
内容的提问来源于stack exchange,提问作者Daniel
相关产品推荐
相关产品推荐

