基于rentrez提取PubMed XML文献的作者及关联信息技术问询
问题描述
已参考相关代码实现PubMed文献收稿、录用、发表日期及间隔提取,现需获取以下学术指标:
- 作者数量
- 第一作者所属机构
- 最后作者所属机构
- 单篇文献引用数
- 第一作者学术度
同时希望了解PubMed XML可提取的完整字段范围。当前已实现首尾作者姓名提取,但无法拆分出首尾作者对应的机构,可复现代码如下:
#load in packages library(reprex) library(devtools) #> Loading required package: usethis install_github("ropensci/rentrez") #> Skipping install of 'rentrez' from a github remote, the SHA1 (a225f213) has not changed since last install. #> Use `force = TRUE` to force installation library(rentrez) require(XML) #> Loading required package: XML require(ggplot2) #> Loading required package: ggplot2 require(ggridges) #> Loading required package: ggridges require(gridExtra) #> Loading required package: gridExtra # search pubmed using a search term (use_history allows retrieval of all records) pp <- entrez_search(db="pubmed", term="cell[ta] AND 2010 : 2021[pdat] AND (journal article[pt] NOT review[pt] NOT comment[pt] NOT autobiography[pt] NOT biography[pt] NOT case reports[pt] NOT clinical trial[pt] NOT historical article[pt] NOT comparative study[pt] NOT evaluation study[pt] NOT evaluation study[pt] NOT introductory journal article[pt])", use_history = TRUE) pp_rec <- entrez_fetch(db="pubmed", web_history=pp$web_history, rettype="xml", parsed=TRUE) # save records as XML file saveXML(pp_rec, file = "Data/records.xml") #> Error in saveXML(pp_rec, file = "Data/records.xml"): cannot create file Data/records.xml. Check the directory exists and permissions are appropriate filename <- "~/Data/records.xml" ## extract a data frame from XML file ## This is modified from christopherBelter's pubmedXML R code extract_xml <- function(theFile) { library(XML) newData <- xmlParse(theFile) records <- getNodeSet(newData, "//PubmedArticle") pmid <- xpathSApply(newData,"//MedlineCitation/PMID", xmlValue) doi <- lapply(records, xpathSApply, ".//ELocationID[@EIdType = \"doi\"]", xmlValue) doi[sapply(doi, is.list)] <- NA doi <- unlist(doi) authLast <- lapply(records, xpathSApply, ".//Author/LastName", xmlValue) authLast[sapply(authLast, is.list)] <- NA authInit <- lapply(records, xpathSApply, ".//Author/Initials", xmlValue) authInit[sapply(authInit, is.list)] <- NA authors <- mapply(paste, authLast, authInit, collapse = "|") authAffil <- lapply(records, xpathSApply, ".//Author/AffiliationInfo", xmlValue) authAffil[sapply(authAffil, is.list)] <- NA authAffil <- sapply(authAffil, paste, collapse = "|") theDF <- data.frame(pmid, doi, authors,authAffil, stringsAsFactors = FALSE) return(theDF) } #extract into a dataframe theData <- extract_xml(filename) #show the author affiliations as bunched print(theData$authAffil[1]) #> [1] "Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA. Electronic address: kjsiddle@broadinstitute.org.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA; Department of Organismic and Evolutionary Biology, Harvard University, Cambridge, MA 02138, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA; Department of Organismic and Evolutionary Biology, Harvard University, Cambridge, MA 02138, USA; Department of Immunology and Infectious Diseases, Harvard T.H. Chan School of Public Health, Harvard University, Boston, MA 02115, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA; Division of Infectious Diseases, Massachusetts General Hospital, Boston, MA 02114, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA; Faculty of Arts and Sciences, Harvard University, Cambridge, MA 02138, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA; Department of Organismic and Evolutionary Biology, Harvard University, Cambridge, MA 02138, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA; Department of Systems Biology, Harvard Medical School, Boston, MA 02115, USA.|Department of Epidemiology, Harvard T.H. Chan School of Public Health, Harvard University, Boston, MA 02115, USA; Center for Communicable Disease Dynamics, Department of Epidemiology, Harvard T. H. Chan School of Public Health, Harvard University, Boston, MA 02115, USA.|Department of Epidemiology, Harvard T.H. Chan School of Public Health, Harvard University, Boston, MA 02115, USA; Center for Communicable Disease Dynamics, Department of Epidemiology, Harvard T. H. Chan School of Public Health, Harvard University, Boston, MA 02115, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA; Applied Epidemiology Fellowship, Council of State and Territorial Epidemiologists, Atlanta, GA 30345, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Barnstable County Department of Health and the Environment, Barnstable, MA 02630, USA.|Barnstable County Department of Health and the Environment, Barnstable, MA 02630, USA.|Barnstable County Department of Health and the Environment, Barnstable, MA 02630, USA.|Barnstable County Department of Human Services, Barnstable, MA 02630, USA.|Community Tracing Collaborative, Commonwealth of Massachusetts, Boston, MA 02199, USA.|Community Tracing Collaborative, Commonwealth of Massachusetts, Boston, MA 02199, USA.|Community Tracing Collaborative, Commonwealth of Massachusetts, Boston, MA 02199, USA.|Department of Epidemiology, Harvard T.H. Chan School of Public Health, Harvard University, Boston, MA 02115, USA; Center for Communicable Disease Dynamics, Department of Epidemiology, Harvard T. H. Chan School of Public Health, Harvard University, Boston, MA 02115, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Massachusetts Department of Public Health, Boston, MA 02199, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA; Department of Immunology and Infectious Diseases, Harvard T.H. Chan School of Public Health, Harvard University, Boston, MA 02115, USA; Massachusetts Consortium for Pathogen Readiness, Boston, MA 02115, USA. Electronic address: bronwyn@broadinstitute.org.|Broad Institute of Harvard and MIT, Cambridge, MA 02142, USA; Department of Organismic and Evolutionary Biology, Harvard University, Cambridge, MA 02138, USA; Department of Immunology and Infectious Diseases, Harvard T.H. Chan School of Public Health, Harvard University, Boston, MA 02115, USA; Howard Hughes Medical Institute, Chevy Chase, MD 20815, USA; Massachusetts Consortium for Pathogen Readiness, Boston, MA 02115, USA."
解决方案
1. 提取首尾作者对应机构
核心思路是针对每个Author节点单独提取机构,而非直接抓取所有机构的集合。修改extract_xml函数,新增首尾机构提取逻辑:
extract_xml <- function(theFile) { library(XML) newData <- xmlParse(theFile) records <- getNodeSet(newData, "//PubmedArticle") # 初始化存储向量 pmid <- c() doi <- c() authors <- c() author_count <- c() first_affil <- c() last_affil <- c() authAffil <- c() for (rec in records) { # 提取PMID pmid_val <- xpathSApply(rec, ".//MedlineCitation/PMID", xmlValue) pmid <- c(pmid, ifelse(length(pmid_val)==0, NA, pmid_val)) # 提取DOI doi_val <- xpathSApply(rec, ".//ELocationID[@EIdType = \"doi\"]", xmlValue) doi <- c(doi, ifelse(length(doi_val)==0, NA, doi_val)) # 提取作者信息 author_nodes <- getNodeSet(rec, ".//Author") auth_last <- sapply(author_nodes, xpathSApply, ".//LastName", xmlValue) auth_init <- sapply(author_nodes, xpathSApply, ".//Initials", xmlValue) auth_full <- mapply(paste, auth_last, auth_init, SIMPLIFY = FALSE) authors <- c(authors, paste(unlist(auth_full), collapse = "|")) # 作者数量 author_count <- c(author_count, length(author_nodes)) # 提取所有机构 all_affils <- sapply(author_nodes, xpathSApply, ".//AffiliationInfo/Affiliation", xmlValue) all_affils <- lapply(all_affils, function(x) ifelse(length(x)==0, NA, paste(x, collapse = "; "))) authAffil <- c(authAffil, paste(unlist(all_affils), collapse = "|")) # 第一作者机构 first_affil_val <- ifelse(length(author_nodes)>=1, all_affils[[1]], NA) first_affil <- c(first_affil, first_affil_val) # 最后作者机构 last_affil_val <- ifelse(length(author_nodes)>=1, all_affils[[length(author_nodes)]], NA) last_affil <- c(last_affil, last_affil_val) } theDF <- data.frame(pmid, doi, authors, author_count, first_affil, last_affil, authAffil, stringsAsFactors = FALSE) return(theDF) }
修改后调用函数,即可得到包含author_count(作者数量)、first_affil(第一作者机构)、last_affil(最后作者机构)的数据集。
2. 其他目标指标获取方向
单篇文献引用数
PubMed XML本身不包含引用数据,需通过外部API获取:
- 使用
rcrossref包,通过DOI查询CrossRef数据库的引用计数 - 调用Web of Science/Scopus的API,通过PMID/DOI查询引用数据(需申请API密钥)
第一作者学术度
无统一提取字段,需结合外部数据源:
- 通过作者姓名+机构匹配ORCID,再调用ORCID API获取作者发表记录
- 使用Scopus/Web of Science API,查询作者的h指数、总引用数等指标
3. PubMed XML可提取的完整字段范围
PubMed的Medline XML包含以下核心字段:
- 文献标识:PMID、DOI、PMCID
- 基本信息:标题、英文摘要、中文摘要(若有)
- 作者信息:所有作者姓名、机构、ORCID、邮箱
- 出版信息:期刊名、ISSN、卷期页、收稿日期、录用日期、在线发表日期、印刷发表日期
- 主题标注:MeSH术语、关键词、化学物质标识
- 研究信息:研究类型、资助机构、伦理声明、临床试验注册号
- 关联文献:参考文献列表、相关PMID
内容的提问来源于stack exchange,提问作者mickmars51
相关产品推荐
相关产品推荐

