You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在Node.js中解析指定URL的XML并提取IdList中的所有ID?

How to Extract PubMed IDs from NCBI ESearch XML Response

Hey there! Let's work through pulling those IdList entries from the NCBI ESearch XML response—sorry to hear node2xml didn't pan out for you. Chances are you might have mixed up the library's purpose (node2xml is typically used to convert JavaScript objects to XML, not parse XML from responses). Here are three reliable approaches to try in Node.js:

1. Use xml2js (XML-to-JavaScript Parser)

This is one of the most popular libraries for turning XML into workable JS objects.

First, install the required packages:

npm install xml2js axios

Then use this code to fetch and parse the XML:

const axios = require('axios');
const xml2js = require('xml2js');

async function fetchPubmedIds() {
  const ncbiUrl = 'https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi?db=pubmed&term=science[journal]+AND+breast+cancer+AND+2008[pdat]';
  
  try {
    // Fetch the XML data
    const response = await axios.get(ncbiUrl);
    const xmlContent = response.data;

    // Parse XML to JS object (explicitArray ensures repeated nodes are arrays)
    const parser = new xml2js.Parser({ explicitArray: true });
    const parsedResult = await parser.parseStringPromise(xmlContent);

    // Extract IDs from the IdList
    const pubmedIds = parsedResult.eSearchResult.IdList.Id;
    console.log('Extracted PubMed IDs:', pubmedIds);
  } catch (error) {
    console.error('Error during fetch/parsing:', error.message);
  }
}

// Run the function
fetchPubmedIds();

2. Use cheerio (jQuery-style XML Manipulation)

If you're comfortable with jQuery syntax, cheerio makes traversing and extracting data from XML a breeze.

Install dependencies first:

npm install cheerio axios

Sample code:

const axios = require('axios');
const cheerio = require('cheerio');

async function fetchIdsWithCheerio() {
  const ncbiUrl = 'https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi?db=pubmed&term=science[journal]+AND+breast+cancer+AND+2008[pdat]';
  
  try {
    const response = await axios.get(ncbiUrl);
    // Load XML in cheerio with xmlMode enabled
    const $ = cheerio.load(response.data, { xmlMode: true });

    // Extract text from all <Id> nodes under <IdList>
    const pubmedIds = $('IdList > Id').map((_, element) => $(element).text()).get();
    console.log('Extracted PubMed IDs:', pubmedIds);
  } catch (error) {
    console.error('Error:', error.message);
  }
}

fetchIdsWithCheerio();

3. Use DOMParser (Browser or Node.js with jsdom)

For browser environments, you can use the native DOMParser. For Node.js, you'll need jsdom to replicate browser-like DOM functionality.

Browser Environment Example:

async function fetchIdsInBrowser() {
  const ncbiUrl = 'https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi?db=pubmed&term=science[journal]+AND+breast+cancer+AND+2008[pdat]';
  
  try {
    const response = await fetch(ncbiUrl);
    const xmlText = await response.text();
    const parser = new DOMParser();
    const xmlDoc = parser.parseFromString(xmlText, 'text/xml');

    // Get all <Id> elements and extract their text
    const idElements = xmlDoc.querySelectorAll('IdList Id');
    const pubmedIds = Array.from(idElements).map(el => el.textContent.trim());
    console.log(pubmedIds);
  } catch (err) {
    console.error('Error:', err.message);
  }
}

fetchIdsInBrowser();

Node.js with jsdom:

First install jsdom and axios:

npm install jsdom axios

Then code:

const axios = require('axios');
const { JSDOM } = require('jsdom');

async function fetchIdsWithJSDOM() {
  const ncbiUrl = 'https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi?db=pubmed&term=science[journal]+AND+breast+cancer+AND+2008[pdat]';
  
  try {
    const response = await axios.get(ncbiUrl);
    // Initialize JSDOM with XML content type
    const dom = new JSDOM(response.data, { contentType: 'text/xml' });
    const idElements = dom.window.document.querySelectorAll('IdList Id');
    const pubmedIds = Array.from(idElements).map(el => el.textContent.trim());
    console.log('Extracted PubMed IDs:', pubmedIds);
  } catch (err) {
    console.error('Error:', err.message);
  }
}

fetchIdsWithJSDOM();

内容的提问来源于stack exchange,提问作者Mathie

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.21 06:38:26