如何在Node.js中解析指定URL的XML并提取IdList中的所有ID?
Hey there! Let's work through pulling those IdList entries from the NCBI ESearch XML response—sorry to hear node2xml didn't pan out for you. Chances are you might have mixed up the library's purpose (node2xml is typically used to convert JavaScript objects to XML, not parse XML from responses). Here are three reliable approaches to try in Node.js:
1. Use xml2js (XML-to-JavaScript Parser)
This is one of the most popular libraries for turning XML into workable JS objects.
First, install the required packages:
npm install xml2js axios
Then use this code to fetch and parse the XML:
const axios = require('axios'); const xml2js = require('xml2js'); async function fetchPubmedIds() { const ncbiUrl = 'https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi?db=pubmed&term=science[journal]+AND+breast+cancer+AND+2008[pdat]'; try { // Fetch the XML data const response = await axios.get(ncbiUrl); const xmlContent = response.data; // Parse XML to JS object (explicitArray ensures repeated nodes are arrays) const parser = new xml2js.Parser({ explicitArray: true }); const parsedResult = await parser.parseStringPromise(xmlContent); // Extract IDs from the IdList const pubmedIds = parsedResult.eSearchResult.IdList.Id; console.log('Extracted PubMed IDs:', pubmedIds); } catch (error) { console.error('Error during fetch/parsing:', error.message); } } // Run the function fetchPubmedIds();
2. Use cheerio (jQuery-style XML Manipulation)
If you're comfortable with jQuery syntax, cheerio makes traversing and extracting data from XML a breeze.
Install dependencies first:
npm install cheerio axios
Sample code:
const axios = require('axios'); const cheerio = require('cheerio'); async function fetchIdsWithCheerio() { const ncbiUrl = 'https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi?db=pubmed&term=science[journal]+AND+breast+cancer+AND+2008[pdat]'; try { const response = await axios.get(ncbiUrl); // Load XML in cheerio with xmlMode enabled const $ = cheerio.load(response.data, { xmlMode: true }); // Extract text from all <Id> nodes under <IdList> const pubmedIds = $('IdList > Id').map((_, element) => $(element).text()).get(); console.log('Extracted PubMed IDs:', pubmedIds); } catch (error) { console.error('Error:', error.message); } } fetchIdsWithCheerio();
3. Use DOMParser (Browser or Node.js with jsdom)
For browser environments, you can use the native DOMParser. For Node.js, you'll need jsdom to replicate browser-like DOM functionality.
Browser Environment Example:
async function fetchIdsInBrowser() { const ncbiUrl = 'https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi?db=pubmed&term=science[journal]+AND+breast+cancer+AND+2008[pdat]'; try { const response = await fetch(ncbiUrl); const xmlText = await response.text(); const parser = new DOMParser(); const xmlDoc = parser.parseFromString(xmlText, 'text/xml'); // Get all <Id> elements and extract their text const idElements = xmlDoc.querySelectorAll('IdList Id'); const pubmedIds = Array.from(idElements).map(el => el.textContent.trim()); console.log(pubmedIds); } catch (err) { console.error('Error:', err.message); } } fetchIdsInBrowser();
Node.js with jsdom:
First install jsdom and axios:
npm install jsdom axios
Then code:
const axios = require('axios'); const { JSDOM } = require('jsdom'); async function fetchIdsWithJSDOM() { const ncbiUrl = 'https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi?db=pubmed&term=science[journal]+AND+breast+cancer+AND+2008[pdat]'; try { const response = await axios.get(ncbiUrl); // Initialize JSDOM with XML content type const dom = new JSDOM(response.data, { contentType: 'text/xml' }); const idElements = dom.window.document.querySelectorAll('IdList Id'); const pubmedIds = Array.from(idElements).map(el => el.textContent.trim()); console.log('Extracted PubMed IDs:', pubmedIds); } catch (err) { console.error('Error:', err.message); } } fetchIdsWithJSDOM();
内容的提问来源于stack exchange,提问作者Mathie

