You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Google Apps Script编写的Reddit Feed解析器报错求助

Fixing "TypeError: Cannot read property 'getChildren' of undefined" in Reddit Scraper Script

The Root Cause

That error on line 29 boils down to a misunderstanding of Reddit's XML feed structure. When you parse the Reddit XML response, the root element is already the <feed> tag—so calling root.getChildren("feed") returns an empty array, and trying to access index [0] gives you undefined. Attempting to run getChildren("entry") on that undefined value throws the TypeError you're seeing.

Step-by-Step Fixes

  1. Correct XML Element Traversal
    Replace the line that fetches entries with this, since root is the feed element itself:

    var entries = root.getChildren("entry");
    
  2. Switch to HTTPS for Reddit Requests
    Reddit now enforces HTTPS, so update the URL in scrapReddit() to use https:// instead of http:// to avoid redirect issues or blocked requests:

    var url = "https://www.reddit.com/r/" + REDDIT + "/new.xml?limit=100" + getLastID_();
    
  3. Handle Empty Spreadsheet in getLastID_()
    If your spreadsheet is brand new (empty), sheet.getLastRow() returns 0, and trying to call sheet.getRange(0, col) will throw another error. Add a check to handle this edge case:

    function getLastID_() {
      var ss = SpreadsheetApp.getActiveSpreadsheet();
      var sheet = ss.getSheets()[0];
      var row = sheet.getLastRow();
      // Exit early if sheet is empty to avoid invalid range errors
      if (row === 0) return "";
      var col = sheet.getLastColumn();
      var url = sheet.getRange(row, col).getValue().toString();
      var pattern = /.*comments\/([^\/]*)\.*/;
      var id = url.match(pattern);
      return id ? "&after=t3_" + id[1] : "";
    }
    
  4. Fix Link Extraction
    The original code tried to get the post link via .getText(), but Reddit's XML <link> element stores the URL in the href attribute. Update that line to pull the correct value:

    var link = entries[i].getChild('link').getAttribute('href').getValue();
    

Full Fixed Code

/* Reddit Scraper written by Amit Agarwal (modified to fix XML parsing error) */
var REDDIT = "HomeImprovement";

function run() {
 deleteTriggers_();
 /* Fetch Reddit posts every 5 minutes to avoid hitting the reddit and Google Script quotas */
 ScriptApp.newTrigger("scrapReddit")
 .timeBased().everyMinutes(5).create();
}

function scrapReddit() {
 // Process 20 Reddit posts in a batch
 var url = "https://www.reddit.com/r/" + REDDIT + "/new.xml?limit=100" + getLastID_();
 // Reddit API returns the results in XML format
 var response = UrlFetchApp.fetch(url).getContentText();
 var doc = XmlService.parse(response);
 var root = doc.getRootElement();
 // Fixed: root is already the <feed> element, so directly get <entry> children
 var entries = root.getChildren("entry");
 var data = new Array();
 for (var i=0; i<entries.length; i++) {
 /* Extract post date, title, description and link from Reddit */
 var date = entries[i].getChild('updated').getText();
 var title = entries[i].getChild('title').getText();
 var desc = entries[i].getChild('content').getText();
 // Fixed: Pull link from href attribute instead of text content
 var link = entries[i].getChild('link').getAttribute('href').getValue();
 data[i] = new Array(date, title, desc, link);
 }
 if (data.length == 0) {
 /* There's no data so stop the background trigger */
 deleteTriggers_();
 } else {
 writeData_(data);
 }
}

/* Write the scrapped data in a batch to the Google Spreadsheet since this is more efficient */
function writeData_(data) {
 if (data.length === 0) {
 return;
 }
 var ss = SpreadsheetApp.getActiveSpreadsheet();
 var sheet = ss.getSheets()[0];
 var row = sheet.getLastRow();
 var col = sheet.getLastColumn();
 var range = sheet.getRange(row+1, 1, data.length, 4);
 try {
 range.setValues(data);
 } catch (e) {
 Logger.log(e.toString());
 }
}

/* Use the ID of the last processed post from Reddit as token */
function getLastID_() {
 var ss = SpreadsheetApp.getActiveSpreadsheet();
 var sheet = ss.getSheets()[0];
 var row = sheet.getLastRow();
 // Handle empty spreadsheet case
 if (row === 0) return "";
 var col = sheet.getLastColumn();
 var url = sheet.getRange(row, col).getValue().toString();
 var pattern = /.*comments\/([^\/]*)\.*/;
 var id = url.match(pattern);
 return id ? "&after=t3_" + id[1] : "";
}

/* Posts Extracted, Delete the Triggers */
function deleteTriggers_() {
 var triggers = ScriptApp.getProjectTriggers();
 for (var i=0; i<triggers.length; i++) {
 ScriptApp.deleteTrigger(triggers[i]);
 }
}

内容的提问来源于stack exchange,提问作者Mauro Bros

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.08 22:27:31