You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Go语言实现无限滚动页面滚到底部并获取HTML内容求助

解决Go语言中无限滚动页面的元素加载与内容提取问题

针对无限滚动页面的爬取,核心思路是模拟滚动行为,直到页面高度不再变化(即没有新内容加载),再提取所需内容。下面给出基于chromedp和rod这两个Go语言主流浏览器自动化库的具体实现方案,都是新手友好的写法:

方案一:使用chromedp实现

chromedp是基于Chrome DevTools Protocol的无依赖自动化工具,适合需要精细控制浏览器行为的场景:

package main

import (
	"context"
	"fmt"
	"io/ioutil"
	"time"

	"github.com/chromedp/chromedp"
)

func main() {
	// 初始化浏览器上下文
	ctx, cancel := chromedp.NewContext(context.Background())
	defer cancel()

	// 设置全局超时,防止无限等待
	ctx, cancel = context.WithTimeout(ctx, 5*time.Minute)
	defer cancel()

	targetURL := "替换为你的目标页面URL"
	var fullHTML string
	var lastPageHeight int64

	err := chromedp.Run(ctx,
		// 导航到目标页面
		chromedp.Navigate(targetURL),
		// 等待页面初始渲染完成
		chromedp.Sleep(2*time.Second),
		// 循环滚动加载内容
		chromedp.ActionFunc(func(ctx context.Context) error {
			for {
				var currentHeight int64
				// 获取当前页面总高度
				if err := chromedp.Evaluate(`document.body.scrollHeight`, &currentHeight).Do(ctx); err != nil {
					return err
				}
				// 高度不再变化,说明已加载全部内容
				if currentHeight == lastPageHeight {
					break
				}
				lastPageHeight = currentHeight
				// 滚动到页面底部
				if err := chromedp.Evaluate(`window.scrollTo(0, document.body.scrollHeight)`, nil).Do(ctx); err != nil {
					return err
				}
				// 等待新内容加载,时间根据网站速度调整
				time.Sleep(3*time.Second)
			}
			return nil
		}),
		// 获取整个页面的HTML内容
		chromedp.OuterHTML("html", &fullHTML),
		// 如果只需要指定标签,比如所有class为"item"的元素,可替换为:
		// chromedp.OuterHTML(".item", &targetElementsHTML),
	)

	if err != nil {
		fmt.Printf("执行出错: %v\n", err)
		return
	}

	// 保存HTML到本地文件
	if err := ioutil.WriteFile("full_page.html", []byte(fullHTML), 0644); err != nil {
		fmt.Printf("保存文件失败: %v\n", err)
		return
	}

	fmt.Println("页面内容已成功保存")
}

方案二:使用rod实现

rod是更简洁易用的浏览器自动化库,API设计更直观:

package main

import (
	"fmt"
	"io/ioutil"
	"time"

	"github.com/go-rod/rod"
)

func main() {
	// 连接浏览器(自动启动本地Chrome/Edge)
	browser := rod.New().MustConnect()
	defer browser.MustClose()

	// 打开目标页面
	page := browser.MustPage("替换为你的目标页面URL")
	defer page.MustClose()

	lastHeight := 0
	for {
		// 获取当前页面总高度
		currentHeight := page.MustEval("document.body.scrollHeight").Int()
		// 高度不变则停止滚动
		if currentHeight == lastHeight {
			break
		}
		lastHeight = currentHeight
		// 滚动到底部
		page.MustEval("window.scrollTo(0, document.body.scrollHeight)")
		// 等待页面空闲(无网络请求和DOM更新),再额外等待1秒确保加载完成
		page.MustWaitIdle()
		time.Sleep(1 * time.Second)
	}

	// 获取整个页面HTML
	fullHTML := page.MustHTML()

	// 如果需要提取指定标签内容,示例:
	// items := page.MustElements(".item")
	// for _, item := range items {
	//     fmt.Println("元素文本:", item.MustText())
	// }

	// 保存内容到文件
	if err := ioutil.WriteFile("rod_full_page.html", []byte(fullHTML), 0644); err != nil {
		fmt.Printf("保存文件失败: %v\n", err)
		return
	}

	fmt.Println("内容保存完成")
}

关键注意事项

  • 等待时间调整:不同网站的内容加载速度不同,可根据实际情况修改time.Sleep的时长,或者用更精准的等待方式(比如等待某个特定元素出现)。
  • 反爬规避:部分网站会检测自动化工具,可通过设置自定义User-Agent绕过:
    • chromedp:在初始化时添加chromedp.Flag("user-agent", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 ...")
    • rod:调用page.SetUserAgent("你的自定义UA")
  • 内存占用:如果页面内容极大,加载全部元素可能消耗较多内存,可考虑分批提取元素并保存,而非一次性获取整个HTML。

内容的提问来源于stack exchange,提问作者ieatglue

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.31 22:35:18