You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Go语言爬虫:如何利用结构体中的URL跳转页面获取食谱数据

解决Golang爬虫关联列表页与详情页数据的问题

你的核心问题是当前代码将列表页的菜品信息(名称、URL)和详情页数据(描述、食材等)分开存储,没有建立对应关联。可以通过Colly的请求上下文(Request.Ctx)传递当前菜品的指针,在详情页回调中直接填充数据到对应的结构体中,实现数据的统一存储。

以下是修改后的完整代码:

package main

import (
	"encoding/json"
	"fmt"
	"net"
	"net/http"
	"os"
	"time"

	"github.com/gocolly/colly"
)

type Product struct {
	Name         string `json:"name"`
	URL          string `json:"url"`
	Description  string `json:"description"`
	Ingredients  string `json:"ingredients"`
	PhotoURL     string `json:"photo_url"`
	Directions   string `json:"directions"`
}

var allProducts []*Product

func main() {
	c := colly.NewCollector()
	c.WithTransport(&http.Transport{
		DialContext: (&net.Dialer{
			Timeout:   60 * time.Second,
			KeepAlive: 30 * time.Second,
			DualStack: true,
		}).DialContext,
		MaxIdleConns:          100,
		IdleConnTimeout:       90 * time.Second,
		TLSHandshakeTimeout:   10 * time.Second,
		ExpectContinueTimeout: 1 * time.Second,
	})

	c.OnRequest(func(r *colly.Request) {
		fmt.Println("正在爬取:", r.URL)
	})

	c.OnResponse(func(r *colly.Response) {
		fmt.Println("请求状态:", r.StatusCode)
	})

	// 爬取列表页,获取菜品名称和详情URL
	c.OnHTML("a.mntl-card", func(h *colly.HTMLElement) {
		product := &Product{
			Name: h.ChildText(".card__title-text"),
			URL:  h.Attr("href"),
		}
		allProducts = append(allProducts, product)
		// 将product指针存入请求上下文,传递到详情页回调
		ctx := colly.NewContext()
		ctx.Put("product", product)
		// 访问详情页,携带上下文
		if err := c.Request("GET", product.URL, nil, ctx, nil); err != nil {
			fmt.Println("访问详情页失败:", err)
		}
	})

	// 爬取详情页,填充菜品的详细数据
	c.OnHTML("article.mntl-recipe-content", func(h *colly.HTMLElement) {
		// 从上下文取出当前对应的product指针
		product, ok := h.Request.Ctx.Get("product").(*Product)
		if !ok {
			fmt.Println("无法从上下文获取product")
			return
		}

		// 调整选择器以匹配网站实际结构(根据allrecipes当前页面结构调整)
		product.Description = h.ChildText("p.mntl-recipe-subheading__text")
		product.PhotoURL = h.ChildAttr("img.mntl-primary-image--img", "src")
		// 拼接所有食材文本
		ingredients := ""
		h.ForEach("li.mntl-structured-ingredients__list-item", func(_ int, el *colly.HTMLElement) {
			ingredients += el.Text + "\n"
		})
		product.Ingredients = ingredients
		// 拼接所有步骤文本
		directions := ""
		h.ForEach("p.mntl-sc-block-html", func(_ int, el *colly.HTMLElement) {
			directions += el.Text + "\n"
		})
		product.Directions = directions
	})

	c.OnError(func(r *colly.Response, err error) {
		fmt.Println("请求URL:", r.Request.URL, "失败,响应:", r, "错误:", err)
	})

	// 开始爬取列表页
	c.Visit("https://www.allrecipes.com/recipes/17562/dinner/")

	// 序列化所有菜品数据到JSON文件
	content, err := json.MarshalIndent(allProducts, "", "  ")
	if err != nil {
		fmt.Println("JSON序列化失败:", err)
		return
	}

	if err := os.WriteFile("recipes.json", content, 0644); err != nil {
		fmt.Println("写入文件失败:", err)
		return
	}

	fmt.Println("爬取完成,共获取", len(allProducts), "道菜品")
}

关键修改说明:

  • 结构体简化:合并原有的products和recettes为一个Product结构体,避免数据分散,更符合逻辑。
  • 上下文传递数据:使用colly.NewContext()创建请求上下文,将当前菜品的指针存入其中,在访问详情页时携带该上下文,确保详情页数据能准确对应到列表页的菜品。
  • 选择器修正:根据allrecipes网站当前的页面结构调整了HTML选择器(比如图片、描述、食材、步骤的选择器),避免因页面结构变化导致爬取不到数据。
  • 数据拼接优化:对食材和步骤使用循环遍历的方式拼接所有内容,避免只获取到第一个元素的文本。
  • 错误处理增强:在请求详情页时增加了错误捕获,方便排查访问失败的问题。

内容的提问来源于stack exchange,提问作者maka

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.25 10:22:03