Go语言爬虫:如何利用结构体中的URL跳转页面获取食谱数据
解决Golang爬虫关联列表页与详情页数据的问题
你的核心问题是当前代码将列表页的菜品信息(名称、URL)和详情页数据(描述、食材等)分开存储,没有建立对应关联。可以通过Colly的请求上下文(Request.Ctx)传递当前菜品的指针,在详情页回调中直接填充数据到对应的结构体中,实现数据的统一存储。
以下是修改后的完整代码:
package main import ( "encoding/json" "fmt" "net" "net/http" "os" "time" "github.com/gocolly/colly" ) type Product struct { Name string `json:"name"` URL string `json:"url"` Description string `json:"description"` Ingredients string `json:"ingredients"` PhotoURL string `json:"photo_url"` Directions string `json:"directions"` } var allProducts []*Product func main() { c := colly.NewCollector() c.WithTransport(&http.Transport{ DialContext: (&net.Dialer{ Timeout: 60 * time.Second, KeepAlive: 30 * time.Second, DualStack: true, }).DialContext, MaxIdleConns: 100, IdleConnTimeout: 90 * time.Second, TLSHandshakeTimeout: 10 * time.Second, ExpectContinueTimeout: 1 * time.Second, }) c.OnRequest(func(r *colly.Request) { fmt.Println("正在爬取:", r.URL) }) c.OnResponse(func(r *colly.Response) { fmt.Println("请求状态:", r.StatusCode) }) // 爬取列表页,获取菜品名称和详情URL c.OnHTML("a.mntl-card", func(h *colly.HTMLElement) { product := &Product{ Name: h.ChildText(".card__title-text"), URL: h.Attr("href"), } allProducts = append(allProducts, product) // 将product指针存入请求上下文,传递到详情页回调 ctx := colly.NewContext() ctx.Put("product", product) // 访问详情页,携带上下文 if err := c.Request("GET", product.URL, nil, ctx, nil); err != nil { fmt.Println("访问详情页失败:", err) } }) // 爬取详情页,填充菜品的详细数据 c.OnHTML("article.mntl-recipe-content", func(h *colly.HTMLElement) { // 从上下文取出当前对应的product指针 product, ok := h.Request.Ctx.Get("product").(*Product) if !ok { fmt.Println("无法从上下文获取product") return } // 调整选择器以匹配网站实际结构(根据allrecipes当前页面结构调整) product.Description = h.ChildText("p.mntl-recipe-subheading__text") product.PhotoURL = h.ChildAttr("img.mntl-primary-image--img", "src") // 拼接所有食材文本 ingredients := "" h.ForEach("li.mntl-structured-ingredients__list-item", func(_ int, el *colly.HTMLElement) { ingredients += el.Text + "\n" }) product.Ingredients = ingredients // 拼接所有步骤文本 directions := "" h.ForEach("p.mntl-sc-block-html", func(_ int, el *colly.HTMLElement) { directions += el.Text + "\n" }) product.Directions = directions }) c.OnError(func(r *colly.Response, err error) { fmt.Println("请求URL:", r.Request.URL, "失败,响应:", r, "错误:", err) }) // 开始爬取列表页 c.Visit("https://www.allrecipes.com/recipes/17562/dinner/") // 序列化所有菜品数据到JSON文件 content, err := json.MarshalIndent(allProducts, "", " ") if err != nil { fmt.Println("JSON序列化失败:", err) return } if err := os.WriteFile("recipes.json", content, 0644); err != nil { fmt.Println("写入文件失败:", err) return } fmt.Println("爬取完成,共获取", len(allProducts), "道菜品") }
关键修改说明:
- 结构体简化:合并原有的
products和recettes为一个Product结构体,避免数据分散,更符合逻辑。 - 上下文传递数据:使用
colly.NewContext()创建请求上下文,将当前菜品的指针存入其中,在访问详情页时携带该上下文,确保详情页数据能准确对应到列表页的菜品。 - 选择器修正:根据allrecipes网站当前的页面结构调整了HTML选择器(比如图片、描述、食材、步骤的选择器),避免因页面结构变化导致爬取不到数据。
- 数据拼接优化:对食材和步骤使用循环遍历的方式拼接所有内容,避免只获取到第一个元素的文本。
- 错误处理增强:在请求详情页时增加了错误捕获,方便排查访问失败的问题。
内容的提问来源于stack exchange,提问作者maka
相关产品推荐
相关产品推荐

