如何用R从moweek.com.uy下载指定分类的图片及属性数据
问题背景
- 目标网站:https://moweek.com.uy/
- 需抓取的主分类:class为
expandedCategory的三个分类:VESTIMENTA(data-category-id="1")、CALZADO(data-category-id="2")、ACCESORIOS(data-category-id="3") - 需排除主分类下的“Ver todo”链接,仅抓取其他子分类
- 每个子分类页面需提取商品信息(分类、子分类、商品名、价格、图片链接)存入数据集,同时下载图片到对应分类/子分类文件夹
原代码问题分析
- 抓取子分类链接时未排除“Ver todo”,导致无效页面请求
- 图片下载部分代码未完成(
file_name <- paste),无法保存图片 - 未处理子分类的分页情况(部分子分类商品多页)
- 部分节点选择可能失效,且未处理节点不存在的异常情况
- 无反爬延迟,可能被网站限制访问
修正后的完整代码
pacman::p_load(tidyverse, rvest, httr, fs) # 基础设置 base_url <- "https://moweek.com.uy" delay_sec <- 2 # 防反爬延迟 # 获取主分类下的有效子分类链接 main_page <- read_html(base_url) category_nodes <- main_page %>% html_nodes(".expandedCategory") # 筛选出目标主分类,并提取子分类链接(排除"Ver todo") subcategory_urls <- map(category_nodes, function(node) { # 获取主分类ID,筛选目标分类 cat_id <- node %>% html_attr("data-category-id") if (!cat_id %in% c("1", "2", "3")) return(NULL) # 提取子分类链接,排除包含"ver-todo"的链接 node %>% html_nodes("a.categoryLevelTwoTitle") %>% html_attr("href") %>% str_subset("ver-todo", negate = TRUE) }) %>% unlist() %>% unique() # 初始化数据集 image_data <- tibble() # 遍历每个子分类链接(含分页处理) for (sub_url in subcategory_urls) { page_num <- 1 while(TRUE) { # 构建当前分页URL current_url <- paste0(base_url, sub_url, ifelse(page_num == 1, "", paste0("?page=", page_num))) cat("正在处理:", current_url, "\n") # 请求页面并解析 Sys.sleep(delay_sec) response <- GET(current_url) if (http_status(response)$category != "Success") { cat("页面请求失败,跳过:", current_url, "\n") break } sub_page <- read_html(content(response, as = "text")) # 提取分类和子分类名称 cat_name <- sub_page %>% html_node(".categoryLevelOneTitle") %>% html_text() %>% str_to_title() subcat_name <- sub_page %>% html_node(".categoryLevelTwoTitle.selected") %>% html_text() %>% str_to_title() # 提取当前页商品信息 product_nodes <- sub_page %>% html_nodes(".productItem") if (length(product_nodes) == 0) break # 无商品则停止分页 current_page_data <- map_dfr(product_nodes, function(prod_node) { tibble( Category = cat_name, Subcategory = subcat_name, Name = prod_node %>% html_node(".productTitle") %>% html_text(trim = TRUE), Price = prod_node %>% html_node(".priceText") %>% html_text(trim = TRUE), Image_URL = prod_node %>% html_node("img") %>% html_attr("src") %>% str_replace("^//", "https://") # 修复图片链接格式 ) }) # 合并数据 image_data <- bind_rows(image_data, current_page_data) # 检查是否有下一页 has_next <- sub_page %>% html_nodes(".pagination .next") %>% length() > 0 if (!has_next) break page_num <- page_num + 1 } } # 下载图片到对应文件夹 for (i in 1:nrow(image_data)) { img_url <- image_data$Image_URL[i] if (is.na(img_url)) next # 创建文件夹路径 category_folder <- image_data$Category[i] %>% str_replace_all(" ", "_") subcategory_folder <- image_data$Subcategory[i] %>% str_replace_all(" ", "_") full_folder_path <- path(category_folder, subcategory_folder) # 创建文件夹(如果不存在) dir_create(full_folder_path, recurse = TRUE) # 生成文件名(避免重复,用商品名+随机字符串) file_name <- image_data$Name[i] %>% str_replace_all("[^a-zA-Z0-9_]", "_") %>% paste0("_", str_trunc(sha1(img_url), 6, "right"), ".jpg") full_file_path <- path(full_folder_path, file_name) # 下载图片 if (!file_exists(full_file_path)) { Sys.sleep(delay_sec) tryCatch({ GET(img_url, write_disk(full_file_path, overwrite = TRUE)) cat("已下载:", full_file_path, "\n") }, error = function(e) { cat("下载失败:", img_url, ",错误信息:", e$message, "\n") }) } else { cat("文件已存在,跳过:", full_file_path, "\n") } } # 保存数据集到CSV write_csv(image_data, "moweek_product_data.csv") cat("数据已保存到moweek_product_data.csv\n")
代码关键改进点
- 精准筛选子分类:通过主分类ID筛选目标分类,并排除"Ver todo"链接
- 分页处理:自动检测并遍历子分类的所有分页页面
- 链接修复:处理图片链接的相对路径格式(将
//转为https://) - 异常处理:添加页面请求失败、图片下载失败的捕获逻辑,避免程序中断
- 防反爬机制:添加固定延迟,避免请求过于频繁
- 文件管理:使用
fs包简化文件夹创建和文件路径处理,生成唯一文件名避免重复 - 数据清洗:对文本字段进行去空格处理,确保数据整洁
内容的提问来源于stack exchange,提问作者Paula
相关产品推荐
相关产品推荐

