Pixabay图片爬虫遇ValueError及None类型错误求助
修复Pixabay图片爬虫的URL错误问题
问题根源分析
- 爬虫抓取到
/static/img/blank.gif这类相对路径URL,urllib.request无法识别相对路径,触发ValueError: unknown url type - 部分图片元素的所有src相关属性均为
None,后续调用len(srcset)时触发类型错误
具体修复措施
- 补全相对路径URL:判断URL是否以
http开头,否则拼接Pixabay域名https://pixabay.com - 处理None值:在使用图片源链接前先判断是否有效,避免调用字符串方法时出错
- 优化图片源选择:优先提取高分辨率图片链接(srcset中最后一个元素),而非最小尺寸
- 容错性提升:遇到HTTP错误时跳过当前图片,继续爬取剩余内容,而非直接退出程序
修改后的完整代码
import os import sys import urllib.request from selenium import webdriver from selenium.webdriver.chrome.service import Service from time import sleep from bs4 import BeautifulSoup keyword = input('검색어 : ') maxImages = int(input('다운로드 시도할 최대 이미지 수 : ')) # 保存路径设置 path = f'crawled_images/{keyword}_{maxImages}' try: if not os.path.exists(path): os.makedirs(path) else: print('이전에 같은 [검색어, 이미지 수]로 다운로드한 폴더가 존재합니다.') sys.exit(0) except OSError: print('os error') sys.exit(0) pages = int((maxImages - 1) / 100) + 1 imgCount = 0 success = 0 finish = False # Chrome驱动配置 service = Service(executable_path='chromedriver') options = webdriver.ChromeOptions() # options.add_argument('headless') options.add_argument('--disable-gpu') options.add_argument('lang=ko_KR') options.add_argument("user-agent=Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/114.0.0.0 Safari/537.36") driver = webdriver.Chrome(service=service, options=options) for i in range(1, pages + 1): driver.get(f'https://pixabay.com/images/search/{keyword}/?pagi={i}') sleep(1) html = driver.page_source soup = BeautifulSoup(html, 'html.parser') imgs = soup.select('div.verticalMasonry--RoKfF.lg--v7yE8 img') lastPage = len(imgs) != 100 for img in imgs: # 获取所有可能的图片源属性 src = img.get('src') srcset = img.get('srcset') lazy_srcset = img.get('data-lazy-srcset') lazy_src = img.get('data-lazy-src') # 优先选择有效的图片链接 target_url = None if lazy_srcset: # 取srcset中最后一个(最高分辨率)的链接 target_url = lazy_srcset.split()[-2] elif srcset: target_url = srcset.split()[-2] elif lazy_src: target_url = lazy_src elif src: target_url = src # 处理相对路径和无效URL if target_url: # 补全相对路径 if not target_url.startswith(('http://', 'https://')): target_url = f'https://pixabay.com{target_url}' # 跳过blank.gif这类占位图 if 'blank.gif' in target_url: imgCount += 1 continue try: filename = target_url.split('/')[-1] save_path = os.path.join(path, filename) req = urllib.request.Request(target_url, headers={'User-Agent': 'Mozilla/5.0'}) img_data = urllib.request.urlopen(req).read() with open(save_path, 'wb') as f: f.write(img_data) success += 1 print(f'성공 저장: {save_path}') except urllib.error.HTTPError as e: print(f'에러 발생 (스킵): {e}') except Exception as e: print(f'기타 에러 (스킵): {e}') imgCount += 1 if imgCount == maxImages: finish = True break if finish or lastPage: break driver.quit() print(f'성공 : {success}, 실패 : {maxImages - success}')
内容的提问来源于stack exchange,提问作者김도열
相关产品推荐
相关产品推荐

