ThreadPoolExecutor批量下载仅成功下载列表最后一个文件的问题排查
问题:多线程下载图片仅最后一张成功的原因及修复方案
我有一个名为images.txt的文件,每行是一个图片URL,内容如下:
https://upload.wikimedia.org/wikipedia/commons/thumb/3/3e/Glenn_Jacobs_%2853122237030%29_-_Cropped.jpg/440px-Glenn_Jacobs_%2853122237030%29_-_Cropped.jpg https://upload.wikimedia.org/wikipedia/commons/thumb/d/de/Kane2003.jpg/340px-Kane2003.jpg https://upload.wikimedia.org/wikipedia/commons/7/7a/Steel_Cage.jpg https://upload.wikimedia.org/wikipedia/commons/thumb/c/cd/Kane_2008.JPG/340px-Kane_2008.JPG https://upload.wikimedia.org/wikipedia/commons/thumb/e/e4/Brothers_of_Destruction.jpg/440px-Brothers_of_Destruction.jpg
我想用20个工作线程的ThreadPoolExecutor把每张图片下载到本地images子目录,但运行代码后控制台显示所有图片都在下载,最终却只有最后一张成功,其余都没下载。我的代码如下:
from os import makedirs from os.path import basename from os.path import join import shutil from concurrent.futures import ThreadPoolExecutor from concurrent.futures import as_completed import requests # load a file from a URL, returns content of downloaded file def download_url(urlpath, dir): # Set the headers for the request headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/85.0.4183.102 Safari/537.36", "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.9", "Accept-Language": "en-US,en;q=0.9" } r = requests.get(urlpath, headers=headers, stream=True) filename = basename(urlpath) outpath = join(dir, filename) if r.status_code == 200: with open(outpath, 'wb') as f: r.raw.decode_content = True shutil.copyfileobj(r.raw, f) # download one file to a local directory def download_url_to_file(link, path): download_url(link, path) return link # download all files on the provided webpage to the provided path def getInBulk(filePath, path): # download the html webpage # create a local directory to save files makedirs(path, exist_ok=True) # parse html and retrieve all href urls listed links = open(filePath).readlines() # report progress print(f'Found {len(links)} links') # create the pool of worker threads with ThreadPoolExecutor(max_workers=20) as exe: # dispatch all download tasks to worker threads futures = [exe.submit(download_url_to_file, link, path) for link in links] # report results as they become available for future in as_completed(futures): # retrieve result link = future.result() # check for a link that was skipped print(f'Downloaded {link} to directory') PATH = 'images' filePath = "images.txt" getInBulk(filePath, PATH)
错误原因
问题出在readlines()读取的URL末尾包含换行符(\n),导致请求的URL无效,服务器返回非200状态码,所以这些图片没有被下载。只有最后一行可能没有换行符,所以能成功下载。
修复方案
对读取到的每个URL进行去除首尾空白字符的处理,比如用strip()方法。同时建议增加错误处理,打印失败的请求信息,方便排查问题。
修复后的代码
from os import makedirs from os.path import basename from os.path import join import shutil from concurrent.futures import ThreadPoolExecutor from concurrent.futures import as_completed import requests # load a file from a URL, returns content of downloaded file def download_url(urlpath, dir): # Set the headers for the request headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/85.0.4183.102 Safari/537.36", "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.9", "Accept-Language": "en-US,en;q=0.9" } # 去除URL首尾的空白字符(包括换行符) urlpath = urlpath.strip() if not urlpath: print(f"Skipping empty URL") return False try: r = requests.get(urlpath, headers=headers, stream=True) r.raise_for_status() # 主动抛出HTTP错误 filename = basename(urlpath) outpath = join(dir, filename) with open(outpath, 'wb') as f: r.raw.decode_content = True shutil.copyfileobj(r.raw, f) return True except Exception as e: print(f"Failed to download {urlpath}: {str(e)}") return False # download one file to a local directory def download_url_to_file(link, path): success = download_url(link, path) return (link, success) # download all files on the provided webpage to the provided path def getInBulk(filePath, path): makedirs(path, exist_ok=True) # 读取文件并处理每个URL with open(filePath, 'r') as f: links = [line.strip() for line in f if line.strip()] print(f'Found {len(links)} valid links') with ThreadPoolExecutor(max_workers=20) as exe: futures = [exe.submit(download_url_to_file, link, path) for link in links] for future in as_completed(futures): link, success = future.result() if success: print(f'Downloaded {link} to directory') else: print(f'Failed to download {link}') PATH = 'images' filePath = "images.txt" getInBulk(filePath, PATH)
关键改动点
- 处理URL中的换行符:在读取文件时用
[line.strip() for line in f if line.strip()]过滤空行并去除每个URL的首尾空白字符;在download_url函数中也增加了urlpath.strip()的二次处理,双重保障。 - 增加错误处理:使用
r.raise_for_status()捕获HTTP错误,并用try-except块包裹请求逻辑,打印失败的具体原因,便于调试。 - 返回下载状态:修改
download_url_to_file的返回值,包含下载是否成功的状态,控制台可以明确显示每个URL的下载结果。
内容的提问来源于stack exchange,提问作者NedStarkOfWinterfell
相关产品推荐
相关产品推荐

