Scrapy爬取PDF遇UnicodeEncodeError及PDF损坏问题求解
解决Scrapy爬虫Base64解码非ASCII字符错误
问题说明
采用StackOverflow方案开发的Scrapy爬虫,在下载巴西CVM网站PDF时,部分URL触发Base64解码错误,提示字符串包含非ASCII字符。手动过滤非ASCII字符后,所有保存的PDF均损坏,但对应网页可正常访问。
错误日志
2023-05-29 19:22:20 [scrapy.core.scraper] ERROR: Spider error processing <POST https://www.rad.cvm.gov.br/ENET/frmExibirArquivoIPEExterno.aspx/ExibirPDF> (referer: https://www.rad.cvm.gov.br/ENET/frmExibirArquivoIPEExterno.aspx?NumeroProtocoloEntrega=1106380) Traceback (most recent call last): File "/home/higo/anaconda3/lib/python3.9/base64.py", line 37, in _bytes_from_decode_data return s.encode('ascii') UnicodeEncodeError: 'ascii' codec can't encode character '\xe3' in position 7: ordinal not in range(128) During handling of the above exception, another exception occurred: Traceback (most recent call last): File "/home/higo/anaconda3/lib/python3.9/site-packages/twisted/internet/defer.py", line 857, in _runCallbacks current.result = callback( # type: ignore[misc] File "/home/higo/Documentos/Doutorado/Artigo/scrape_fatos/scrape_fatos/spiders/fatos.py", line 63, in download_pdf pdf = base64.b64decode(b64) File "/home/higo/anaconda3/lib/python3.9/base64.py", line 80, in b64decode s = _bytes_from_decode_data(s) File "/home/higo/anaconda3/lib/python3.9/base64.py", line 39, in _bytes_from_decode_data raise ValueError('string argument should contain only ASCII characters') ValueError: string argument should contain only ASCII characters
完整爬虫代码
import base64 import logging import os import re from urllib.parse import unquote import scrapy class FatosSpider(scrapy.Spider): name = 'fatos' allowed_domains = ['cvm.gov.br'] with open("urls.txt", "rt") as f: start_urls = [url.strip() for url in f.readlines()] base_dir = './pdf_downloads' def parse(self, response): id_ = self.get_parameter_by_name("ID", response.url) if id_: numeroProtocolo = id_ codInstituicao = 2 else: numeroProtocolo = self.get_parameter_by_name("NumeroProtocoloEntrega", response.url) codInstituicao = 1 dataValue = "{ codigoInstituicao: '" + str(codInstituicao) + "', numeroProtocolo: '" + str(numeroProtocolo) + "'" token = response.xpath('//*[@id="hdnTokenB3"]/@value').get(default='') versaoCaptcha = '' if response.xpath('//*[@id="hdnHabilitaCaptcha"]/@value').get(default='') == 'S': if not token: versaoCaptcha = 'V3' payload = dataValue + ", token: '" + token + "', versaoCaptcha: '" + versaoCaptcha + "'}" url = 'https://www.rad.cvm.gov.br/ENET/frmExibirArquivoIPEExterno.aspx/ExibirPDF' headers = { "Accept": "application/json, text/javascript, */*; q=0.01", "Accept-Encoding": "gzip, deflate, br", "Accept-Language": "en-US,en;q=0.5", "Cache-Control": "no-cache", "Connection": "keep-alive", "Content-Type": "application/json; charset=utf-8", "DNT": "1", "Host": "www.rad.cvm.gov.br", "Origin": "https://www.rad.cvm.gov.br", "Pragma": "no-cache", "Referer": f"https://www.rad.cvm.gov.br/ENET/frmExibirArquivoIPEExterno.aspx?NumeroProtocoloEntrega={numeroProtocolo}", "Sec-Fetch-Dest": "empty", "Sec-Fetch-Mode": "cors", "Sec-Fetch-Site": "same-origin", "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:109.0) Gecko/20100101 Firefox/113.0", "X-Requested-With": "XMLHttpRequest" } yield scrapy.Request(url=url, headers=headers, body=payload, method='POST', callback=self.download_pdf, cb_kwargs={'protocol_num': numeroProtocolo}) def download_pdf(self, response, protocol_num): json_data = response.json() b64 = json_data.get('d') if b64: pdf = base64.b64decode(b64) filename = f'{protocol_num}.pdf' p = os.path.join(self.base_dir, filename) if not os.path.isdir(self.base_dir): os.mkdir(self.base_dir) with open(p, 'wb') as f: f.write(pdf) self.log(f"Saved {filename} in {self.base_dir}") else: self.log("Couldn't download pdf", logging.ERROR) @staticmethod def get_parameter_by_name(name, url): name = name.replace('[', '\\[').replace(']', '\\]') results = re.search(r"[?&]" + name + r"(=([^&#]*)|&|#|$)", url) if not results: return None if len(results.groups()) < 2 or not results[2]: return '' return unquote(results[2])
解决方案
问题根源
返回的Base64字符串可能经过了URL编码(比如特殊字符被转义),直接解码会因非ASCII字符失败;直接删除非ASCII字符会破坏Base64结构,导致PDF损坏。
修复后的download_pdf方法
def download_pdf(self, response, protocol_num): json_data = response.json() b64 = json_data.get('d') if b64: try: # 还原URL编码的字符,比如转义的特殊字符 b64_decoded = unquote(b64) # 仅保留Base64规范允许的字符(A-Za-z0-9+/=),过滤无效字符 valid_b64 = re.sub(r'[^A-Za-z0-9+/=]', '', b64_decoded) # 执行Base64解码 pdf = base64.b64decode(valid_b64) filename = f'{protocol_num}.pdf' p = os.path.join(self.base_dir, filename) if not os.path.isdir(self.base_dir): os.mkdir(self.base_dir) with open(p, 'wb') as f: f.write(pdf) self.log(f"Saved {filename} in {self.base_dir}") except (ValueError, TypeError) as e: # 捕获解码异常,不中断爬虫 self.log(f"Failed to decode PDF {protocol_num}: {str(e)}", logging.ERROR) else: self.log("Couldn't download pdf", logging.ERROR)
修复说明
- URL解码:用
unquote还原被URL转义的字符,避免因转义字符导致的非ASCII错误 - 过滤无效字符:仅保留Base64规范内的字符,既解决非ASCII问题,又不破坏Base64结构
- 异常捕获:单个PDF解码失败时记录错误,继续处理其他URL,保证爬虫稳定性
内容的提问来源于stack exchange,提问作者Higo Felipe Silva Pires
相关产品推荐
相关产品推荐

