使用Python抓取在线文章生成PNG摘要时遇Unicode编码错误
问题
我用Python批量抓取多个网页的标题和前4段内容生成摘要,再把结果保存为PNG图片时,遇到了Unicode编码错误。
报错详情
UnicodeEncodeError Traceback (most recent call last) in <cell line: 70>() 78 79 # Calculate height for summary ---> 80 summary_height = calculate_text_height(draw, summary, font_summary) 81 current_height += summary_height + 30 # 30 pixels spacing after summary 82 3 frames /usr/local/lib/python3.10/dist-packages/PIL/ImageFont.py in getlength(self, text, *args, **kwargs) 203 .. versionadded:: 9.2.0 204 """ ---> 205 width, height = self.font.getsize(text) 206 return width 207 UnicodeEncodeError: 'latin-1' codec can't encode character '\u2019' in position 4: ordinal not in range(256)
相关代码
import requests from bs4 import BeautifulSoup from PIL import Image, ImageDraw, ImageFont # Function to get title and summary from a URL def extract_info(url): try: response = requests.get(url) response.encoding = 'utf-8' # Ensure UTF-8 encoding soup = BeautifulSoup(response.text, 'html.parser') # Extract title title = soup.title.string.strip() if soup.title and soup.title.string else 'No Title Found' # Extract the first 4 paragraphs as summary paragraphs = soup.find_all('p') if paragraphs: summary_paragraphs = [p.get_text(strip=True) for p in paragraphs[:4] if p.get_text(strip=True)] summary = '\n\n'.join(summary_paragraphs) if summary_paragraphs else 'No Summary Found' else: summary = 'No Summary Found' return title, summary except Exception as e: return 'Error fetching details', str(e) # List of URLs urls = [ "https://www.cisa.gov/news-events/events", "https://www.cisa.gov/news-events/cybersecurity-advisories", "https://www.cisa.gov/news-events/alerts/2024/08/14/adobe-releases-security-updates-multiple-products", "https://www.youtube.com/@cisagov", "https://www.dhs.gov", "https://www.kaspersky.com/home-security?icid=gl_securelisheader_acq_ona_smm__onl_b2c_securelist_prodmen_______", ] # Collect all the data data = [] for url in urls: print(f"Processing URL: {url}") title, summary = extract_info(url) data.append((title, url, summary)) # Determine the image height dynamically width = 1200 padding = 50 # Padding around text current_height = padding # Use a Unicode-compatible font try: font_path = "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf" font_title = ImageFont.truetype(font_path, 24) font_url = ImageFont.truetype(font_path, 20) font_summary = ImageFont.truetype(font_path, 18) except: font_title = ImageFont.load_default() font_url = ImageFont.load_default() font_summary = ImageFont.load_default() # Create a dummy image to calculate text size dummy_img = Image.new('RGB', (width, 1000)) draw = ImageDraw.Draw(dummy_img) def calculate_text_height(draw, text, font): # Calculate the bounding box of the text and return its height bbox = draw.multiline_textbbox((0, 0), text, font=font) return bbox[3] - bbox[1] for title, url, summary in data: # Calculate height for title title_height = calculate_text_height(draw, title, font_title) current_height += title_height + 10 # 10 pixels spacing # Calculate height for url url_height = calculate_text_height(draw, url, font_url) current_height += url_height + 10 # Calculate height for summary summary_height = calculate_text_height(draw, summary, font_summary) current_height += summary_height + 30 # 30 pixels spacing after summary # Add some bottom padding current_height += padding # Create the final image image = Image.new('RGB', (width, current_height), color=(255, 255, 255)) draw = ImageDraw.Draw(image) y_text = padding for title, url, summary in data: # Draw title draw.text((padding, y_text), title, font=font_title, fill="black") y_text += calculate_text_height(draw, title, font_title) + 10 # 10 pixels spacing # Draw URL draw.text((padding, y_text), url, font=font_url, fill="blue") y_text += calculate_text_height(draw, url, font_url) + 10 # Draw Summary draw.multiline_text((padding, y_text), summary, font=font_summary, fill="black", spacing=4) y_text += calculate_text_height(draw, summary, font_summary) + 30 # 30 pixels spacing after summary # Save the image in the current working directory image.save('website_info.png') # Display the image image.show()
解决方案
这个错误是因为PIL在处理包含特殊Unicode字符(比如’,对应\u2019)的文本时,默认使用latin-1编码,无法处理超出ASCII范围的字符。可通过以下方式解决:
1. 明确加载支持Unicode的字体并指定编码
修改字体加载逻辑,确保加载支持Unicode的字体,并明确指定UTF-8编码:
# 替换原字体加载部分 font_path = "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf" try: font_title = ImageFont.truetype(font_path, 24, encoding="utf-8") font_url = ImageFont.truetype(font_path, 20, encoding="utf-8") font_summary = ImageFont.truetype(font_path, 18, encoding="utf-8") except IOError: # 根据系统自动选择备选Unicode字体 import sys if sys.platform == 'win32': font_path = "arial.ttf" elif sys.platform == 'darwin': font_path = "/Library/Fonts/Arial Unicode.ttf" else: font_path = "/usr/share/fonts/truetype/freefont/FreeSans.ttf" font_title = ImageFont.truetype(font_path, 24, encoding="utf-8") font_url = ImageFont.truetype(font_path, 20, encoding="utf-8") font_summary = ImageFont.truetype(font_path, 18, encoding="utf-8")
2. 预处理文本,替换/过滤特殊字符
如果无法加载合适字体,可提前将特殊Unicode字符替换为ASCII等效字符,或过滤无法编码的字符:
def clean_unicode_text(text): # 替换常见特殊字符为ASCII等效字符 replacements = { '\u2019': "'", '\u201c': '"', '\u201d': '"', '\u2013': '-', '\u2014': '--' } for orig, repl in replacements.items(): text = text.replace(orig, repl) # 移除剩余无法用latin-1编码的字符 return text.encode('latin-1', 'ignore').decode('latin-1') # 在extract_info函数中调用清理函数 title = clean_unicode_text(soup.title.string.strip()) if soup.title and soup.title.string else 'No Title Found' # 摘要部分同样处理 summary_paragraphs = [clean_unicode_text(p.get_text(strip=True)) for p in paragraphs[:4] if p.get_text(strip=True)]
3. 升级Pillow版本
旧版本Pillow存在Unicode处理bug,升级到最新版本可修复部分问题:
pip install --upgrade pillow
内容的提问来源于stack exchange,提问作者New2015
相关产品推荐
相关产品推荐

