You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Python抓取在线文章生成PNG摘要时遇Unicode编码错误

问题

我用Python批量抓取多个网页的标题和前4段内容生成摘要,再把结果保存为PNG图片时,遇到了Unicode编码错误。

报错详情
UnicodeEncodeError                        Traceback (most recent call last)
 in <cell line: 70>()
78
79     # Calculate height for summary
---> 80     summary_height = calculate_text_height(draw, summary, font_summary)
81     current_height += summary_height + 30  # 30 pixels spacing after summary
82

3 frames
/usr/local/lib/python3.10/dist-packages/PIL/ImageFont.py in getlength(self, text, *args, **kwargs)
203         .. versionadded:: 9.2.0
204         """
---> 205         width, height = self.font.getsize(text)
206         return width
207

UnicodeEncodeError: 'latin-1' codec can't encode character '\u2019' in position 4: ordinal not in range(256)
相关代码
import requests
from bs4 import BeautifulSoup
from PIL import Image, ImageDraw, ImageFont

# Function to get title and summary from a URL
def extract_info(url):
    try:
        response = requests.get(url)
        response.encoding = 'utf-8'  # Ensure UTF-8 encoding
        soup = BeautifulSoup(response.text, 'html.parser')
        
        # Extract title
        title = soup.title.string.strip() if soup.title and soup.title.string else 'No Title Found'
        
        # Extract the first 4 paragraphs as summary
        paragraphs = soup.find_all('p')
        if paragraphs:
            summary_paragraphs = [p.get_text(strip=True) for p in paragraphs[:4] if p.get_text(strip=True)]
            summary = '\n\n'.join(summary_paragraphs) if summary_paragraphs else 'No Summary Found'
        else:
            summary = 'No Summary Found'
        
        return title, summary
    except Exception as e:
        return 'Error fetching details', str(e)

# List of URLs
urls = [
     "https://www.cisa.gov/news-events/events",
    "https://www.cisa.gov/news-events/cybersecurity-advisories",
    "https://www.cisa.gov/news-events/alerts/2024/08/14/adobe-releases-security-updates-multiple-products",
    "https://www.youtube.com/@cisagov",
    "https://www.dhs.gov",
    "https://www.kaspersky.com/home-security?icid=gl_securelisheader_acq_ona_smm__onl_b2c_securelist_prodmen_______",
  ]

  # Collect all the data
data = []
for url in urls:
    print(f"Processing URL: {url}")
    title, summary = extract_info(url)
    data.append((title, url, summary))

# Determine the image height dynamically
 width = 1200
 padding = 50  # Padding around text
 current_height = padding

# Use a Unicode-compatible font
try:
   font_path = "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf"  
   font_title = ImageFont.truetype(font_path, 24)
   font_url = ImageFont.truetype(font_path, 20)
   font_summary = ImageFont.truetype(font_path, 18)
except:
   font_title = ImageFont.load_default()
   font_url = ImageFont.load_default()
   font_summary = ImageFont.load_default()

# Create a dummy image to calculate text size
dummy_img = Image.new('RGB', (width, 1000))
draw = ImageDraw.Draw(dummy_img)

def calculate_text_height(draw, text, font):
# Calculate the bounding box of the text and return its height
   bbox = draw.multiline_textbbox((0, 0), text, font=font)
   return bbox[3] - bbox[1]

for title, url, summary in data:
   # Calculate height for title
    title_height = calculate_text_height(draw, title, font_title)
    current_height += title_height + 10  # 10 pixels spacing

 # Calculate height for url
url_height = calculate_text_height(draw, url, font_url)
current_height += url_height + 10

# Calculate height for summary
summary_height = calculate_text_height(draw, summary, font_summary)
current_height += summary_height + 30  # 30 pixels spacing after summary

# Add some bottom padding
current_height += padding

  # Create the final image
image = Image.new('RGB', (width, current_height), color=(255, 255, 255))
draw = ImageDraw.Draw(image)

y_text = padding
for title, url, summary in data:
    # Draw title
    draw.text((padding, y_text), title, font=font_title, fill="black")
    y_text += calculate_text_height(draw, title, font_title) + 10  # 10 pixels spacing

# Draw URL
draw.text((padding, y_text), url, font=font_url, fill="blue")
y_text += calculate_text_height(draw, url, font_url) + 10

# Draw Summary
draw.multiline_text((padding, y_text), summary, font=font_summary, fill="black", spacing=4)
y_text += calculate_text_height(draw, summary, font_summary) + 30  # 30 pixels spacing after summary

# Save the image in the current working directory
image.save('website_info.png')

# Display the image
image.show()
解决方案

这个错误是因为PIL在处理包含特殊Unicode字符(比如’,对应\u2019)的文本时,默认使用latin-1编码,无法处理超出ASCII范围的字符。可通过以下方式解决:

1. 明确加载支持Unicode的字体并指定编码

修改字体加载逻辑,确保加载支持Unicode的字体,并明确指定UTF-8编码:

# 替换原字体加载部分
font_path = "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf"
try:
    font_title = ImageFont.truetype(font_path, 24, encoding="utf-8")
    font_url = ImageFont.truetype(font_path, 20, encoding="utf-8")
    font_summary = ImageFont.truetype(font_path, 18, encoding="utf-8")
except IOError:
    # 根据系统自动选择备选Unicode字体
    import sys
    if sys.platform == 'win32':
        font_path = "arial.ttf"
    elif sys.platform == 'darwin':
        font_path = "/Library/Fonts/Arial Unicode.ttf"
    else:
        font_path = "/usr/share/fonts/truetype/freefont/FreeSans.ttf"
    font_title = ImageFont.truetype(font_path, 24, encoding="utf-8")
    font_url = ImageFont.truetype(font_path, 20, encoding="utf-8")
    font_summary = ImageFont.truetype(font_path, 18, encoding="utf-8")

2. 预处理文本,替换/过滤特殊字符

如果无法加载合适字体,可提前将特殊Unicode字符替换为ASCII等效字符,或过滤无法编码的字符:

def clean_unicode_text(text):
    # 替换常见特殊字符为ASCII等效字符
    replacements = {
        '\u2019': "'",
        '\u201c': '"',
        '\u201d': '"',
        '\u2013': '-',
        '\u2014': '--'
    }
    for orig, repl in replacements.items():
        text = text.replace(orig, repl)
    # 移除剩余无法用latin-1编码的字符
    return text.encode('latin-1', 'ignore').decode('latin-1')

# 在extract_info函数中调用清理函数
title = clean_unicode_text(soup.title.string.strip()) if soup.title and soup.title.string else 'No Title Found'
# 摘要部分同样处理
summary_paragraphs = [clean_unicode_text(p.get_text(strip=True)) for p in paragraphs[:4] if p.get_text(strip=True)]

3. 升级Pillow版本

旧版本Pillow存在Unicode处理bug,升级到最新版本可修复部分问题:

pip install --upgrade pillow

内容的提问来源于stack exchange,提问作者New2015

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.19 15:43:11