You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用pandas读取西语维基百科表格遇UnicodeEncodeError求助

问题

使用pandas的read_html函数读取西班牙维基百科页面中的特定表格时,触发Unicode编码错误:
执行代码:

df_mx = pd.read_html('https://es.wikipedia.org/wiki/Economía_de_México', match='Indicadores macroeconómicos, financieros y de bienestar')

出现报错:

UnicodeEncodeError: 'ascii' codec can't encode character '\xed' in position 16: ordinal not in range(128)

完整报错堆栈:

---------------------------------------------------------------------------
UnicodeEncodeError                        Traceback (most recent call last)
Cell In[14], line 1
----> 1 df_mx = pd.read_html('https://es.wikipedia.org/wiki/Economía_de_México', match='Indicadores macroeconómicos, financieros y de bienestar')

File ~/anaconda3/lib/python3.11/site-packages/pandas/io/html.py:1212, in read_html(io, match, flavor, header, index_col, skiprows, attrs, parse_dates, thousands, encoding, decimal, converters, na_values, keep_default_na, displayed_only, extract_links, dtype_backend)
   1208 check_dtype_backend(dtype_backend)
   1210 io = stringify_path(io)
-> 1212 return _parse(
   1213     flavor=flavor,
   1214     io=io,
   1215     match=match,
   1216     header=header,
   1217     index_col=index_col,
   1218     skiprows=skiprows,
   1219     parse_dates=parse_dates,
   1220     thousands=thousands,
   1221     attrs=attrs,
   1222     encoding=encoding,
   1223     decimal=decimal,
   1224     converters=converters,
   1225     na_values=na_values,
   1226     keep_default_na=keep_default_na,
   1227     displayed_only=displayed_only,
   1228     extract_links=extract_links,
   1229     dtype_backend=dtype_backend,
   1230 )

File ~/anaconda3/lib/python3.11/site-packages/pandas/io/html.py:1001, in _parse(flavor, io, match, attrs, encoding, displayed_only, extract_links, **kwargs)
    999 else:
   1000     assert retained is not None  # for mypy
-> 1001     raise retained
   1003 ret = []
   1004 for table in tables:

File ~/anaconda3/lib/python3.11/site-packages/pandas/io/html.py:981, in _parse(flavor, io, match, attrs, encoding, displayed_only, extract_links, **kwargs)
    978 p = parser(io, compiled_match, attrs, encoding, displayed_only, extract_links)
    980 try:
-> 981     tables = p.parse_tables()
    982 except ValueError as caught:
    983     # if `io` is an io-like object, check if it's seekable
    984     # and try to rewind it before trying the next parser
    985     if hasattr(io, "seekable") and io.seekable():

File ~/anaconda3/lib/python3.11/site-packages/pandas/io/html.py:257, in _HtmlFrameParser.parse_tables(self)
    249 def parse_tables(self):
    250     """
    251     Parse and return all tables from the DOM.
    252 
   (...) 
    255     list of parsed (header, body, footer) tuples from tables.
    256     """
-> 257     tables = self._parse_tables(self._build_doc(), self.match, self.attrs)
    258     return (self._parse_thead_tbody_tfoot(table) for table in tables)

File ~/anaconda3/lib/python3.11/site-packages/pandas/io/html.py:666, in _BeautifulSoupHtml5LibFrameParser._build_doc(self)
    663 def _build_doc(self):
    664     from bs4 import BeautifulSoup
-> 666     bdoc = self._setup_build_doc()
    667     if isinstance(bdoc, bytes) and self.encoding is not None:
    668         udoc = bdoc.decode(self.encoding)

File ~/anaconda3/lib/python3.11/site-packages/pandas/io/html.py:658, in _BeautifulSoupHtml5LibFrameParser._setup_build_doc(self)
    657 def _setup_build_doc(self):
-> 658     raw_text = _read(self.io, self.encoding)
    659     if not raw_text:
    660         raise ValueError(f"No text parsed from document: {self.io}")

File ~/anaconda3/lib/python3.11/site-packages/pandas/io/html.py:155, in _read(obj, encoding)
    149 text: str | bytes
    150 if (
    151     is_url(obj)
    152     or hasattr(obj, "read")
    153     or (isinstance(obj, str) and file_exists(obj))
    154 ):
-> 155     with get_handle(obj, "r", encoding=encoding) as handles:
    156         text = handles.handle.read()
    157 elif isinstance(obj, (str, bytes)):

File ~/anaconda3/lib/python3.11/site-packages/pandas/io/common.py:716, in get_handle(path_or_buf, mode, encoding, compression, memory_map, is_text, errors, storage_options)
    713     codecs.lookup_error(errors)
    715 # open URLs
-> 716 ioargs = _get_filepath_or_buffer(
    717     path_or_buf,
    718     encoding=encoding,
    719     compression=compression,
    720     mode=mode,
    721     storage_options=storage_options,
    722 )
    724 handle = ioargs.filepath_or_buffer
    725 handles: list[BaseBuffer]

File ~/anaconda3/lib/python3.11/site-packages/pandas/io/common.py:368, in _get_filepath_or_buffer(filepath_or_buffer, encoding, compression, mode, storage_options)
    366 # assuming storage_options is to be interpreted as headers
    367 req_info = urllib.request.Request(filepath_or_buffer, headers=storage_options)
-> 368 with urlopen(req_info) as req:
    369     content_encoding = req.headers.get("Content-Encoding", None)
    370     if content_encoding == "gzip":
    371         # Override compression based on Content-Encoding header

File ~/anaconda3/lib/python3.11/site-packages/pandas/io/common.py:270, in urlopen(*args, **kwargs)
    264 """
    265 Lazy-import wrapper for stdlib urlopen, as that imports a big chunk of
    266 the stdlib.
    267 """
    268 import urllib.request
-> 270 return urllib.request.urlopen(*args, **kwargs)

File ~/anaconda3/lib/python3.11/urllib/request.py:216, in urlopen(url, data, timeout, cafile, capath, cadefault, context)
    214 else:
    215     opener = _opener
-> 216 return opener.open(url, data, timeout)

File ~/anaconda3/lib/python3.11/urllib/request.py:519, in OpenerDirector.open(self, fullurl, data, timeout)
    516     req = meth(req)
    518 sys.audit('urllib.Request', req.full_url, req.data, req.headers, req.get_method())
-> 519 response = self._open(req, data)
    521 # post-process response
    522 meth_name = protocol+"_response"

File ~/anaconda3/lib/python3.11/urllib/request.py:536, in OpenerDirector._open(self, req, data)
    533     return result
    535 protocol = req.type
-> 536 result = self._call_chain(self.handle_open, protocol, protocol +
    537                           '_open', req)
    538 if result:
    539     return result

File ~/anaconda3/lib/python3.11/urllib/request.py:496, in OpenerDirector._call_chain(self, chain, kind, meth_name, *args)
    494 for handler in handlers:
    495     func = getattr(handler, meth_name)
-> 496     result = func(*args)
    497     if result is not None:
    498         return result

File ~/anaconda3/lib/python3.11/urllib/request.py:1391, in HTTPSHandler.https_open(self, req)
   1390 def https_open(self, req):
-> 1391     return self.do_open(http.client.HTTPSConnection, req,
   1392         context=self._context, check_hostname=self._check_hostname)

File ~/anaconda3/lib/python3.11/urllib/request.py:1348, in AbstractHTTPHandler.do_open(self, http_class, req, **http_conn_args)
   1346 try:
   1347     try:
-> 1348         h.request(req.get_method(), req.selector, req.data, headers,
   1349                   encode_chunked=req.has_header('Transfer-encoding'))
   1350     except OSError as err: # timeout error
   1351         raise URLError(err)

File ~/anaconda3/lib/python3.11/http/client.py:1286, in HTTPConnection.request(self, method, url, body, headers, encode_chunked)
   1283 def request(self, method, url, body=None, headers={}, *,
   1284             encode_chunked=False):
   1285     """Send a complete request to the server."""
-> 1286     self._send_request(method, url, body, headers, encode_chunked)

File ~/anaconda3/lib/python3.11/http/client.py:1297, in HTTPConnection._send_request(self, method, url, body, headers, encode_chunked)
   1294 if 'accept-encoding' in header_names:
   1295     skips['skip_accept_encoding'] = 1
-> 1297 self.putrequest(method, url, **skips)
   1299 # chunked encoding will happen if HTTP/1.1 is used and either
   1300 # the caller passes encode_chunked=True or the following
   1301 # conditions hold:
   1302 # 1. content-length has not been explicitly set
   1303 # 2. the body is a file or iterable, but not a str or bytes-like
   1304 # 3. Transfer-Encoding has NOT been explicitly set by the caller
   1306 if 'content-length' not in header_names:
   1307     # only chunk body if not explicitly set for backwards
   1308     # compatibility, assuming the client code is already handling the
   1309     # chunking

File ~/anaconda3/lib/python3.11/http/client.py:1135, in HTTPConnection.putrequest(self, method, url, skip_host, skip_accept_encoding)
   1131 self._validate_path(url)
   1133 request = '%s %s %s' % (method, url, self._http_vsn_str)
-> 1135 self._output(self._encode_request(request))
   1137 if self._http_vsn == 11:
   1138     # Issue some standard headers for better HTTP/1.1 compliance
   1140     if not skip_host:
   1141         # this header is issued *only* for HTTP/1.1
   1142         # connections. more specifically, this means it is
   (...) 
   1152         # but the host of the actual URL, not the host of the
   1153         # proxy.

File ~/anaconda3/lib/python3.11/http/client.py:1215, in HTTPConnection._encode_request(self, request)
   1213 def _encode_request(self, request):
   1214     # ASCII also helps prevent CVE-2019-9740.
-> 1215     return request.encode('ascii')

UnicodeEncodeError: 'ascii' codec can't encode character '\xed' in position 16: ordinal not in range(128)

直接对URL字符串使用.encode('UTF-8')无法解决问题,因为错误发生在URL请求阶段,urllib尝试将URL转换为ASCII编码时失败。


解决方案

方法1:对URL进行URL编码

URL中的非ASCII字符(如é)需要转换为URL安全格式,使用urllib.parse.quote处理路径部分:

import pandas as pd
from urllib.parse import quote

# 对包含特殊字符的路径部分进行编码
url_path = quote('Economía_de_México')
full_url = f'https://es.wikipedia.org/wiki/{url_path}'

# 读取表格
df_mx = pd.read_html(full_url, match='Indicadores macroeconómicos, financieros y de bienestar')

方法2:使用requests库先获取页面内容

绕过pandas内置的urllib请求,手动用requests获取页面并指定编码:

import pandas as pd
import requests

url = 'https://es.wikipedia.org/wiki/Economía_de_México'
# 获取页面内容,明确指定UTF-8编码
response = requests.get(url)
response.encoding = 'utf-8'

# 传入已获取的页面文本给read_html
df_mx = pd.read_html(response.text, match='Indicadores macroeconómicos, financieros y de bienestar')

方法3:全局设置UTF-8编码

通过设置环境变量强制Python使用UTF-8编码,适合解决全局编码问题:

import sys
import os

# 设置环境变量
os.environ['PYTHONIOENCODING'] = 'utf-8'
sys.stdout.encoding = 'utf-8'

# 然后执行原代码
import pandas as pd
df_mx = pd.read_html('https://es.wikipedia.org/wiki/Economía_de_México', match='Indicadores macroeconómicos, financieros y de bienestar')

内容的提问来源于stack exchange,提问作者Kevin Duran

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.29 23:27:33