Python爬取Google搜索结果:标题链接显示元素对象而非内容的问题
解决Google搜索结果爬取:标题和链接提取问题
我想用Python获取Google搜索第一页结果的标题和链接,并以列表或字典形式存储。但当前输出中,标题显示为<Element 'h3' class=('LC20lb', 'MBeuO', 'DKV0Md')>,链接显示带有Element 'a' href=前缀的元素对象,而非实际的标题文本和链接地址。我查阅了诸多示例,多数需要订阅付费API,且尝试的BeautifulSoup方法也无效,以下是我目前编写的代码及输出:
import requests import urllib import pandas as pd from requests_html import HTML from requests_html import HTMLSession def get_source(url): """Return the source code for the provided URL. Args: url (string): URL of the page to scrape. Returns: response (object): HTTP response object from requests_html. """ try: session = HTMLSession() response = session.get(url) return response except requests.exceptions.RequestException as e: print(e) def scrape_google(query): query = urllib.parse.quote_plus(query) response = get_source("https://www.google.co.uk/search?q=" + query) links = list(response.html.absolute_links) google_domains = ('https://www.google.', 'https://google.', 'https://webcache.googleusercontent.', 'http://webcache.googleusercontent.', 'https://policies.google.', 'https://support.google.', 'https://maps.google.') for url in links[:]: if url.startswith(google_domains): links.remove(url) return links def get_results(query): query = urllib.parse.quote_plus(query) response = get_source("https://www.google.co.uk/search?q=" + query) return response def parse_results(response): css_identifier_result = ".MjjYud" css_identifier_title = "h3.LC20lb.MBeuO.DKV0Md" css_identifier_link = ".yuRUbf a" results = response.html.find(css_identifier_result) output = [] for result in results: item = { 'title': result.find(css_identifier_title, first=True), 'link': result.find(css_identifier_link, first=True) } output.append(item) return output def google_search(query): response = get_results(query) return parse_results(response) results = google_search("Elon Musk twitter") results
输出:
[{'title': <Element 'h3' class=('LC20lb', 'MBeuO', 'DKV0Md')>, 'link': <Element 'a' href='https://www.theguardian.com/technology/2022/oct/30/twitter-trolls-bombard-platform-after-elon-musk-takeover' data-jsarwt='1' data-usg='AOvVaw3lmUr0p6yL70Nhr1Y5jurH' data-ved='2ahUKEwjA9-TFw4j7AhX3QzABHVyyBwcQFnoECBgQAQ'>}]
解决方案
问题核心是你直接返回了requests_html库的Element对象,而非提取对象中的实际内容。只需修改parse_results函数,获取元素的文本和属性值即可:
- 标题:通过Element对象的
.text属性提取h3标签的文本内容 - 链接:通过Element对象的
.attrs['href']提取a标签的href属性值
修改后的parse_results函数代码:
def parse_results(response): css_identifier_result = ".MjjYud" css_identifier_title = "h3.LC20lb.MBeuO.DKV0Md" css_identifier_link = ".yuRUbf a" results = response.html.find(css_identifier_result) output = [] for result in results: # 提取标题文本和链接地址,添加空值判断避免报错 title_elem = result.find(css_identifier_title, first=True) link_elem = result.find(css_identifier_link, first=True) item = { 'title': title_elem.text if title_elem else None, 'link': link_elem.attrs['href'] if link_elem else None } output.append(item) return output
完整可运行代码
import requests import urllib from requests_html import HTMLSession def get_source(url): try: session = HTMLSession() response = session.get(url) return response except requests.exceptions.RequestException as e: print(e) def get_results(query): query = urllib.parse.quote_plus(query) return get_source("https://www.google.co.uk/search?q=" + query) def parse_results(response): css_identifier_result = ".MjjYud" css_identifier_title = "h3.LC20lb.MBeuO.DKV0Md" css_identifier_link = ".yuRUbf a" results = response.html.find(css_identifier_result) output = [] for result in results: title_elem = result.find(css_identifier_title, first=True) link_elem = result.find(css_identifier_link, first=True) item = { 'title': title_elem.text if title_elem else None, 'link': link_elem.attrs['href'] if link_elem else None } output.append(item) return output def google_search(query): response = get_results(query) return parse_results(response) # 测试运行 results = google_search("Elon Musk twitter") for item in results: print(f"标题: {item['title']}") print(f"链接: {item['link']}\n")
说明
- 添加了空值判断,避免因页面元素结构变化找不到元素导致报错
- 移除了未使用的
pandas和HTML导入,精简代码结构 - 运行后会输出清晰的标题文本和实际可访问的链接地址
内容的提问来源于stack exchange,提问作者kiestuthridge23
相关产品推荐
相关产品推荐

