如何在Google Cloud Functions部署运行Scrapy网页爬虫?
如何将Scrapy爬虫部署到Google Cloud Functions并集成到聊天机器人
我第一次做网页爬虫,现在写了一个从指定URL提取地址并和用户输入匹配的Scrapy爬虫,打算集成到聊天机器人里。想知道怎么把它部署到Google Cloud Functions,具体流程是什么,有没有相关的操作指引?
我当前的爬虫代码
from scrapy.spiders import CrawlSpider, Rule from scrapy.linkextractors import LinkExtractor from ..items import DataItem from fuzzywuzzy import fuzz from urllib.parse import urljoin import scrapy class AddressesSpider(scrapy.Spider): name = 'Addresses' allowed_domains = ['find-energy-certificate.service.gov.uk'] postcode = "bh10+4ah" start_urls = ['https://find-energy-certificate.service.gov.uk/find-a-certificate/search-by-postcode?postcode=' + postcode] ## def start_requests(self): ## self.first = input("Please enter the address you would like to match: ") ## yield scrapy.Request(url=self.start_urls[0], callback=self.parse) def parse(self, response): first = input("Please enter the address you would like to match: ") highest_ratios = [] highest_item = None for row in response.xpath('//table[@class="govuk-table"]//tr'): address = row.xpath("normalize-space(.//a[@class='govuk-link']/text())").extract()[0].lower() address = address.rsplit(',', 2)[0] link = row.xpath('.//a[@class="govuk-link"]/@href').extract() details = row.xpath("normalize-space(.//td/following-sibling::td)").extract() ratio = fuzz.token_set_ratio(address, first) item = DataItem() item['link'] = link item['details'] = details item['address'] = address item['ratioresult'] = ratio if len(highest_ratios) < 3: highest_ratios.append(item) elif ratio > min(highest_ratios, key=lambda x: x['ratioresult'])['ratioresult']: highest_ratios.remove(min(highest_ratios, key=lambda x: x['ratioresult'])) highest_ratios.append(item) highest_ratios_100 = [item for item in highest_ratios if item['ratioresult'] == 100] if highest_ratios_100: for item in highest_ratios_100: yield item else: yield max(highest_ratios, key=lambda x: x['ratioresult']) if len(highest_ratios_100) > 1: for i, item in enumerate(highest_ratios_100): print(f"{i+1}: {item['address']}") selected = int(input("Please select the correct address by entering the number corresponding to the address: ")) - 1 selected_item = highest_ratios_100[selected] else: selected_item = highest_ratios_100[0] if highest_ratios_100 else max(highest_ratios, key=lambda x: x['ratioresult']) new_url = selected_item['link'][0] new_url = str(new_url) if new_url: base_url = 'https://find-energy-certificate.service.gov.uk' print(f'Base URL: {base_url}') print(f'New URL: {new_url}') new_url = urljoin(base_url, new_url) print(f'Combined URL: {new_url}') yield scrapy.Request(new_url, callback=self.parse_new_page) def parse_new_page(self, response): Postcode = response.xpath('normalize-space((//p[@class="epc-address govuk-body"]/text())[last()])').extract() Town = response.xpath('normalize-space((//p[@class="epc-address govuk-body"]/text())[last()-1])').extract() First = response.xpath(".//p[@class='epc-address govuk-body']").extract() Type = response.xpath('normalize-space(//dd[1]/text())').extract_first() Walls = response.xpath("//th[contains(text(), 'Wall')]/following-sibling::td[1]/text()").extract() Roof = response.xpath("//th[contains(text(), 'Roof')]/following-sibling::td[1]/text()").extract() Heating = response.xpath("//th[text()='Main heating']/following-sibling::td[1]/text()").extract_first() CurrentScore = response.xpath('//body[1]/div[2]/main[1]/div[1]/div[3]/div[3]/svg[1]/svg[1]/text[1]/text()').re_first("[0-9+]{1,2}") Maxscore = response.xpath('//body[1]/div[2]/main[1]/div[1]/div[3]/div[3]/svg[1]/svg[2]/text[1]/text()').re_first("[0-9+]{2}") Expiry = response.xpath('normalize-space(//b)').extract_first() FloorArea = response.xpath('//dt[contains(text(), "floor area")]/following-sibling::dd/text()').re_first("[0-9+]{2,3}") Steps = response.xpath("//h3[contains(text(),'Step')]/text()").extract() yield { 'Postcode': Postcode, 'Town': Town, 'First': First, 'Type': Type, 'Walls': Walls, 'Roof': Roof, 'Heating': Heating, 'CurrentScore': CurrentScore, 'Maxscore': Maxscore, 'Expiry': Expiry, 'FloorArea': FloorArea, 'Steps': Steps }
配套的items.py文件示例
import scrapy class DataItem(scrapy.Item): link = scrapy.Field() details = scrapy.Field() address = scrapy.Field() ratioresult = scrapy.Field()
部署到Google Cloud Functions的流程
1. 代码改造(关键步骤)
Cloud Functions是无服务器HTTP服务,不能用input()做交互,必须改成通过HTTP请求接收参数,同时要调整Scrapy的运行方式:
- 移除所有
input()和print(),改用HTTP请求参数传递用户输入的地址、邮编 - 使用
scrapy.crawler.CrawlerProcess同步运行爬虫,收集结果后返回HTTP响应 - 把Spider的硬编码参数(比如
postcode)改成从请求中获取
改造后的main.py示例(作为Cloud Functions入口):
from scrapy.crawler import CrawlerProcess import scrapy from fuzzywuzzy import fuzz from urllib.parse import urljoin import json # 定义DataItem(如果不想单独放items.py,可直接写在这里) class DataItem(scrapy.Item): link = scrapy.Field() details = scrapy.Field() address = scrapy.Field() ratioresult = scrapy.Field() class AddressesSpider(scrapy.Spider): name = 'Addresses' allowed_domains = ['find-energy-certificate.service.gov.uk'] def __init__(self, target_address=None, postcode=None, *args, **kwargs): super().__init__(*args, **kwargs) self.target_address = target_address.lower() if target_address else '' self.postcode = postcode.replace(' ', '+') if postcode else 'bh10+4ah' self.start_urls = [f'https://find-energy-certificate.service.gov.uk/find-a-certificate/search-by-postcode?postcode={self.postcode}'] self.final_result = [] def parse(self, response): highest_ratios = [] for row in response.xpath('//table[@class="govuk-table"]//tr'): try: address = row.xpath("normalize-space(.//a[@class='govuk-link']/text())").extract()[0].lower() address = address.rsplit(',', 2)[0] link = row.xpath('.//a[@class="govuk-link"]/@href').extract() details = row.xpath("normalize-space(.//td/following-sibling::td)").extract() ratio = fuzz.token_set_ratio(address, self.target_address) item = DataItem() item['link'] = link item['details'] = details item['address'] = address item['ratioresult'] = ratio if len(highest_ratios) < 3: highest_ratios.append(item) elif ratio > min(highest_ratios, key=lambda x: x['ratioresult'])['ratioresult']: highest_ratios.remove(min(highest_ratios, key=lambda x: x['ratioresult'])) highest_ratios.append(item) except IndexError: continue # 跳过表头行 # 选择最优地址 highest_ratios_100 = [item for item in highest_ratios if item['ratioresult'] == 100] if highest_ratios_100: selected_item = highest_ratios_100[0] if len(highest_ratios_100) ==1 else max(highest_ratios_100, key=lambda x: x['ratioresult']) else: selected_item = max(highest_ratios, key=lambda x: x['ratioresult']) # 爬取详情页 new_url = selected_item['link'][0] full_url = urljoin('https://find-energy-certificate.service.gov.uk', new_url) yield scrapy.Request(full_url, callback=self.parse_new_page) def parse_new_page(self, response): result = { 'Postcode': response.xpath('normalize-space((//p[@class="epc-address govuk-body"]/text())[last()])').extract_first(), 'Town': response.xpath('normalize-space((//p[@class="epc-address govuk-body"]/text())[last()-1])').extract_first(), 'FullAddress': response.xpath("normalize-space(.//p[@class='epc-address govuk-body'])").extract_first(), 'PropertyType': response.xpath('normalize-space(//dd[1]/text())').extract_first(), 'Walls': response.xpath("//th[contains(text(), 'Wall')]/following-sibling::td[1]/text()").extract_first(), 'Roof': response.xpath("//th[contains(text(), 'Roof')]/following-sibling::td[1]/text()").extract_first(), 'Heating': response.xpath("//th[text()='Main heating']/following-sibling::td[1]/text()").extract_first(), 'CurrentScore': response.xpath('//body[1]/div[2]/main[1]/div[1]/div[3]/div[3]/svg[1]/svg[1]/text[1]/text()').re_first("[0-9]{1,2}"), 'Maxscore': response.xpath('//body[1]/div[2]/main[1]/div[1]/div[3]/div[3]/svg[1]/svg[2]/text[1]/text()').re_first("[0-9]{2}"), 'Expiry': response.xpath('normalize-space(//b)').extract_first(), 'FloorArea': response.xpath('//dt[contains(text(), "floor area")]/following-sibling::dd/text()').re_first("[0-9]{2,3}"), 'ImprovementSteps': response.xpath("//h3[contains(text(),'Step')]/text()").extract() } self.final_result.append(result) # Cloud Functions入口函数 def crawl_epc(request): # 获取请求参数(支持GET或POST) request_json = request.get_json(silent=True) request_args = request.args target_address = request_json.get('address') if request_json else request_args.get('address') postcode = request_json.get('postcode') if request_json else request_args.get('postcode') if not target_address or not postcode: return json.dumps({'error': '请提供address和postcode参数'}), 400 # 运行爬虫 process = CrawlerProcess(settings={ 'USER_AGENT': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36', 'LOG_LEVEL': 'ERROR' # 关闭Scrapy日志,避免干扰响应 }) spider = AddressesSpider(target_address=target_address, postcode=postcode) process.crawl(spider) process.start() # 返回结果 return json.dumps(spider.final_result[0]) if spider.final_result else json.dumps({'error': '未找到匹配地址'}), 200
2. 准备依赖文件requirements.txt
scrapy==2.11.0 fuzzywuzzy==0.18.0 python-Levenshtein==0.21.1
3. 本地环境准备
- 安装Google Cloud SDK,完成账号登录和项目配置
- 确保本地Python版本和Cloud Functions选择的版本一致(推荐3.10或3.11)
4. 部署命令
在项目目录下执行:
gcloud functions deploy crawl_epc --runtime python311 --trigger-http --allow-unauthenticated --memory 512MB
参数说明:
crawl_epc:入口函数名--runtime:Python版本--trigger-http:触发方式为HTTP请求--allow-unauthenticated:允许未授权请求(如果聊天机器人有认证,可去掉此参数)--memory:分配的内存(根据爬虫复杂度调整)
5. 测试部署
部署完成后,会得到一个HTTP触发URL,可通过GET请求测试:
https://<REGION>-<PROJECT_ID>.cloudfunctions.net/crawl_epc?address=YourTargetAddress&postcode=YourPostcode
或者用POST请求发送JSON参数:
{ "address": "YourTargetAddress", "postcode": "YourPostcode" }
注意事项
- Cloud Functions单次执行最长时间为9分钟,要确保爬虫逻辑在这个时间内完成
- 目标网站可能有反爬机制,建议设置合理的
USER_AGENT,必要时添加延迟 - 不要频繁触发爬虫,避免被目标网站封禁IP
- 如果需要集成到聊天机器人,直接让机器人调用这个HTTP接口即可
内容的提问来源于stack exchange,提问作者Designer
相关产品推荐
相关产品推荐

