Python爬虫:如何获取PokémonDB中宝可梦技能的URL、名称与效果描述
解决宝可梦技能信息爬取问题
要获取宝可梦的技能URL、名称及效果描述,你需要在现有Scrapy代码基础上添加技能链接提取、详情页跳转和效果抓取的逻辑,以下是修改后的完整实现:
import scrapy class PokeSpider(scrapy.Spider): name = 'pokespider' start_urls = ['https://pokemondb.net/pokedex/all'] def parse(self, response): # 遍历所有宝可梦条目(原代码仅抓取第一个,此处改为全量爬取) for linha in response.css('table#pokedex > tbody > tr'): link = linha.css("td:nth-child(2) > a::attr(href)") yield response.follow(link.get(), self.parser_pokemon) def parser_pokemon(self, response): nome = response.css('h1::text').get() id = response.css('table.vitals-table > tbody > tr:nth-child(1) > td > strong::text').get() tamanho = response.css('table.vitals-table > tbody > tr:nth-child(4) > td::text').get() peso = response.css('table.vitals-table > tbody > tr:nth-child(5) > td::text').get() url_pokemon = response.url tipos = response.css('table.vitals-table tbody tr:nth-child(2) td a::text').getall()[:2] evolucoes = [] evolucoes_possiveis = response.css('#main div.infocard-list-evo div span.infocard-lg-data.text-muted') for evolucao in evolucoes_possiveis: nome_evolucao = evolucao.css('a::text').get() id_evolucao = evolucao.css('small:nth-child(1)::text').get() url_evolucao = evolucao.css('a::attr(href)').get() url_evolucao_completinha = response.urljoin(url_evolucao) # 使用urljoin避免URL拼接错误 evolucoes.append({ "nome_evolucao": nome_evolucao, "id_evolucao": id_evolucao, "url_evolucao": url_evolucao_completinha }) # 提取当前宝可梦的所有技能行 move_rows = response.css('table.moves-table tbody tr') for row in move_rows: move_name = row.css('td:nth-child(2) a::text').get() move_relative_url = row.css('td:nth-child(2) a::attr(href)').get() if move_name and move_relative_url: full_move_url = response.urljoin(move_relative_url) # 组装包含宝可梦基础数据的临时条目 pokemon_item = { "nome": nome, "id": id, "tamanho": tamanho, "peso": peso, "url_pokemon": url_pokemon, "tipos": tipos, "evolucoes": evolucoes, "move_name": move_name, "move_url": full_move_url } # 请求技能详情页,通过meta传递宝可梦数据 yield scrapy.Request( full_move_url, callback=self.parse_move_effect, meta={'pokemon_item': pokemon_item} ) def parse_move_effect(self, response): pokemon_item = response.meta['pokemon_item'] # 抓取技能效果描述(适配页面不同结构) effect_paragraphs = response.css('h2:contains("Effect") ~ p::text').getall() if not effect_paragraphs: effect_paragraphs = response.css('div.effect p::text').getall() move_effect = '\n'.join([p.strip() for p in effect_paragraphs if p.strip()]) pokemon_item['move_effect'] = move_effect if move_effect else None yield pokemon_item
核心修改说明:
- 全量宝可梦爬取:将原代码仅抓取首个宝可梦的逻辑改为遍历所有宝可梦条目
- 技能链接提取:从宝可梦详情页的技能表格中提取每个技能的名称和详情页URL
- 异步请求技能详情:通过
scrapy.Request跳转至技能详情页,并用meta参数传递宝可梦基础数据 - 效果描述抓取:新增
parse_move_effect回调函数,从技能详情页提取效果文本并合并到最终条目
注:此代码会为每个技能生成一条包含对应宝可梦信息的独立条目。若需将单个宝可梦的所有技能合并为一条条目,需使用异步数据聚合逻辑(如基于Twisted Deferred的回调管理),属于进阶用法。
内容的提问来源于stack exchange,提问作者Uemura
相关产品推荐
相关产品推荐

