如何用Scrapy爬取页面内链?现有代码存重复与遗漏问题求助
Scrapy爬取内链数据问题修复
问题描述
我使用Scrapy爬取页面https://icsstrive.com/incident/lockbit-ransomware-attack-significantly-impacts-owens-group-operations/中的victims、malware、threat_source三类内链内容,编写代码后出现两个问题:
- victim内容重复输出
- 有时会遗漏malware或threat_source的数据
原代码如下:
# importing the scrapy module import random import scrapy import logging from scrapy.utils.log import configure_logging from pycti import OpenCTIApiClient import stix2 from pycti import ( Identity, ThreatActor, Malware, Location, StixCoreRelationship, Report, ) import json class icsstriveSpider(scrapy.Spider): stix_objects=[] name = "icsstrive" start_urls = ['https://icsstrive.com/'] baseUrl="https://icsstrive.com" pages = None def parse(self, response, **kwargs): links = response.css('div.search-r-title a::attr(href)').getall() for i in range(len(links)): links[i] = links[i] yield from response.follow_all(links, self.parse_icsstrive) if self.pages is None: self.pages=response.xpath('//a[@class="wpv-filter-pagination-link js-wpv-pagination-link page-link"]/@href').getall() if len(self.pages) >0: url=self.pages[0] self.pages.remove(url) yield response.follow(self.baseUrl+url, self.parse,dont_filter=True) def parse_icsstrive(self, response, **kwargs): title = "" published = "" type = "" summary = "" incident_Date = "" location = "" estimated_Cost = "" victims_url = "" victim_title = "" malwares_urls = "" threat_source_urls = "" references_name = "" references_url = "" industries = "" impacts = "" title=response.xpath('//h1[@class="entry-title"]/text()').get() published=response.xpath('//p[@class="et_pb_title_meta_container"]/span/text()').get() type=response.xpath('//div[@class="et_pb_text_inner"]/text()').get() summary=response.xpath('//div[@class="et_pb_text_inner"]/p/text()').get() incident_Date = response.xpath('//h3[text()="Incident Date"]/following-sibling::*//text()').get() location = response.xpath('//h3[text()="Location"]/following-sibling::p/a/text()').get() estimated_Cost = response.xpath('//h3[text()="Estimated Cost"]/following-sibling::p/text()').get() victims_url = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Victims"]/following-sibling::div/ul/li/a/@href').getall() malwares_urls = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Type of Malware"]/following-sibling::div/ul/li/a/@href').getall() threat_source_urls = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Threat Source"]/following-sibling::div/ul/li/a/@href').getall() references_name = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="References"]/following-sibling::div/ul/li/a/text()').getall() references_url = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="References"]/following-sibling::div/ul/li/a/@href').getall() industries = response.xpath('//h3[text()="Industries"]/following-sibling::p/a/text()').get() impacts = response.xpath('//h3[text()="Impacts"]/following-sibling::*//text()').get() item = { "title": title, "published": published, "type": type, "summary": summary, "incident_Date": incident_Date, "estimated_Cost": estimated_Cost, "references": ",".join(references_name), "industries": industries, } if location is not None: item["location"]= location.replace("'", '"') if impacts is not None: item["impacts"]= impacts.replace("'", '"') # Extract malware URLs if len(victims_url) > 0: for url in victims_url: request= scrapy.Request(url + "?dummy=" + str(random.random()),callback=self.parse_victims,dont_filter=True,meta={'item': item, 'malwares_urls': malwares_urls, 'threat_source_urls':threat_source_urls}) request.meta['dont_cache'] = True yield request else: yield item def parse_victims(self, response, **kwargs): victim_title = "" victim_published = "" victim_des = "" victim_title=response.xpath('//h1[@class="entry-title"]/text()').get() victim_des=response.xpath('//div[@class="et_pb_text_inner"]/p/text()').get() victim_published = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Incidents"]/following-sibling::div/ul/li/strong/text()').getall() item = response.meta['item'] malwares_urls = response.meta['malwares_urls'] threat_source_urls = response.meta['threat_source_urls'] item["victim_title"] = victim_title item["victim_des"] = victim_des item["victim_url"] = response.url if victim_published: item["victim_published"] = victim_published[0] if item["title"]=="Chinese Identified Hackers Targeting Hawaii Water Utilities and unidentified Oil & Gas Pipeline in US": print(item) if len(malwares_urls) > 0: for malware_url in malwares_urls: request= scrapy.Request(malware_url+ "?dummy=" + str(random.random()), callback=self.parse_malware,dont_filter=True, meta={'item': item, 'threat_source_urls':threat_source_urls}) request.meta['dont_cache'] = True yield request elif len(threat_source_urls) > 0: for threat_source_url in threat_source_urls: request= scrapy.Request(threat_source_url+ "?dummy=" + str(random.random()), callback=self.parse_threat_source,dont_filter=True,meta={'item': item}) request.meta['dont_cache'] = True yield request else: yield item def parse_malware(self, response, **kwargs): malware_title = "" malware_published = "" malware_des = "" malware_title=response.xpath('//h1[@class="entry-title"]/text()').get() malware_des=response.xpath('//div[@class="et_pb_text_inner"]/p/text()').get() malware_published = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Incidents"]/following-sibling::div/ul/li/strong/text()').getall() item = response.meta['item'] threat_source_urls = response.meta['threat_source_urls'] item["malware_title"] = malware_title item["malware_des"] = malware_des if malware_published: item["malware_published"] = malware_published[0] if len(threat_source_urls) > 0: for threat_source_url in threat_source_urls: request= scrapy.Request(threat_source_url+ "?dummy=" + str(random.random()), callback=self.parse_threat_source,dont_filter=True,meta={'item': item}) request.meta['dont_cache'] = True yield request else: yield item def parse_threat_source(self, response, **kwargs): threat_source_title = "" threat_source_published = "" threat_source_des = "" threat_source_title=response.xpath('//h1[@class="entry-title"]/text()').get() threat_source_des=response.xpath('//div[@class="et_pb_text_inner"]/p/text()').get() threat_source_published = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Incidents"]/following-sibling::div/ul/li/strong/text()').getall() item = response.meta['item'] item["threat_source_title"] = threat_source_title item["threat_source_des"] = threat_source_des if item["title"]=="Chinese Identified Hackers Targeting Hawaii Water Utilities and unidentified Oil & Gas Pipeline in US": print(item) if threat_source_published: item["threat_source_published"] = threat_source_published[0] yield item
问题原因
- Victim内容重复:传递给
parse_victims的是同一个item字典的引用,多个victim请求共享同一个item对象,后续对item的修改会覆盖之前的数据,最终导致所有victim条目输出相同内容。 - 遗漏malware/threat_source:当前逻辑采用串行分支判断(
if...elif...else),如果存在malware URL,就只会处理malware,不会触发threat_source的请求;且如果有多个同类型内链,后续请求会覆盖item中的对应字段,导致只保留最后一个内链的数据。
修复方案
1. 复制item副本,避免引用传递
每次生成请求时,传递item的副本(比如用item.copy()),确保每个请求的item都是独立的对象。
2. 重构内链处理逻辑,确保所有内链都被处理
取消串行依赖,改为统一处理剩余内链,确保只要存在对应内链就会触发请求,同时每个请求携带独立的item副本。
修复后的代码
# importing the scrapy module import random import scrapy import logging from scrapy.utils.log import configure_logging from pycti import OpenCTIApiClient import stix2 from pycti import ( Identity, ThreatActor, Malware, Location, StixCoreRelationship, Report, ) import json class icsstriveSpider(scrapy.Spider): stix_objects=[] name = "icsstrive" start_urls = ['https://icsstrive.com/'] baseUrl="https://icsstrive.com" pages = None def parse(self, response, **kwargs): links = response.css('div.search-r-title a::attr(href)').getall() yield from response.follow_all(links, self.parse_icsstrive) if self.pages is None: self.pages=response.xpath('//a[@class="wpv-filter-pagination-link js-wpv-pagination-link page-link"]/@href').getall() if len(self.pages) >0: url=self.pages[0] self.pages.remove(url) yield response.follow(self.baseUrl+url, self.parse,dont_filter=True) def parse_icsstrive(self, response, **kwargs): title = response.xpath('//h1[@class="entry-title"]/text()').get() published = response.xpath('//p[@class="et_pb_title_meta_container"]/span/text()').get() type = response.xpath('//div[@class="et_pb_text_inner"]/text()').get() summary = response.xpath('//div[@class="et_pb_text_inner"]/p/text()').get() incident_Date = response.xpath('//h3[text()="Incident Date"]/following-sibling::*//text()').get() location = response.xpath('//h3[text()="Location"]/following-sibling::p/a/text()').get() estimated_Cost = response.xpath('//h3[text()="Estimated Cost"]/following-sibling::p/text()').get() victims_url = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Victims"]/following-sibling::div/ul/li/a/@href').getall() malwares_urls = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Type of Malware"]/following-sibling::div/ul/li/a/@href').getall() threat_source_urls = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Threat Source"]/following-sibling::div/ul/li/a/@href').getall() references_name = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="References"]/following-sibling::div/ul/li/a/text()').getall() industries = response.xpath('//h3[text()="Industries"]/following-sibling::p/a/text()').get() impacts = response.xpath('//h3[text()="Impacts"]/following-sibling::*//text()').get() base_item = { "title": title, "published": published, "type": type, "summary": summary, "incident_Date": incident_Date, "estimated_Cost": estimated_Cost, "references": ",".join(references_name), "industries": industries, } if location is not None: base_item["location"]= location.replace("'", '"') if impacts is not None: base_item["impacts"]= impacts.replace("'", '"') # 处理所有内链,每个内链请求携带独立的item副本 if victims_url: for url in victims_url: item_copy = base_item.copy() request = scrapy.Request( url + "?dummy=" + str(random.random()), callback=self.parse_victims, dont_filter=True, meta={ 'item': item_copy, 'malwares_urls': malwares_urls, 'threat_source_urls': threat_source_urls } ) request.meta['dont_cache'] = True yield request else: # 没有victim时,直接处理剩余内链 yield from self.process_remaining_links(base_item, malwares_urls, threat_source_urls) def process_remaining_links(self, item, malwares_urls, threat_source_urls): """统一处理剩余的malware和threat_source链接""" if malwares_urls: for url in malwares_urls: item_copy = item.copy() request = scrapy.Request( url + "?dummy=" + str(random.random()), callback=self.parse_malware, dont_filter=True, meta={ 'item': item_copy, 'threat_source_urls': threat_source_urls } ) request.meta['dont_cache'] = True yield request elif threat_source_urls: for url in threat_source_urls: item_copy = item.copy() request = scrapy.Request( url + "?dummy=" + str(random.random()), callback=self.parse_threat_source, dont_filter=True, meta={'item': item_copy} ) request.meta['dont_cache'] = True yield request else: yield item def parse_victims(self, response, **kwargs): victim_title = response.xpath('//h1[@class="entry-title"]/text()').get() victim_des = response.xpath('//div[@class="et_pb_text_inner"]/p/text()').get() victim_published = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Incidents"]/following-sibling::div/ul/li/strong/text()').getall() item = response.meta['item'] malwares_urls = response.meta['malwares_urls'] threat_source_urls = response.meta['threat_source_urls'] # 写入victim数据 item["victim_title"] = victim_title item["victim_des"] = victim_des item["victim_url"] = response.url if victim_published: item["victim_published"] = victim_published[0] # 处理剩余内链 yield from self.process_remaining_links(item, malwares_urls, threat_source_urls) def parse_malware(self, response, **kwargs): malware_title = response.xpath('//h1[@class="entry-title"]/text()').get() malware_des = response.xpath('//div[@class="et_pb_text_inner"]/p/text()').get() malware_published = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Incidents"]/following-sibling::div/ul/li/strong/text()').getall() item = response.meta['item']
相关产品推荐
相关产品推荐

