You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用Scrapy爬取页面内链?现有代码存重复与遗漏问题求助

Scrapy爬取内链数据问题修复

问题描述

我使用Scrapy爬取页面https://icsstrive.com/incident/lockbit-ransomware-attack-significantly-impacts-owens-group-operations/中的victims、malware、threat_source三类内链内容,编写代码后出现两个问题:

  • victim内容重复输出
  • 有时会遗漏malware或threat_source的数据

原代码如下:

# importing the scrapy module
import random
import scrapy
import logging
from scrapy.utils.log import configure_logging
from pycti import OpenCTIApiClient
import stix2
from pycti import (
    Identity,
    ThreatActor,
    Malware,
    Location,
    StixCoreRelationship,
    Report,
)
import json

class icsstriveSpider(scrapy.Spider):

    stix_objects=[]

    name = "icsstrive"
    start_urls = ['https://icsstrive.com/']
    baseUrl="https://icsstrive.com"
    pages = None

    def parse(self, response, **kwargs):
        links = response.css('div.search-r-title a::attr(href)').getall()
        for i in range(len(links)):
            links[i] = links[i]
        yield from response.follow_all(links, self.parse_icsstrive)
      
        if self.pages is None:
            self.pages=response.xpath('//a[@class="wpv-filter-pagination-link js-wpv-pagination-link page-link"]/@href').getall()
        if len(self.pages) >0:
            url=self.pages[0]
            self.pages.remove(url)
            yield response.follow(self.baseUrl+url, self.parse,dont_filter=True)
       
        
    def parse_icsstrive(self, response, **kwargs):
        title = ""
        published = ""
        type = ""
        summary = ""
        incident_Date = ""
        location = ""
        estimated_Cost = ""
        victims_url = ""
        victim_title = ""
        malwares_urls = ""
        threat_source_urls = ""
        references_name = ""
        references_url = ""
        industries = ""
        impacts = ""
        title=response.xpath('//h1[@class="entry-title"]/text()').get()
        published=response.xpath('//p[@class="et_pb_title_meta_container"]/span/text()').get()
        type=response.xpath('//div[@class="et_pb_text_inner"]/text()').get()
        summary=response.xpath('//div[@class="et_pb_text_inner"]/p/text()').get()
        incident_Date = response.xpath('//h3[text()="Incident Date"]/following-sibling::*//text()').get()
        location = response.xpath('//h3[text()="Location"]/following-sibling::p/a/text()').get()
        estimated_Cost = response.xpath('//h3[text()="Estimated Cost"]/following-sibling::p/text()').get()
        victims_url = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Victims"]/following-sibling::div/ul/li/a/@href').getall()
        malwares_urls = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Type of Malware"]/following-sibling::div/ul/li/a/@href').getall()
        threat_source_urls = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Threat Source"]/following-sibling::div/ul/li/a/@href').getall()
        references_name = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="References"]/following-sibling::div/ul/li/a/text()').getall()
        references_url = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="References"]/following-sibling::div/ul/li/a/@href').getall()
        industries = response.xpath('//h3[text()="Industries"]/following-sibling::p/a/text()').get()
        impacts = response.xpath('//h3[text()="Impacts"]/following-sibling::*//text()').get()

        item = {
            "title": title,
            "published": published,
            "type": type,
            "summary": summary,
            "incident_Date": incident_Date,
            "estimated_Cost": estimated_Cost,
            "references": ",".join(references_name),
            "industries": industries,
        }
        if location is not None:
            item["location"]= location.replace("'", '"')
        if impacts is not None:
            item["impacts"]= impacts.replace("'", '"')
       
        # Extract malware URLs
        if len(victims_url) > 0: 
            for url in victims_url:
                request= scrapy.Request(url + "?dummy=" + str(random.random()),callback=self.parse_victims,dont_filter=True,meta={'item': item, 'malwares_urls': malwares_urls, 'threat_source_urls':threat_source_urls})
                request.meta['dont_cache'] = True
                yield request
        else:
            yield item  

    def parse_victims(self, response, **kwargs):
        victim_title = ""
        victim_published = ""
        victim_des = ""
        victim_title=response.xpath('//h1[@class="entry-title"]/text()').get()
        victim_des=response.xpath('//div[@class="et_pb_text_inner"]/p/text()').get()
        victim_published = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Incidents"]/following-sibling::div/ul/li/strong/text()').getall()
    
        item = response.meta['item']
       
        malwares_urls = response.meta['malwares_urls']
        threat_source_urls = response.meta['threat_source_urls']
        item["victim_title"] = victim_title
        item["victim_des"] = victim_des
        item["victim_url"] = response.url
        if victim_published:
            item["victim_published"] = victim_published[0]
        if item["title"]=="Chinese Identified Hackers Targeting Hawaii Water Utilities and unidentified Oil & Gas Pipeline in US":
             print(item)
       
        if len(malwares_urls) > 0:
            for malware_url in malwares_urls:
                request= scrapy.Request(malware_url+ "?dummy=" + str(random.random()), callback=self.parse_malware,dont_filter=True, meta={'item': item, 'threat_source_urls':threat_source_urls})
                request.meta['dont_cache'] = True
                yield request
        elif len(threat_source_urls) > 0:
            for threat_source_url in threat_source_urls:
                request= scrapy.Request(threat_source_url+ "?dummy=" + str(random.random()), callback=self.parse_threat_source,dont_filter=True,meta={'item': item})
                request.meta['dont_cache'] = True
                yield request
        else:
            yield item   
    def parse_malware(self, response, **kwargs):
        malware_title = ""
        malware_published = ""
        malware_des = ""
        malware_title=response.xpath('//h1[@class="entry-title"]/text()').get()
        malware_des=response.xpath('//div[@class="et_pb_text_inner"]/p/text()').get()
        malware_published = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Incidents"]/following-sibling::div/ul/li/strong/text()').getall()
        item = response.meta['item']
        threat_source_urls = response.meta['threat_source_urls']
        item["malware_title"] = malware_title
        item["malware_des"] = malware_des
        
        if malware_published:
            item["malware_published"] = malware_published[0]
        if len(threat_source_urls) > 0:
            for threat_source_url in threat_source_urls:
                request= scrapy.Request(threat_source_url+ "?dummy=" + str(random.random()), callback=self.parse_threat_source,dont_filter=True,meta={'item': item})
                request.meta['dont_cache'] = True
                yield request
        else:
            yield item   

    def parse_threat_source(self, response, **kwargs):
        threat_source_title = ""
        threat_source_published = ""
        threat_source_des = ""
        threat_source_title=response.xpath('//h1[@class="entry-title"]/text()').get()
        threat_source_des=response.xpath('//div[@class="et_pb_text_inner"]/p/text()').get()
        threat_source_published = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Incidents"]/following-sibling::div/ul/li/strong/text()').getall()
        item = response.meta['item']
        item["threat_source_title"] = threat_source_title
        item["threat_source_des"] = threat_source_des
        if item["title"]=="Chinese Identified Hackers Targeting Hawaii Water Utilities and unidentified Oil & Gas Pipeline in US":
             print(item)
        if threat_source_published:
            item["threat_source_published"] = threat_source_published[0]
        yield item

问题原因

  1. Victim内容重复:传递给parse_victims的是同一个item字典的引用,多个victim请求共享同一个item对象,后续对item的修改会覆盖之前的数据,最终导致所有victim条目输出相同内容。
  2. 遗漏malware/threat_source:当前逻辑采用串行分支判断(if...elif...else),如果存在malware URL,就只会处理malware,不会触发threat_source的请求;且如果有多个同类型内链,后续请求会覆盖item中的对应字段,导致只保留最后一个内链的数据。

修复方案

1. 复制item副本,避免引用传递

每次生成请求时,传递item的副本(比如用item.copy()),确保每个请求的item都是独立的对象。

2. 重构内链处理逻辑,确保所有内链都被处理

取消串行依赖,改为统一处理剩余内链,确保只要存在对应内链就会触发请求,同时每个请求携带独立的item副本。

修复后的代码

# importing the scrapy module
import random
import scrapy
import logging
from scrapy.utils.log import configure_logging
from pycti import OpenCTIApiClient
import stix2
from pycti import (
    Identity,
    ThreatActor,
    Malware,
    Location,
    StixCoreRelationship,
    Report,
)
import json

class icsstriveSpider(scrapy.Spider):

    stix_objects=[]

    name = "icsstrive"
    start_urls = ['https://icsstrive.com/']
    baseUrl="https://icsstrive.com"
    pages = None

    def parse(self, response, **kwargs):
        links = response.css('div.search-r-title a::attr(href)').getall()
        yield from response.follow_all(links, self.parse_icsstrive)
      
        if self.pages is None:
            self.pages=response.xpath('//a[@class="wpv-filter-pagination-link js-wpv-pagination-link page-link"]/@href').getall()
        if len(self.pages) >0:
            url=self.pages[0]
            self.pages.remove(url)
            yield response.follow(self.baseUrl+url, self.parse,dont_filter=True)
       
        
    def parse_icsstrive(self, response, **kwargs):
        title = response.xpath('//h1[@class="entry-title"]/text()').get()
        published = response.xpath('//p[@class="et_pb_title_meta_container"]/span/text()').get()
        type = response.xpath('//div[@class="et_pb_text_inner"]/text()').get()
        summary = response.xpath('//div[@class="et_pb_text_inner"]/p/text()').get()
        incident_Date = response.xpath('//h3[text()="Incident Date"]/following-sibling::*//text()').get()
        location = response.xpath('//h3[text()="Location"]/following-sibling::p/a/text()').get()
        estimated_Cost = response.xpath('//h3[text()="Estimated Cost"]/following-sibling::p/text()').get()
        victims_url = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Victims"]/following-sibling::div/ul/li/a/@href').getall()
        malwares_urls = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Type of Malware"]/following-sibling::div/ul/li/a/@href').getall()
        threat_source_urls = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Threat Source"]/following-sibling::div/ul/li/a/@href').getall()
        references_name = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="References"]/following-sibling::div/ul/li/a/text()').getall()
        industries = response.xpath('//h3[text()="Industries"]/following-sibling::p/a/text()').get()
        impacts = response.xpath('//h3[text()="Impacts"]/following-sibling::*//text()').get()

        base_item = {
            "title": title,
            "published": published,
            "type": type,
            "summary": summary,
            "incident_Date": incident_Date,
            "estimated_Cost": estimated_Cost,
            "references": ",".join(references_name),
            "industries": industries,
        }
        if location is not None:
            base_item["location"]= location.replace("'", '"')
        if impacts is not None:
            base_item["impacts"]= impacts.replace("'", '"')
       
        # 处理所有内链,每个内链请求携带独立的item副本
        if victims_url:
            for url in victims_url:
                item_copy = base_item.copy()
                request = scrapy.Request(
                    url + "?dummy=" + str(random.random()),
                    callback=self.parse_victims,
                    dont_filter=True,
                    meta={
                        'item': item_copy,
                        'malwares_urls': malwares_urls,
                        'threat_source_urls': threat_source_urls
                    }
                )
                request.meta['dont_cache'] = True
                yield request
        else:
            # 没有victim时,直接处理剩余内链
            yield from self.process_remaining_links(base_item, malwares_urls, threat_source_urls)

    def process_remaining_links(self, item, malwares_urls, threat_source_urls):
        """统一处理剩余的malware和threat_source链接"""
        if malwares_urls:
            for url in malwares_urls:
                item_copy = item.copy()
                request = scrapy.Request(
                    url + "?dummy=" + str(random.random()),
                    callback=self.parse_malware,
                    dont_filter=True,
                    meta={
                        'item': item_copy,
                        'threat_source_urls': threat_source_urls
                    }
                )
                request.meta['dont_cache'] = True
                yield request
        elif threat_source_urls:
            for url in threat_source_urls:
                item_copy = item.copy()
                request = scrapy.Request(
                    url + "?dummy=" + str(random.random()),
                    callback=self.parse_threat_source,
                    dont_filter=True,
                    meta={'item': item_copy}
                )
                request.meta['dont_cache'] = True
                yield request
        else:
            yield item

    def parse_victims(self, response, **kwargs):
        victim_title = response.xpath('//h1[@class="entry-title"]/text()').get()
        victim_des = response.xpath('//div[@class="et_pb_text_inner"]/p/text()').get()
        victim_published = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Incidents"]/following-sibling::div/ul/li/strong/text()').getall()
    
        item = response.meta['item']
        malwares_urls = response.meta['malwares_urls']
        threat_source_urls = response.meta['threat_source_urls']
        
        # 写入victim数据
        item["victim_title"] = victim_title
        item["victim_des"] = victim_des
        item["victim_url"] = response.url
        if victim_published:
            item["victim_published"] = victim_published[0]
       
        # 处理剩余内链
        yield from self.process_remaining_links(item, malwares_urls, threat_source_urls)

    def parse_malware(self, response, **kwargs):
        malware_title = response.xpath('//h1[@class="entry-title"]/text()').get()
        malware_des = response.xpath('//div[@class="et_pb_text_inner"]/p/text()').get()
        malware_published = response.xpath('//div[@class="et_pb_text_inner"]/h3[text()="Incidents"]/following-sibling::div/ul/li/strong/text()').getall()
        
        item = response.meta['item']
相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.03 04:33:46