You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

While循环二次运行触发脚本重启及数据匹配异常求助

解决你的PyQt4网页渲染循环重启问题

嘿,我帮你找到了问题的核心,咱们一步步来修复它!

问题根源

你遇到的脚本第二次运行重启的问题,本质是Qt框架不允许重复创建QApplication实例。在你的Render类里,每次实例化都会新建一个QApplication(sys.argv),但Qt规定一个程序生命周期内只能有一个QApplication对象,第二次创建时就会触发内部冲突,导致脚本直接重启。

修复方案

1. 全局初始化QApplication

把QApplication的创建移到脚本最开头,只执行一次,所有Render实例都复用这个全局对象。

2. 修改Render类的事件循环

原来的app.exec_()会启动一个独立的事件循环,第二次调用会出问题,改成用processEvents()来等待页面加载完成,这样就能在全局事件循环里处理加载逻辑。

3. 资源清理优化

每次使用完Render对象后,手动删除它,避免内存堆积(虽然Python有GC,但显式清理更稳妥)。

修改后的完整代码

import sys, re, pyperclip, requests, csv, fileinput, os
import bs4 as bs
from PyQt4.QtGui import *
from PyQt4.QtCore import *
from PyQt4.QtWebKit import *
from lxml import html
from itertools import zip_longest

# 全局初始化QApplication,整个程序只创建一次
app = QApplication(sys.argv)

#Rendering the Webpage
class Render(QWebPage):
    def __init__(self, url):
        QWebPage.__init__(self)
        self.loadFinished.connect(self._loadFinished)
        self.mainFrame().load(QUrl(url))
        # 用全局app的事件循环等待加载完成,不再启动新的exec_()
        self._loaded = False
        while not self._loaded:
            app.processEvents()

    def _loadFinished(self, result):
        self.frame = self.mainFrame()
        self._loaded = True

def scraping_yellow(inputUrl):
    url = str(inputUrl)
    # 现在创建Render不会再新建QApplication了
    r = Render(url)
    result = r.frame.toHtml()
    #Converting QString to Ascii for lxml to process
    formatted_result = str(result.encode('utf-8'))
    #Next build lxml tree from formatted_result
    tree = html.fromstring(formatted_result)
    treeNoUTF8 = html.fromstring(result)
    #getContent
    name = treeNoUTF8.xpath('//span[@itemprop="name"]/text()')
    street = treeNoUTF8.xpath('//span[@itemprop="streetAddress"]/text()')
    zipcode = treeNoUTF8.xpath('//span[@itemprop="postalCode"]/text()')
    town = treeNoUTF8.xpath('//span[@itemprop="addressLocality"]/text()')
    distance = treeNoUTF8.xpath('//span[@class="teilnehmerentfernung"]/text()')
    phone = treeNoUTF8.xpath('//span[@class="text nummer_ganz"]//span/text()')
    #ListComprehension to make it clean
    street = [str(w).replace('\xa0', ' ') for w in street]
    distance = [str(w).replace('\xa0',' ') for w in distance]
    def concatenate_list_data(list):
        result2= ''
        for element in list:
            result2 += str(element)
        return result2
    #print('String: '+concatenate_list_data(phone))
    suffix = "alltext"
    with open("C://Python//Zwischenablage_{}.txt".format(suffix), "w") as out_f:
        out_f.write(concatenate_list_data(phone))
    fo = open("C://Python//Zwischenablage_{}.txt".format(suffix), 'r').read()
    #patterns to search for
    phoneRegex = re.compile(r"([\(][0-9]{4,5}[\)][\s]?[0-9]{1,10}[\s]?[0-9]{1,10}[\s]?[0-9]{1,3}[-]?[0-9]?)")
    #emailRegex = re.compile(r"([a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+(.[a-zA-Z]{2,4}))",re.VERBOSE)
    #create lists out of matches
    matches = []
    for groups in phoneRegex.findall(fo):
        matches.append(groups)
    #for groups in emailRegex.findall(text):
    # matches.append(groups)
    if len(matches) > 0:
        pyperclip.copy('\n'.join(matches))
    else:
        print('\n'+'No phone numbers or email addresses found.')
    def remove_duplicates(values):
        output = []
        seen = set()
        for value in values:
            # If value has not been encountered yet,
            # ... add it to both list and set.
            if value not in seen:
                output.append(value)
                seen.add(value)
        return output
    matches1 = remove_duplicates(matches)
    #show lists
    print(name,'\n','\n', street,'\n','\n', zipcode,'\n','\n', town, '\n','\n', distance,'\n','\n', matches1)
    #to check if lists are complete
    print( '\n' + 'Prüfsumme - Elemente pro Liste:')
    print(len(name),len(street),len(zipcode),len(town),len(distance),len(matches1))
    #to write a csv
    d = [name, street, zipcode, town, distance, matches1]
    export_data = zip_longest(*d, fillvalue = '')
    with open('C://Python/'+ending+'.csv', 'a', encoding="ISO-8859-1", newline='') as myfile:
        wr = csv.writer(myfile, delimiter=';')
        #wr.writerow(("Name", "Str.", "PLZ", "Ort","Entfernung","Telefon","Email"))
        wr.writerows(export_data)
    #to skip duplicates
    seen = set() # set for fast O(1) amortized lookup
    for line in fileinput.FileInput('C://Python/'+ending+'.csv', inplace=1):
        if line in seen:
            continue # skip duplicate
        seen.add(line)
        print(line), # standard output is now redirected to the file
    #to remove blanks and delete the temporary file
    with open('C://Python/'+ending+'.csv') as input, open('C://Python/'+ending+' noblank.csv', 'w') as output:
        non_blank = (line for line in input if line.strip())
        output.writelines(non_blank)
    os.remove('C://Python/'+ending+'.csv')
    print(link1)
    # 显式清理Render对象,释放资源
    del r

################# Start - Programm #################
ending = input('Bitte geben Sie den gewünschten Dateinamen ein: ')
link = 'https://www.gelbeseiten.de/'+input('Bitte geben Sie das gesuchte Gewerk ein!(ohne Umlaute)')+'/bergheim,,,,,umkreis-50000/s'
i=0
n = 4 #n = int(input('Bitte gib die Seitenanzahl ein: '))
while i<=n :
    link1 = link+str(1+i)
    print('Link: '+link1)
    scraping_yellow(link1)
    i = i + 1
else:
    #to confirm printing
    print('>>>> csv-Datei wurde beschrieben! <<<<')

关于数据匹配错误的额外建议

你提到输入Test2和Fliesenleger时,手机号和公司名称不对应,这是因为你现在是全局提取所有名称和所有号码,没有关联它们的父容器。建议调整XPath逻辑:

  • 先定位每个公司的卡片节点(比如//div[contains(@class, "teilnehmer")],具体要看页面实际结构)
  • 然后遍历每个卡片,在卡片内部分别提取名称、地址、号码等信息
  • 这样每个公司的数据都会对应起来,不会出现错位

另外提一句,PyQt4已经停止维护很多年了,如果你后续要扩展功能,建议迁移到PyQt5或者PySide2,生态更活跃,文档也更全。

内容的提问来源于stack exchange,提问作者DanielHe

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.15 07:28:54