You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Pandas KeyError 'date'排查:网页爬虫代码突然失效求助

问题:BD25.eu爬虫突然抛出KeyError: 'date'异常

此前可正常运行的BD25.eu网页爬虫,近期未修改代码、重装Python及所有依赖库后,仍抛出KeyError: 'date'异常。该爬虫用于爬取指定日期之后的站点条目,报错发生在执行语句:

final_df['Upload_date'] = pd.to_datetime(final_df['date'],dayfirst=True)

完整代码

import shutil
from selenium import webdriver
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.common.by import By
import pandas as pd
from bs4 import BeautifulSoup
import warnings
import json
import shutil
import os
from difflib import get_close_matches
import time
import imdb
import datetime

# creating an instance of the IMDB()
##ia = imdb.IMDb()
ia = imdb.Cinemagoer()
# Using the Search movie method

warnings.filterwarnings("ignore")

def get_movie_name(name):
    try:
        items = []
        while len(items) == 0 and len(name) > 0:
            time.sleep(2)
            items = ia.search_movie(name)
            if len(items) == 0:
                name = name.split(' ')
                name = name[:-1]
                name = ' '.join(name)

        closest_match = get_close_matches(name, [str(i) for i in items])
        if len(closest_match) ==0:
            return False
        else:
            return closest_match[0]
    except Exception as e:
        print(e)
        return False

def read_inputs():
    f = open('config.json')
    data = json.load(f)
    return data
    


data = read_inputs()

download_folder = 'C:\\Users\\jerem\\Downloads'

date  = data["date"]
username = data["username"]   
password = data["password"]   

# initialize the Chrome driver
options = webdriver.ChromeOptions() 
options.add_experimental_option("excludeSwitches", ["enable-logging",'--headless','--disable-gpu'])
driver = webdriver.Chrome(options = options,executable_path = "chromedriver")
driver.get("http://www.bd25.eu/index.php")
#username
##driver.find_element_by_xpath("//input[@class='lista'][@type='text']").send_keys(username)
driver.find_element(By.XPATH, "//input[@type='text']").send_keys(username)
##driver.find_element('type',"text").send_keys(username)


#password
driver.find_element(By.XPATH, "//input[@type='password']").send_keys(password)
#submit
driver.find_element(By.XPATH, "//input[@type='submit']").click()
#list of nzbs
driver.find_element(By.LINK_TEXT, 'List of NZBs').click()
movie_df,final_df = pd.DataFrame(),pd.DataFrame()
get_url = driver.current_url
for page in range(1,int(data['page_number_max'])):
    try:
        if page > 1:
            add_on = '&order=3&by=2&pages='+str(page)
            driver.get(get_url+add_on)
        body = driver.find_element(By.XPATH, "/html/body/table/tbody/tr")
        all_text = body.text
        y = all_text.split('\n')
        
        for i in range(330,430):
            movie_df = movie_df.append({'text':y[i]},ignore_index=True)
            value_list = y[i].split(' ')
            name = value_list[1:-7]
            name = ' '.join(name)
            if value_list[-3] != 'Upload':
                final_df = final_df.append({'date':value_list[-3],
                                        'name':name},ignore_index=True)
            value_list  = []
        print(str(page)+' page is done')
        
    except:
        break


#final_df = final_df[1:]
date = datetime.datetime.strptime(date,'%d/%m/%Y')
final_df['Upload_date'] = pd.to_datetime(final_df['date'],dayfirst=True)
final_df = final_df.loc[final_df['Upload_date'] > date]

for i in range(len(final_df)):
    try:
        driver.find_element(By.LINK_TEXT, 'List of NZBs').click()
        driver.find_element(By.ID, "searchinput").send_keys(final_df['name'].iloc[i])
        driver.find_element(By.XPATH, "//input[@type='submit']").click()
        driver.find_element(By.XPATH, "//img[@src='images/download.gif']").click()
        time.sleep(2)
        space_list = final_df['name'].iloc[i].split(' ')
        foldername = '.'.join(space_list)
        try:
            list_of_folder = [x for x in os.listdir(download_folder)]
            closest_match = get_close_matches(foldername, list_of_folder)
            src = closest_match[0]
        except:
            time.sleep(2)
            print('waiting for download')
            list_of_folder = [x for x in os.listdir(download_folder)]
            closest_match = get_close_matches(foldername, list_of_folder)
            src = closest_match[0]


        
        ##click on link to open password
        driver.find_element(By.PARTIAL_LINK_TEXT, final_df['name'].iloc[i]).click()
        time.sleep(10)
        ##driver.find_element(By.ID, 'ty').click()
        driver.find_element(By.XPATH, "//input[@type='button']").click()
        password_to_save = driver.find_element(By.XPATH, "//div[@id='thanks_div']").text
        while len(password_to_save)==0:
            password_to_save = driver.find_element(By.XPATH, "//div[@id='thanks_div']").text
            print('getting password')
            time.sleep(2)
        print('got password')
        movie_name = get_movie_name(final_df['name'].iloc[i])
        if movie_name == False:
            movie_name = final_df['name'].iloc[i]
        print(movie_name)
        try:
            # Destination
            try:
                dest = movie_name + ' password='+password_to_save+'.rar'
                # Renaming the file
                os.rename(download_folder+src, download_folder+dest)
            except:
                try:
                    dest = movie_name+'.rar' 
                    os.rename(download_folder+src, download_folder+dest)
                except:
                    try:
                        movie_name = ''.join(e for e in movie_name if e.isalnum())
                        dest = movie_name + ' password='+password_to_save+'.rar'
                        #dest = movie_name+'.rar' 
                        os.rename(download_folder+src, download_folder+dest)  
                    except:
                        movie_name = ''.join(e for e in movie_name if e.isalnum())
                        dest = movie_name + '.rar'
                        os.rename(download_folder+src, download_folder+dest)  
        except Exception as e:
            print(e)
        try:
            print('renamed file')
            shutil.move( download_folder+dest, data['target_directory_file'])

            # get description
            body = driver.find_element(By.XPATH, "/html/body/table/tbody/tr")
            a  = body.text
            soup  = BeautifulSoup(a)
            description = (soup.prettify())
            desc_index  =description.find('Description')
            screen_index = description.find('Screenshots')
            final_description = description[desc_index:screen_index]
            #create_info_file(movie_name,description)
            os.chdir(data['target_directory_info_file'])
            with open(movie_name+'.txt', 'w') as f:
                f.write(final_description)
                f.write('password = '+password_to_save)

            print(i)
        except:
            print('already exists')
    except Exception as e:
        print(e)

报错信息

Traceback (most recent call last):
  File "C:\Python310\lib\site-packages\pandas\core\indexes\base.py", line 3800, in get_loc
    return self._engine.get_loc(casted_key)
  File "pandas\_libs\index.pyx", line 138, in pandas._libs.index.IndexEngine.get_loc
  File "pandas\_libs\index.pyx", line 165, in pandas._libs.index.IndexEngine.get_loc
  File "pandas\_libs\hashtable_class_helper.pxi", line 5745, in pandas._libs.hashtable.PyObjectHashTable.get_item
  File "pandas\_libs\hashtable_class_helper.pxi", line 5753, in pandas._libs.hashtable.PyObjectHashTable.get_item
KeyError: 'date'

The above exception was the direct cause of the following exception:

Traceback (most recent call last):
  File "C:\Users\jerem\Desktop\Final\BD25 Scraper\project2.py", line 104, in <module>
    final_df['Upload_date'] = pd.to_datetime(final_df['date'],dayfirst=True)
  File "C:\Python310\lib\site-packages\pandas\core\frame.py", line 3805, in __getitem__
    indexer = self.columns.get_loc(key)
  File "C:\Python310\lib\site-packages\pandas\core\indexes\base.py", line 3802, in get_loc
    raise KeyError(key) from err
KeyError: 'date'

问题原因

KeyError: 'date'说明final_df中根本没有date列,核心原因是目标网站的页面结构发生了变化:

  • 代码依赖固定索引(range(330,430)、value_list[-3])提取数据,网站更新布局后,这些索引位置不再对应日期或有效条目。
  • 筛选条件if value_list[-3] != 'Upload'可能永远不满足,导致没有任何数据被添加到final_df中。

修复方案

1. 先调试确认当前页面结构

在爬取数据的循环中添加打印,查看当前页面的内容格式:

# 在body = driver.find_element(...)后添加以下代码
print("当前页面总行数:", len(y))
print("前5行内容:", y[:5])
print("某一行的分割结果:", y[330].split(' ') if len(y)>330 else "行号超出范围")

通过输出确认当前页面的有效数据位置、日期的实际格式和索引。

2. 替换固定索引为动态定位

不要依赖固定行号和索引,改用正则表达式匹配日期,或通过元素属性定位有效条目:

# 替换原有的for i in range(330,430)循环部分
import re
date_pattern = re.compile(r'\d{2}/\d{2}/\d{4}')  # 匹配DD/MM/YYYY格式的日期

# 先收集所有有效行
valid_rows = [row for row in y if date_pattern.search(row)]

final_data = []  # 用列表收集数据,替代append(已弃用)
for row in valid_rows:
    value_list = row.split(' ')
    # 找到日期在列表中的位置
    date_idx = next((i for i, val in enumerate(value_list) if date_pattern.match(val)), -1)
    if date_idx == -1:
        continue
    # 提取名称:从第1个元素到日期前一个元素
    name = ' '.join(value_list[1:date_idx]).strip()
    final_data.append({'date': value_list[date_idx], 'name': name})

final_df = pd.DataFrame(final_data)

3. 添加空数据判断

在处理final_df前检查是否为空,避免后续报错:

if final_df.empty:
    print("未爬取到任何数据,请检查页面结构或筛选规则")
    driver.quit()
    exit()

4. 优化元素定位方式

原代码中用/html/body/table/tbody/tr获取页面内容过于脆弱,改用更具体的XPath定位目标表格:

# 替换原有的body定位代码
table = driver.find_element(By.XPATH, "//table[@class='your-table-class']")  # 替换为实际表格的class或其他属性
all_text = table.text
y = all_text.split('\n')

内容的提问来源于stack exchange,提问作者Jeremy Backup

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.14 01:00:57