构建网络爬虫时,如何不使用sleep()定义crawlSleep休眠函数?
实现crawlSleep休眠函数需求及代码修复
需求说明
需要实现crawlSleep函数,核心要求如下:
- 借助datetime模块获取当前日期
- 自行实现天数计算函数(禁止使用内置天数计算函数),计算距离上次爬取的间隔天数
- 通过while循环维持休眠状态,直到间隔满一周时触发重新爬取,并更新爬取日期
- 暂无需关注HTML解析逻辑,重点完成休眠与触发逻辑
现有代码
基础爬虫代码
import datetime def get_page(url): try: import urllib.request page = urllib.request.urlopen(url).read() page = page.decode("utf-8") return page except: return "" def get_next_target(page): start_link = page.find('<a href=') if start_link == -1: return None, 0 start_quote = page.find('"', start_link) end_quote = page.find('"', start_quote+1) url = page[start_quote + 1:end_quote] return url, end_quote def get_all_links(page): links = [] while True: url, endpos = get_next_target(page) if url: links.append(url) page = page[endpos:] else: break return links def union(p,q): for e in q: if e not in p: p.append(e) def updateCrawlDate(index, keyword, url, date): for entry in index: if entry[0] == keyword: entry[entry.index(url)+1] = date return index.append([keyword, url, date]) def add_toIndex(index, keyword, url,date):### we must be also changed here #date = datetime.datetime.now().strftime("%d-%m-%y") if keyword in index: index[keyword].append({url,date}) #index[keyword].append({date}) else: index[keyword]=[{url}] index[keyword]=[{'test:'+date}] def getclearpage(content): title = content[content.find("<title>")+7:content.find("</title>")] body = content[content.find("<body>")+6:content.find("</body>")] while body.find(">") != -1: start = body.find("<") end = body.find(">") body = body[:start] + body[end+1:] return title + body def addPageToIndex(index, url, content,date): content = getclearpage(content) words = content.split() #date= datetime.datetime.now().strftime("%d-%m-%y") for word in words: add_toIndex(index, word, url,date) def crawlWeb(seed): tocrawl = [seed] crawled = [] index = {} ### i changed list to dictionary global last_crawldate ## last_crawldate= datetime.datetime.now().strftime("%y-%m-%d") while tocrawl: page = tocrawl.pop() if page not in crawled: content = get_page(page) addPageToIndex(index, page, content,"LastcrawlDate:" +last_crawldate) union(tocrawl, get_all_links(get_page(page))) crawled.append(page) return index #, last_crawldate
未完成的crawlSleep代码
def crawlSleep(index,seed): global last_crawldate while True: current_time = datetime.datetime.now().strftime("%y-%m-%d") daysPassed = current_time - last_crawldate ## i know that these are not the data types we need to apply on them .. if daysPassed >= 7 : return crawlWeb(seed) else: return crawlSleep(index,seed)
问题分析与修复方案
原crawlSleep存在以下问题:
- 用字符串类型的日期做减法,无法计算天数差
- 递归调用会导致栈溢出,且无休眠等待,持续占用CPU
- 未实现自定义天数计算逻辑
修复后的完整代码
import datetime import time # 自定义天数计算函数:计算两个datetime对象之间的天数差 def calculate_days_diff(date1, date2): # 统一转换为无时区的datetime对象 d1 = date1.replace(tzinfo=None) d2 = date2.replace(tzinfo=None) # 计算单个日期从1970年开始的总天数 def date_to_total_days(date): # 计算年份累计天数(含闰年判断) year_total = 0 for year in range(1970, date.year): if (year % 4 == 0 and year % 100 != 0) or (year % 400 == 0): year_total += 366 else: year_total += 365 # 计算当月之前的月份天数(含闰年2月调整) month_days = [31,28,31,30,31,30,31,31,30,31,30,31] if (date.year %4 ==0 and date.year%100 !=0) or (date.year%400 ==0): month_days[1] = 29 month_total = sum(month_days[:date.month-1]) # 累加当月天数 return year_total + month_total + date.day return abs(date_to_total_days(d1) - date_to_total_days(d2)) # 全局变量存储上次爬取日期 last_crawldate = None def crawlWeb(seed): tocrawl = [seed] crawled = [] index = {} global last_crawldate last_crawldate = datetime.datetime.now() while tocrawl: page = tocrawl.pop() if page not in crawled: content = get_page(page) addPageToIndex(index, page, content,"LastcrawlDate:" + last_crawldate.strftime("%y-%m-%d")) union(tocrawl, get_all_links(get_page(page))) crawled.append(page) return index def crawlSleep(index, seed): global last_crawldate # 初始化上次爬取日期(首次执行时) if last_crawldate is None: last_crawldate = datetime.datetime.now() while True: current_time = datetime.datetime.now() # 使用自定义函数计算天数差 days_passed = calculate_days_diff(current_time, last_crawldate) if days_passed >=7: # 满一周触发重新爬取 print("已间隔7天,开始重新爬取...") new_index = crawlWeb(seed) # 合并新索引到原索引(可根据需求调整逻辑) for keyword, entries in new_index.items(): if keyword in index: index[keyword].extend(entries) else: index[keyword] = entries return index else: # 每小时检查一次,避免资源浪费 print(f"距离下次爬取还有 {7 - days_passed} 天,休眠中...") time.sleep(3600) # 休眠1小时
关键说明
- 自定义天数计算:
calculate_days_diff通过手动累加年、月、日的天数总和得到差值,未使用datetime内置的days属性 - 休眠优化:用
time.sleep()每小时检查一次,避免CPU空转 - 日期类型调整:将
last_crawldate改为datetime对象,便于后续计算 - 索引更新:重新爬取后合并新旧索引,可根据实际需求修改合并逻辑
内容的提问来源于stack exchange,提问作者nord süd
相关产品推荐
相关产品推荐

