使用Feign调用FastAPI多网站爬虫服务时请求挂起问题
问题描述
Spring Boot应用通过Feign客户端调用FastAPI网页爬虫服务时,请求会卡在ScrapeResponse response = this.scrapeDate();处,且无任何错误抛出。FastAPI服务会重启爬取进程,但Spring Boot函数无法继续执行API调用之后的逻辑。该问题仅在爬取多个网站时出现,爬取单个网站时一切正常。通常此类请求耗时可达1小时,有时甚至2小时。
Spring Boot调用爬虫的定时函数
@Scheduled(cron = "0 40 16 * * ?") public void scrape() { log.info("Calling web scraping service..."); Instant start = Instant.now(); ScrapeResponse response = this.scrapeDate(); if (response == null) { log.error("Failed to scrape the web"); return; } List<ArticleEntity> scrappedArticles = response.data().stream() .filter(this::isValidArticle) // 检查文章是否有效 .flatMap( article -> { boolean existsInResponse1 = response.data().stream().anyMatch(a -> a.title().equals(article.title())); if (existsInResponse1) { return Stream.of(this.buildArticle(article), this.buildArticle(article)); } else { return Stream.of(this.buildArticle(article)); } }) .toList(); articleRepository.saveAll(scrappedArticles); Instant end = Instant.now(); long durationInSeconds = end.getEpochSecond() - start.getEpochSecond(); long minutes = durationInSeconds / 60; long seconds = durationInSeconds % 60; log.info( "Web scraping completed in {} minutes and {} seconds, Scrapped articles: {}", minutes, seconds, scrappedArticles.size()); }
Feign客户端配置类
@Configuration public class FeignClientConfig { private final ObjectMapper objectMapper; public FeignClientConfig(ObjectMapper objectMapper) { this.objectMapper = objectMapper; } @Bean public Retryer feignRetryer() { return new Retryer.Default(100, 1000, 3); // 初始间隔,最大间隔,最大重试次数 } @Bean public Request.Options options() { return new Request.Options( 180, TimeUnit.MINUTES, // 连接超时(3小时) 180, TimeUnit.MINUTES, // 读取超时(3小时) true ); } @Bean Logger.Level feignLoggerLevel() { return Logger.Level.FULL; } @Bean public Encoder feignEncoder() { return new JacksonEncoder(objectMapper); } @Bean public Decoder feignDecoder() { return new JacksonDecoder(objectMapper); } }
FastAPI爬虫接口及实现
接口定义
@app.post("/scrape/news") async def scrape_news_articles(): thematics_file_path = 'files/thematics.json' thematics_data = load_items(thematics_file_path) thematics = [speciality.name['fr'] for speciality in thematics_data] try: data = scrape_news_articles_function(thematics) except requests.exceptions.ReadTimeout: # 重试时使用base64编码 encoded_thematics = base64.b64encode(str(thematics).encode('utf-8')).decode('utf-8') data = scrape_news_articles_function(encoded_thematics, base64_encoded=True) return {"data": data}
爬虫实现函数
def scrape_news_articles_function(thematics, base64_encoded=False): if base64_encoded: thematics = base64.b64decode(thematics).decode('utf-8') driver = configure_webdriver() response = [] response.extend(scrape_data_business_news(thematics, driver)) response.extend(scrape_data_leconomiste(thematics, driver)) response.extend(scrape_data_kapitalis(thematics, driver)) response.extend(scrape_data_lapresse(thematics, driver)) #Yemchi response.extend(scrape_data_le_temps(thematics, driver)) response.extend(scrape_data_sante_tunisie(thematics, driver)) response.extend(scrape_data_tuniscope(thematics, driver)) response.extend(scrape_data_tunisie_numerique(thematics, driver)) response.extend(scrape_data_webdo(thematics, driver)) #Yemchi response.extend(scrape_data_unicef(thematics, driver)) driver.quit() print("Scraping news articles done.") return response
内容的提问来源于stack exchange,提问作者Aziz Zina
相关产品推荐
相关产品推荐

