如何爬取IEEE网站?附基于BeautifulSoup与PyQt5的代码求助
Hey there! Let's walk through how to wrap up your IEEE web scraping project using the code you've started. I'll break this down into actionable steps with adjusted code snippets:
Your current Client class loads the page but doesn't save the rendered content. Let's modify it to store the HTML so we can parse it with BeautifulSoup:
import bs4 as bs import sys import time from PyQt5.QtWidgets import QApplication from PyQt5.QtCore import QUrl from PyQt5.QtWebKitWidgets import QWebPage from PyQt5.QtWebKit import QWebSettings class Client(QWebPage): def __init__(self, url, login_creds=None): self.app = QApplication(sys.argv) QWebPage.__init__(self) # Set a realistic user agent to avoid being blocked self.settings().setAttribute(QWebSettings.UserAgentEnabled, True) self.settings().setUserAgent( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36" ) self.html = "" self.login_creds = login_creds self.current_step = "login" if login_creds else "scrape" # Connect load finish signal to our handler self.loadFinished.connect(self.on_page_load) # Start with login page if credentials are provided if login_creds: self.mainFrame().load(QUrl("https://ieeexplore.ieee.org/Xplore/login.jsp")) else: self.mainFrame().load(QUrl(url)) self.app.exec_() def on_page_load(self): if self.current_step == "login": # Fill in login form username_field = self.mainFrame().findElementById("username") if username_field.isValid(): username_field.setAttribute("value", self.login_creds["username"]) password_field = self.mainFrame().findElementById("password") if password_field.isValid(): password_field.setAttribute("value", self.login_creds["password"]) # Click login button login_btn = self.mainFrame().findElementById("sign_in") if login_btn.isValid(): login_btn.evaluateJavaScript("this.click();") self.current_step = "scrape" # Add a small delay to let the login process complete time.sleep(2) elif self.current_step == "scrape": # Capture the fully rendered HTML self.html = self.mainFrame().toHtml() self.app.quit()
Now that we have the rendered HTML, let's build a function to parse it and extract the data you need (adjust selectors based on what you're scraping):
def scrape_ieee_content(url, login_creds=None): # Get rendered HTML from our Client client = Client(url, login_creds) soup = bs.BeautifulSoup(client.html, "html.parser") # Example: Extract paper titles and authors from search results scraped_data = [] # Find all paper entries (adjust selector to match IEEE's current page structure) paper_cards = soup.select("div.List-results-items") for card in paper_cards: # Extract title title_tag = card.select_one("h2.article-title a") title = title_tag.get_text(strip=True) if title_tag else "No title found" # Extract authors author_tags = card.select("div.author-name a") authors = [auth.get_text(strip=True) for auth in author_tags] # Extract publication date date_tag = card.select_one("div.pub-info span") pub_date = date_tag.get_text(strip=True) if date_tag else "No date found" scraped_data.append({ "title": title, "authors": authors, "publication_date": pub_date, "url": title_tag["href"] if title_tag else "" }) return scraped_data
IEEE has strict anti-bot policies, so keep these tips in mind:
- Add delays: Use
time.sleep(2-5)between requests to mimic human behavior - Avoid rapid-fire requests: Don't scrape hundreds of pages in a row without breaks
- Use login credentials: Most full-text content requires a subscription or institutional access
- Rotate user agents: If you run into blocks, switch up the user agent string periodically
Here's how to use your finished scraper:
if __name__ == "__main__": # Example: Scrape IEEE search results for "machine learning" target_url = "https://ieeexplore.ieee.org/search/searchresult.jsp?queryText=machine+learning" # If you need access to restricted content, pass your login credentials # login_details = {"username": "your_ieee_email", "password": "your_password"} # results = scrape_ieee_content(target_url, login_details) results = scrape_ieee_content(target_url) # Print the first 3 results to test for idx, paper in enumerate(results[:3], 1): print(f"Paper {idx}:") print(f"Title: {paper['title']}") print(f"Authors: {', '.join(paper['authors'])}") print(f"Date: {paper['publication_date']}\n")
Note: IEEE's page structure might change over time, so always use your browser's developer tools (F12) to verify CSS selectors if your scraper stops working.
内容的提问来源于stack exchange,提问作者Bose Sanamchai

