使用rvest抓取网页后,如何清理冗余内容并生成结构化data.frame
处理带分区标签的HTML数据并生成Data Frame
Got it, let's work through this problem step by step. Since you have the raw HTML snippet (and even if you only had the messy text output from html_text()), we can clean this up and turn it into a structured data.frame easily in R with rvest and some basic string manipulation.
Option 1: Parse the HTML Directly (Recommended)
Working with the actual HTML nodes is more reliable than dealing with flattened text, since we can target labels and values directly without guessing how to split messy strings. Here's how to do it:
library(rvest) library(dplyr) library(stringr) # Wrap your raw HTML snippet in a basic HTML structure for proper parsing html_raw <- '<html><body><table><tr><td><span style="color:#ffb76b;"> Member Information:</span></td><br /> <td><span style="margin-left:145px; color:white;"> Name:</span></td> <td><span style="color:white; padding-left:100px;"> John Doe</span></td><br /><td><span style="margin-left:145px; color:white;"> City:</span></td> <td><span style="color:white; padding-left:112px;"> Milwaukee</span></td><br /><td><span style="margin-left:145px; color:white;"> State:</span></td> <td><span style="color:white; padding-left:107px;"> WI</span></td><br /><td><span style="margin-left:145px; color:white;"> Zip:</span></td> <td><span style="color:white; padding-left:118px;"> 53045</span></td><br /><td><span style="margin-left:145px; color:white;"> Angler Class:</span></td> <td><span style="color:white; padding-left:62px;"> Male</span></td><br /><td><span style="color:#ffb76b"> Car Information:</span></td><br /> <td><span style="margin-left:145px; color:white;"> Date Bought:</span></td> <td><span style="color:#ffb76b; padding-left:62px;"> 09/13/1999</span></td><br /><td><span style="margin-left:145px; color:white;"> Time:</span></td> <td><span style="color:white; padding-left:107px;"> 8 pm</span></td><br /><td><span style="margin-left:145px; color:white;"> Length:</span></td> <td><span style="color:white; padding-left:97px;"> 16ft</span></td><br /><td><span style="margin-left:145px; color:white;"> Weight:</span></td> <td><span style="color:white; padding-left:95px;"> Not Specified</span></td><br /><td><span style="margin-left:145px; color:white;"> Tire:</span></td> <td><span style="color:white; padding-left:108px;"> Not Specified</span></td><br /><td><span style="margin-left:145px; color:white;"> Age:</span></td> <td><span style="color:white; padding-left:72px;"> 20 years</span></td><br /><td><span style="margin-left:145px; color:white;"> Mileage:</span></td> <td><span style="color:white; padding-left:67px;"> 65,305</span></td><br /><td><span style="margin-left:145px; color:white;"> Damage:</span></td> <td><span style="color:white; padding-left:58px;"> None</span></td><br /><td><span style="margin-left:145px; color:white;"> Model:</span></td> <td><span style="color:white; padding-left:93px;"> Taurus</span></td><br /><td><span style="color:#ffb76b"> Dealer Information:</span></td><br /> <td><span style="margin-left:145px; color:white;"> Name:</span></td> <td><span style="color:white; padding-left:114px;"> ABC Auto</span></td><br /><td><span style="margin-left:145px; color:white;"> Zip:</span></td> <td><span style="color:white; padding-left:47px;"> 15101</span></td><br /><td><span style="margin-left:145px; color:white;"> Rating:</span></td> <td><span style="color:white; padding-left:63px;"> 5</span></td><br /><td><span style="color:#ffb76b"> Buyer Information:</span></td><br /> <td><span style="margin-left:145px; color:white;"> Sales Person:</span></td> <td><span style="color:white; padding-left:66px;"> John Wick</span></td><br /><td><span style="margin-left:145px; color:white;"> County:</span></td> <td><span style="color:white; padding-left:94px;"> Waukesha</span></td><br /><td><span style="margin-left:145px; color:white;"> State:</span></td> <td><span style="color:white; padding-left:107px;"> Wisconsin</span></td><br /><td><span style="margin-left:145px; color:white;"> Rating:</span></td> <td><span style="color:white; padding-left:58px;"> Good</span></td><br /><td><span style="margin-left:145px; color:white;"> Experience:</span></td> <td><span style="color:white; padding-left:83px;"> Not Specified</span></td><br /><td><span style="color:#ffb76b"> Insurance Information:</span></td><br /> <td><span style="margin-left:145px; color:white;"> Insurer Name:</span></td> <td><span style="color:white; padding-left:68px;"> Allstate</span></td><br /><td><span style="margin-left:145px; color:white;"> Primary Coverage:</span></td> <td><span style="color:white; padding-left:52px;"> Tier3</span></td><br /><td><span style="margin-left:145px; color:white;"> Secondary Coverage:</span></td> <td><span style="color:white; padding-left:34px;"> Not Specified</span></td><br /><td><span style="margin-left:145px; color:white;"> Policy Period:</span></td> <td><span style="color:white; padding-left:64px;"> 2019</span></td><br /></tr></table></body></html>' # Parse the HTML content page <- read_html(html_raw) # Extract all span text, automatically trimming extra whitespace (newlines/tabs included) all_spans <- page %>% html_elements("span") %>% html_text(trim = TRUE) # Define the section headers we want to exclude section_headers <- c( "Member Information:", "Car Information:", "Dealer Information:", "Buyer Information:", "Insurance Information:" ) # Filter out the section headers to keep only labels and values clean_content <- all_spans[!all_spans %in% section_headers] # Split into labels (every odd index) and values (every even index) labels <- clean_content[seq(1, length(clean_content), 2)] %>% str_remove(":$") # Remove trailing colon from each label values <- clean_content[seq(2, length(clean_content), 2)] # Convert to a data frame, keeping original label names (no forced renaming) result_df <- data.frame( setNames(as.list(values), labels), check.names = FALSE ) # View the final structured data print(result_df)
Option 2: Clean the Flattened Text Output
If you only have the messy text from html_text() (like your initial example), we can use string splitting to salvage the data:
library(stringr) # Your raw text output raw_text <- '[1] "window.location = \\"https://example.com\\" Submission Information: Name:\\n\\t\\t\\t\\t\\t Jon Doe City:\\n\\t\\t\\t\\t\\t Milwaukee State:\\n\\t\\t\\t\\t\\t WI Zip:\\n\\t\\t\\t\\t\\t 53045 Car Information: Date Bought:\\n\\t\\t\\t\\t\\t 07/13/1999 Time:\\n\\t\\t\\t\\t\\t 8 pm Brand:\\n\\t\\t\\t\\t\\t Ford Color:\\n\\t\\t\\t\\t\\t Blue"' # Step 1: Strip unnecessary prefixes, escape characters, and section headers clean_text <- raw_text %>% str_remove('^\\[1\\] "') %>% # Remove the [1] " prefix str_remove('"$') %>% # Remove trailing quote str_remove('window.location = \\"https://example.com\\" ') %>% # Remove redirect line str_replace_all('\\n\\t+', ' ') %>% # Replace newlines/tabs with spaces str_remove('Submission Information: ') %>% str_remove('Car Information: ') # Step 2: Split into label-value pairs by detecting the start of a new label (capitalized word + colon) split_markers <- str_locate_all(clean_text, '\\s[A-Z][a-z]+:')[[1]][,1] segments <- str_sub(clean_text, c(1, split_markers), c(split_markers-1, nchar(clean_text))) # Extract cleaned labels and values labels <- segments %>% str_extract('^.+:') %>% str_remove(':') %>% str_trim() values <- segments %>% str_remove('^.+:') %>% str_trim() # Convert to data frame result_df <- data.frame(setNames(as.list(values), labels)) print(result_df)
Quick Tips:
- Stick with Option 1 whenever possible—HTML parsing avoids the headaches of dealing with unpredictable text formatting.
- If you have duplicate labels (like "Name" or "State" across sections), you can modify the code to prefix labels with their section name (e.g., "Member Name", "Dealer Name")—just let me know if you need help with that!
内容的提问来源于stack exchange,提问作者jmp88
相关产品推荐
相关产品推荐

