将多行格式TXT文件转为CSV以导入Pandas DataFrame
Hey there! Let's tackle this problem step by step. Since you didn't share the exact TXT file format or your current code attempt, I'll cover common scenarios and flexible solutions you can adapt to your specific needs.
First, clarify how your TXT files are formatted—this dictates the parsing approach. Common examples include:
Case 1: Fixed delimiter (e.g., tabs, spaces, semicolons)
Name Age City
Alice 30 New York
Bob 25 London
Case 2: Unstructured/labeled lines
User: Alice | Age: 30 | Location: New York
User: Bob | Age: 25 | Location: London
If your TXT uses a consistent separator (like spaces, tabs, or pipes), Pandas can directly read and convert it to CSV in a few lines:
import pandas as pd import os from glob import glob # Define your folder path containing TXT files txt_folder = "/path/to/your/txt/directory" # Loop through all TXT files for txt_path in glob(os.path.join(txt_folder, "*.txt")): # Read TXT: adjust `sep` to match your delimiter (e.g., "\t" for tabs, "|" for pipes) # Add `header=0` if your TXT has a header row; use `header=None` otherwise df = pd.read_csv(txt_path, sep="\s+", header=0) # Generate CSV filename (replace .txt with .csv) csv_filename = os.path.splitext(os.path.basename(txt_path))[0] + ".csv" csv_path = os.path.join(txt_folder, csv_filename) # Save to CSV (skip index with `index=False` to avoid extra columns) df.to_csv(csv_path, index=False)
If your TXT has a unique structure (like labeled lines), write a custom parser to extract fields:
import pandas as pd import os import re from glob import glob def parse_txt_line(line): """Extract fields from a single line of TXT (adjust regex to match your format)""" line = line.strip() if not line: return None # Example regex for labeled lines like "User: Alice | Age: 30 | Location: New York" name = re.search(r"User: (\w+)", line).group(1) if re.search(r"User: (\w+)", line) else None age = int(re.search(r"Age: (\d+)", line).group(1)) if re.search(r"Age: (\d+)", line) else None city = re.search(r"Location: (\w+\s?\w+)", line).group(1) if re.search(r"Location: (\w+\s?\w+)", line) else None return {"Name": name, "Age": age, "City": city} txt_folder = "/path/to/your/txt/directory" for txt_path in glob(os.path.join(txt_folder, "*.txt")): data = [] with open(txt_path, "r", encoding="utf-8") as f: for line in f: parsed_row = parse_txt_line(line) if parsed_row: data.append(parsed_row) # Convert to DataFrame and save df = pd.DataFrame(data) csv_path = os.path.splitext(txt_path)[0] + ".csv" df.to_csv(csv_path, index=False)
If you want to merge all TXT data into a single CSV for easier Pandas analysis:
import pandas as pd import os from glob import glob txt_folder = "/path/to/your/txt/directory" all_data = [] for txt_path in glob(os.path.join(txt_folder, "*.txt")): df = pd.read_csv(txt_path, sep="\s+", header=0) all_data.append(df) # Combine all DataFrames combined_df = pd.concat(all_data, ignore_index=True) combined_df.to_csv(os.path.join(txt_folder, "combined_data.csv"), index=False)
- Encoding issues: Add
encoding="latin-1"or"utf-8"topd.read_csv()oropen()if your TXT uses non-standard encoding. - Mismatched columns: Use
on_bad_lines="skip"inpd.read_csv()(Pandas 1.4+) to skip malformed lines. - No header rows: Define column names manually with
names=["Col1", "Col2", "Col3"]inpd.read_csv().
内容的提问来源于stack exchange,提问作者zappagt

