Java实现CSV文件合并:基于UniqueID字段去重并补全缺失数据的问题求助
Looks like your current code just appends all records from both files together, which is why you're getting duplicate rows instead of merging them based on the uniqueID field. Let's fix this step by step, plus address some hidden issues in your existing code.
First: Fix the CSV Parsing Delimiter Bug
Looking at your file examples, your first CSV uses | as a delimiter, but your code splits on ", " (comma plus space). That's going to break parsing entirely—you'll end up with a single field containing the entire line! Let's update the CsvParser methods to accept a delimiter parameter so it works with both files:
import java.io.BufferedReader; import java.io.File; import java.io.FileReader; import java.io.FileWriter; import java.io.IOException; import java.util.ArrayList; import java.util.Arrays; import java.util.List; import java.util.Set; import org.slf4j.Logger; import org.slf4j.LoggerFactory; public class CsvParser { private static final Logger log = LoggerFactory.getLogger(CsvParser.class); // Add a delimiter parameter to handle different CSV formats public static List<CsvVo> getRecodrsFromACsv(File file, List<String> keys, String delimiter) throws IOException { BufferedReader br = new BufferedReader(new FileReader(file)); List<CsvVo> records = new ArrayList<>(); boolean isHeader = true; String line = null; while ((line = br.readLine()) != null) { if (isHeader) { isHeader = false; continue; } // Skip empty lines to avoid errors if (line.trim().isEmpty()) continue; CsvVo record = new CsvVo(file.getName()); String[] lineSplit = line.split(delimiter); for (int i = 0; i < lineSplit.length; i++) { if (i < keys.size()) { // Avoid index out of bounds for short lines record.put(keys.get(i), lineSplit[i].trim()); } } records.add(record); } br.close(); return records; } public static List<String> getHeadersFromACsv(File file, String delimiter) throws IOException { BufferedReader br = new BufferedReader(new FileReader(file)); List<String> headers = new ArrayList<>(); String line = br.readLine(); if (line != null) { String[] lineSplit = line.split(delimiter); for (String header : lineSplit) { headers.add(header.trim()); } log.info("HEADERS : " + headers); } br.close(); return headers; } // Keep write logic, now handles empty fields gracefully public static void writeToCsv(final File file, final Set<String> headers, final List<CsvVo> records) throws IOException { FileWriter csvWriter = new FileWriter(file); String sep = ""; String[] headersArr = headers.toArray(new String[headers.size()]); // Write headers for (String header : headersArr) { csvWriter.append(sep); csvWriter.append(header); sep = "|"; } csvWriter.append("\n"); // Write merged records for (CsvVo record : records) { sep = ""; for (String s : headersArr) { csvWriter.append(sep); // Handle null/empty fields to avoid gaps in output csvWriter.append(record.get(s) != null ? record.get(s) : ""); sep = "|"; } csvWriter.append("\n"); } csvWriter.flush(); csvWriter.close(); } }
Second: Implement Merging by UniqueID
Instead of concatenating record lists, we'll use a Map<String, CsvVo> where the key is the uniqueID value. This ensures we only keep one record per ID, and fill in missing fields from the second file into the first.
import java.io.File; import java.io.IOException; import java.util.ArrayList; import java.util.HashMap; import java.util.HashSet; import java.util.List; import java.util.Map; import java.util.Set; import java.util.UUID; import org.slf4j.Logger; import org.slf4j.LoggerFactory; public class CsvMergeMain { private static final Logger log = LoggerFactory.getLogger(CsvMergeMain.class); public static void main(String[] args) throws IOException { File aseFile = new File("merge/mergeFile.txt"); File newFile = new File("dpcFileReturn.txt"); log.info("File To Be Processed : " + newFile.getName()); // Define delimiters matching your actual file formats String aseDelimiter = "\\|"; // Escape | since it's a regex special character String newFileDelimiter = ","; // Fetch headers from both files List<String> csv1Headers = CsvParser.getHeadersFromACsv(aseFile, aseDelimiter); List<String> csv2Headers = CsvParser.getHeadersFromACsv(newFile, newFileDelimiter); // Combine and deduplicate headers Set<String> uniqueHeaders = new HashSet<>(); uniqueHeaders.addAll(csv1Headers); uniqueHeaders.addAll(csv2Headers); // Fetch records with correct delimiters List<CsvVo> csv1Records = CsvParser.getRecodrsFromACsv(aseFile, csv1Headers, aseDelimiter); List<CsvVo> csv2Records = CsvParser.getRecodrsFromACsv(newFile, csv2Headers, newFileDelimiter); // Merge records using uniqueID as the key Map<String, CsvVo> mergedRecordsMap = new HashMap<>(); // First add all records from the first file for (CsvVo record : csv1Records) { String uniqueId = record.get("uniqueID"); if (uniqueId != null && !uniqueId.isEmpty()) { mergedRecordsMap.put(uniqueId, record); } else { // Handle records without uniqueID (use a random ID to avoid conflicts) mergedRecordsMap.put(UUID.randomUUID().toString(), record); } } // Merge records from the second file into the map for (CsvVo record : csv2Records) { String uniqueId = record.get("uniqueID"); if (uniqueId != null && !uniqueId.isEmpty()) { CsvVo existingRecord = mergedRecordsMap.get(uniqueId); if (existingRecord != null) { // Fill empty fields in existing record with values from the second file for (String header : csv2Headers) { String existingValue = existingRecord.get(header); String newValue = record.get(header); if ((existingValue == null || existingValue.isEmpty()) && newValue != null && !newValue.isEmpty()) { existingRecord.put(header, newValue); } } } else { // No matching record found, add the new record mergedRecordsMap.put(uniqueId, record); } } else { // Handle records without uniqueID mergedRecordsMap.put(UUID.randomUUID().toString(), record); } } // Convert map values to a list for writing List<CsvVo> allMergedRecords = new ArrayList<>(mergedRecordsMap.values()); // Write the final merged file File mergedFile = new File("mergedFile.txt"); CsvParser.writeToCsv(mergedFile, uniqueHeaders, allMergedRecords); log.info("Merged File Created : " + mergedFile.getAbsolutePath()); } }
Key Changes Explained
- Delimiter Fix: We now pass the correct delimiter for each file to avoid parsing errors.
- UniqueID Merging: Using a map ensures only one record per
uniqueIDexists. When a match is found, empty fields in the first file's record are filled with values from the second. - Empty Field Handling: We avoid overwriting existing data unless the original field is empty (adjust this logic if you want to prioritize the second file's values instead).
- Header Deduplication: A
HashSetensures each header appears only once in the output.
Optional: Add Fallback Merge Keys
If you want to merge using additional fields (like foreName, surName, or emailAddress) when uniqueID is missing, create a composite key:
// Example composite key for fallback merging String compositeKey = record.get("foreName") + "_" + record.get("surName") + "_" + record.get("emailAddress");
内容的提问来源于stack exchange,提问作者Gary Campbell

