如何适配不同结构CSV,高效读取特定物业数据?
I get it, your current fixed-format fscanf approach is rigid and inefficient for variable CSV structures like different Monopoly-style boards with varying property counts. Let's break down better, adaptable solutions that handle dynamic CSV layouts and only extract rows where Type is Property:
Recommended: Use a Mature CSV Parsing Library
Writing your own CSV parser from scratch means handling edge cases like quoted fields (with commas inside), empty values, and line breaks—all of which fscanf fails at. Instead, use a lightweight C library designed for CSV parsing to save time and avoid bugs.
Example with libcsv
Libcsv is a popular, open-source option. Here's a simplified workflow:
- Initialize the parser and set up callback functions to handle fields and rows.
- First, parse the header row to map column names to their indices (e.g., where
Type,Set Id, etc., are located). - For each subsequent row, check if the
Typefield equalsProperty—if so, extract the relevant data into your property struct.
#include <csv.h> #include <stdio.h> #include <stdlib.h> #include <string.h> // Define your property structure typedef struct { int set_id; int intraset_id; char name[100]; int property_cost; int house_cost; int hotel_cost; int rent; int rent_with_house; int rent_with_hotel; } Property; // Track column indices and properties during parsing typedef struct { int type_col; int set_id_col; int intraset_id_col; int name_col; int prop_cost_col; int house_cost_col; int hotel_cost_col; int rent_col; int rent_house_col; int rent_hotel_col; Property* properties; int prop_count; int is_header; } ParseData; // Callback for individual CSV fields int handle_field(void* s, size_t len, void* data) { ParseData* parse_data = (ParseData*)data; char field[100]; strncpy(field, (char*)s, len); field[len] = '\0'; if (parse_data->is_header) { // Map column names to indices if (strcmp(field, "Type") == 0) parse_data->type_col = parse_data->prop_count; else if (strcmp(field, "Set Id") == 0) parse_data->set_id_col = parse_data->prop_count; else if (strcmp(field, "Intraset Id") == 0) parse_data->intraset_id_col = parse_data->prop_count; else if (strcmp(field, "Name") == 0) parse_data->name_col = parse_data->prop_count; else if (strcmp(field, "Property Cost") == 0) parse_data->prop_cost_col = parse_data->prop_count; else if (strcmp(field, "House Cost") == 0) parse_data->house_cost_col = parse_data->prop_count; else if (strcmp(field, "Hotel Cost") == 0) parse_data->hotel_cost_col = parse_data->prop_count; else if (strcmp(field, "Rent") == 0) parse_data->rent_col = parse_data->prop_count; else if (strcmp(field, "Rent with House") == 0) parse_data->rent_house_col = parse_data->prop_count; else if (strcmp(field, "Rent With Hotel") == 0) parse_data->rent_hotel_col = parse_data->prop_count; parse_data->prop_count++; } else { // Store field data temporarily (expand this to track per-row fields as needed) } return 0; } // Callback for end of a CSV row int handle_row(int c, void* data) { ParseData* parse_data = (ParseData*)data; if (parse_data->is_header) { parse_data->is_header = 0; parse_data->prop_count = 0; // Initialize dynamic array for properties parse_data->properties = malloc(10 * sizeof(Property)); } else { // Check if this row is a Property (retrieve Type field from your row buffer) // Populate the Property struct and expand the array as needed parse_data->prop_count++; parse_data->properties = realloc(parse_data->properties, (parse_data->prop_count + 1) * sizeof(Property)); } return 0; } int main() { FILE* file = fopen("board.csv", "r"); if (!file) { perror("Failed to open file"); return 1; } csv_parser parser; if (csv_init(&parser, CSV_APPEND_NULL) != 0) { perror("Failed to initialize parser"); fclose(file); return 1; } ParseData parse_data = {.is_header = 1}; char buf[1024]; size_t bytes_read; while ((bytes_read = fread(buf, 1, sizeof(buf), file)) > 0) { if (csv_parse(&parser, buf, bytes_read, handle_field, handle_row, &parse_data) != bytes_read) { fprintf(stderr, "CSV parse error: %s\n", csv_strerror(csv_error(&parser))); break; } } csv_fini(&parser, handle_field, handle_row, &parse_data); csv_free(&parser); fclose(file); // Use your properties array here... free(parse_data.properties); return 0; }
Alternative: Implement a Flexible Custom CSV Reader
If you don't want to rely on third-party libraries, build a lightweight parser that first maps column names to indices, then processes rows dynamically:
#include <stdio.h> #include <stdlib.h> #include <string.h> typedef struct { int set_id; int intraset_id; char name[100]; int property_cost; int house_cost; int hotel_cost; int rent; int rent_with_house; int rent_with_hotel; } Property; // Split a CSV line into fields (handles basic cases; expand for quoted fields if needed) int split_csv_line(const char* line, char*** fields) { int count = 1; char* temp = strdup(line); for (char* p = temp; *p; p++) if (*p == ',') count++; *fields = malloc(count * sizeof(char*)); int idx = 0; char* start = temp; for (char* p = temp; *p; p++) { if (*p == ',') { *p = '\0'; (*fields)[idx++] = strdup(start); start = p + 1; } } (*fields)[idx++] = strdup(start); free(temp); return count; } void free_fields(char** fields, int count) { for (int i = 0; i < count; i++) free(fields[i]); free(fields); } int main() { FILE* file = fopen("board.csv", "r"); if (!file) { perror("Failed to open file"); return 1; } char line[1024]; // Read header to map columns fgets(line, sizeof(line), file); line[strcspn(line, "\n")] = '\0'; char** headers; int header_count = split_csv_line(line, &headers); // Track column indices int type_col = -1, set_id_col = -1, intraset_id_col = -1; int name_col = -1, prop_cost_col = -1, house_cost_col = -1; int hotel_cost_col = -1, rent_col = -1, rent_house_col = -1, rent_hotel_col = -1; for (int i = 0; i < header_count; i++) { if (!strcmp(headers[i], "Type")) type_col = i; else if (!strcmp(headers[i], "Set Id")) set_id_col = i; else if (!strcmp(headers[i], "Intraset Id")) intraset_id_col = i; else if (!strcmp(headers[i], "Name")) name_col = i; else if (!strcmp(headers[i], "Property Cost")) prop_cost_col = i; else if (!strcmp(headers[i], "House Cost")) house_cost_col = i; else if (!strcmp(headers[i], "Hotel Cost")) hotel_cost_col = i; else if (!strcmp(headers[i], "Rent")) rent_col = i; else if (!strcmp(headers[i], "Rent with House")) rent_house_col = i; else if (!strcmp(headers[i], "Rent With Hotel")) rent_hotel_col = i; } // Dynamic array for properties Property* properties = NULL; int prop_count = 0; // Process each row while (fgets(line, sizeof(line), file)) { line[strcspn(line, "\n")] = '\0'; char** fields; int field_count = split_csv_line(line, &fields); // Check if this is a Property row if (type_col != -1 && field_count > type_col && !strcmp(fields[type_col], "Property")) { // Expand properties array properties = realloc(properties, (prop_count + 1) * sizeof(Property)); if (!properties) { perror("Realloc failed"); break; } Property* p = &properties[prop_count]; // Populate struct (handle empty fields gracefully) p->set_id = (set_id_col != -1 && field_count > set_id_col && strlen(fields[set_id_col])) ? atoi(fields[set_id_col]) : 0; p->intraset_id = (intraset_id_col != -1 && field_count > intraset_id_col && strlen(fields[intraset_id_col])) ? atoi(fields[intraset_id_col]) : 0; strncpy(p->name, (name_col != -1 && field_count > name_col) ? fields[name_col] : "", sizeof(p->name)-1); p->property_cost = (prop_cost_col != -1 && field_count > prop_cost_col && strlen(fields[prop_cost_col])) ? atoi(fields[prop_cost_col]) : 0; p->house_cost = (house_cost_col != -1 && field_count > house_cost_col && strlen(fields[house_cost_col])) ? atoi(fields[house_cost_col]) : 0; p->hotel_cost = (hotel_cost_col != -1 && field_count > hotel_cost_col && strlen(fields[hotel_cost_col])) ? atoi(fields[hotel_cost_col]) : 0; p->rent = (rent_col != -1 && field_count > rent_col && strlen(fields[rent_col])) ? atoi(fields[rent_col]) : 0; p->rent_with_house = (rent_house_col != -1 && field_count > rent_house_col && strlen(fields[rent_house_col])) ? atoi(fields[rent_house_col]) : 0; p->rent_with_hotel = (rent_hotel_col != -1 && field_count > rent_hotel_col && strlen(fields[rent_hotel_col])) ? atoi(fields[rent_hotel_col]) : 0; prop_count++; } free_fields(fields, field_count); } // Example: Print extracted properties for (int i = 0; i < prop_count; i++) { printf("Property: %s | Set ID: %d | Base Rent: %d\n", properties[i].name, properties[i].set_id, properties[i].rent); } // Cleanup free_fields(headers, header_count); free(properties); fclose(file); return 0; }
Key Advantages Over Fixed fscanf
- Flexibility: Works with any CSV layout as long as column names match—no hardcoded positions.
- Scalability: Handles variable numbers of properties without changing code.
- Robustness: Gracefully handles empty fields, column reordering, and other real-world CSV quirks.
- Maintainability: Easy to add support for new columns or modify data extraction logic.
内容的提问来源于stack exchange,提问作者andyong5

