You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Swift中如何正确转换含多值数组的JSON为CSV?

Swift中处理JSON数组字段转CSV的正确方法

在Swift开发AI/ML模型时,需将外部API获取的电影数据转为CSV格式用于模型训练。当前通过JSONEncoder编码本地模型为JSON,再用DataFrame转CSV的方式,存在数组类型字段(如Director、Ratings)格式异常的问题,生成的CSV中这类字段会显示包含Optional和结构的冗余内容,示例如下:

[Optional({
    Director =     (
        "Anthony Russo,
    );

}),
Optional({
    Director =     (
      Joe Russo"
    );
})]

当前实现代码

public func createCSV(startDate: Date, endDate: Date) async {
    // 从CoreData获取存储的电影数据并转为本地模型
    let movies = await fetchMovies(startDate: startDate, endDate: endDate)

    let encoder = JSONEncoder()
    if let encoded = try? encoder.encode(movies) {
        if let json = String(data: encoded, encoding: .utf8) {
            do {
                // 基于JSON数据创建DataFrame
                let dataFrame = try DataFrame(jsonData: encoded)
                writeDataframeToCsv(dataFrame: dataFrame)
            } catch {
                Swift.debugPrint("无法创建DataFrame: \(error)")
            }
        }
    }
}

private func writeDataframeToCsv(dataFrame: DataFrame, csvName: String = "MyCsv", fileExtension: String = "csv") {
    let directoryURL = FileManager.default.urls(for: .documentDirectory, in: .userDomainMask)[0]
    let fileURL = URL(fileURLWithPath: csvName, relativeTo: directoryURL).appendingPathExtension(fileExtension)

    do {
        try dataFrame.writeCSV(to: fileURL, options: .init())
        print(fileURL)
    } catch {
        Swift.debugPrint("无法从DataFrame创建CSV: \(error.localizedDescription)")
    }
}

示例JSON数据

{
    "Title": "Avengers: Endgame",
    "Year": "2019",
    "Rated": "N/A",
    "Released": "26 Apr 2019",
    "Runtime": "N/A",
    "Genre": "Action, Adventure, Fantasy, Sci-Fi",
    "Director": [
        "Anthony Russo",
        "Joe Russo"
    ],
    "Writer": "Christopher Markus, Stephen McFeely, Stan Lee (based on the Marvel comics by), Jack Kirby (based on the Marvel comics by), Jim Starlin (comic book)",
    "Actors": "Bradley Cooper, Brie Larson, Chris Hemsworth, Chris Evans",
    "Plot": "After the devastating events of Avengers: Infinity War (2018), the universe is in ruins. With the help of remaining allies, the Avengers assemble once more in order to undo Thanos' actions and restore order to the universe.",
    "Language": "English",
    "Country": "USA",
    "Awards": "N/A",
    "Poster": "https://m.media-amazon.com/images/M/MV5BNGZiMzBkZjMtNjE3Mi00MWNlLWIyYjItYTk3MjY0Yjg5ODZkXkEyXkFqcGdeQXVyNDg4NjY5OTQ@._V1_SX300.jpg",
    "Ratings": [
        "Good",
        "Great"
    ],
    "Metascore": "N/A",
    "imdbRating": "N/A",
    "imdbVotes": "N/A",
    "imdbID": "tt4154796",
    "Type": "movie",
    "DVD": "N/A",
    "BoxOffice": "N/A",
    "Production": "Marvel Studios",
    "Website": "N/A",
    "Response": "True"
}

解决方案

问题根源在于DataFrame默认会保留JSON数组的结构信息,甚至包含Optional的描述,导致CSV格式混乱。需先将数组字段转换为适合CSV的字符串格式(如逗号分隔),再进行转换。

方案一:预处理模型数据后转DataFrame

先将本地模型中的数组字段转为逗号分隔的字符串,再生成DataFrame:

// 定义预处理后的模型结构(或扩展现有模型)
struct ProcessedMovie {
    let title: String
    let year: String
    let rated: String
    let released: String
    let runtime: String
    let genre: String
    let director: String
    let writer: String
    let actors: String
    let plot: String
    let language: String
    let country: String
    let awards: String
    let poster: String
    let ratings: String
    let metascore: String
    let imdbRating: String
    let imdbVotes: String
    let imdbID: String
    let type: String
    let dvd: String
    let boxOffice: String
    let production: String
    let website: String
    let response: String
}

public func createCSV(startDate: Date, endDate: Date) async {
    let movies = await fetchMovies(startDate: startDate, endDate: endDate)
    
    // 预处理数组字段为字符串
    let processedMovies = movies.map { movie in
        ProcessedMovie(
            title: movie.title,
            year: movie.year,
            rated: movie.rated,
            released: movie.released,
            runtime: movie.runtime,
            genre: movie.genre,
            director: movie.director.joined(separator: ", "),
            writer: movie.writer,
            actors: movie.actors,
            plot: movie.plot,
            language: movie.language,
            country: movie.country,
            awards: movie.awards,
            poster: movie.poster,
            ratings: movie.ratings.joined(separator: ", "),
            metascore: movie.metascore,
            imdbRating: movie.imdbRating,
            imdbVotes: movie.imdbVotes,
            imdbID: movie.imdbID,
            type: movie.type,
            dvd: movie.dvd,
            boxOffice: movie.boxOffice,
            production: movie.production,
            website: movie.website,
            response: movie.response
        )
    }
    
    let encoder = JSONEncoder()
    if let encoded = try? encoder.encode(processedMovies) {
        do {
            let dataFrame = try DataFrame(jsonData: encoded)
            writeDataframeToCsv(dataFrame: dataFrame)
        } catch {
            Swift.debugPrint("无法创建DataFrame: \(error)")
        }
    }
}

方案二:自定义CSV写入逻辑

绕过DataFrame,直接手动构建CSV内容,更灵活控制字段格式:

public func createCSV(startDate: Date, endDate: Date) async {
    let movies = await fetchMovies(startDate: startDate, endDate: endDate)
    writeMoviesToCsv(movies: movies)
}

private func writeMoviesToCsv(movies: [Movie]) {
    let directoryURL = FileManager.default.urls(for: .documentDirectory, in: .userDomainMask)[0]
    let fileURL = directoryURL.appendingPathComponent("MyCsv.csv")
    
    // 构建CSV表头
    let headers = [
        "Title", "Year", "Rated", "Released", "Runtime", "Genre",
        "Director", "Writer", "Actors", "Plot", "Language", "Country",
        "Awards", "Poster", "Ratings", "Metascore", "imdbRating",
        "imdbVotes", "imdbID", "Type", "DVD", "BoxOffice", "Production",
        "Website", "Response"
    ]
    var csvContent = headers.joined(separator: ",") + "\n"
    
    // 处理每一行数据,对含逗号或引号的字段进行转义
    for movie in movies {
        let escapedTitle = "\"\(movie.title.replacingOccurrences(of: "\"", with: "\"\""))\""
        let escapedDirector = "\"\(movie.director.joined(separator: ", ").replacingOccurrences(of: "\"", with: "\"\""))\""
        let escapedRatings = "\"\(movie.ratings.joined(separator: ", ").replacingOccurrences(of: "\"", with: "\"\""))\""
        
        let row = [
            escapedTitle,
            "\"\(movie.year)\"",
            "\"\(movie.rated)\"",
            "\"\(movie.released)\"",
            "\"\(movie.runtime)\"",
            "\"\(movie.genre)\"",
            escapedDirector,
            "\"\(movie.writer.replacingOccurrences(of: "\"", with: "\"\""))\"",
            "\"\(movie.actors.replacingOccurrences(of: "\"", with: "\"\""))\"",
            "\"\(movie.plot.replacingOccurrences(of: "\"", with: "\"\""))\"",
            "\"\(movie.language)\"",
            "\"\(movie.country)\"",
            "\"\(movie.awards)\"",
            "\"\(movie.poster)\"",
            escapedRatings,
            "\"\(movie.metascore)\"",
            "\"\(movie.imdbRating)\"",
            "\"\(movie.imdbVotes)\"",
            "\"\(movie.imdbID)\"",
            "\"\(movie.type)\"",
            "\"\(movie.dvd)\"",
            "\"\(movie.boxOffice)\"",
            "\"\(movie.production)\"",
            "\"\(movie.website)\"",
            "\"\(movie.response)\""
        ].joined(separator: ",")
        
        csvContent += row + "\n"
    }
    
    do {
        try csvContent.write(to: fileURL, atomically: true, encoding: .utf8)
        print(fileURL)
    } catch {
        Swift.debugPrint("写入CSV失败: \(error.localizedDescription)")
    }
}

两种方案都能将数组字段转换为清晰的逗号分隔字符串,满足CSV格式要求,适合后续AI/ML模型训练使用。

内容的提问来源于stack exchange,提问作者Mickey223

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.04 21:29:53