Swift解析Docx文件时无法获取XML的w:t元素问题排查
问题:解析DOCX文件时
w:t元素查找返回空数组 我尝试解析Word的docx文件,但使用代码xmlDocument.rootElement()?.elements(forName: "w:t")时返回空数组。相关代码如下:
func readDocXFile(at url: URL) { do { let fileManager = FileManager() let tempDirectoryURL = fileManager.temporaryDirectory.appendingPathComponent(UUID().uuidString) try fileManager.createDirectory(at: tempDirectoryURL, withIntermediateDirectories: true, attributes: nil) try fileManager.unzipItem(at: url, to: tempDirectoryURL) let documentXMLURL = tempDirectoryURL.appendingPathComponent("word/document.xml") let xmlData = try Data(contentsOf: documentXMLURL) let xmlDocument = try XMLDocument(data: xmlData) let text = extractText(from: xmlDocument) print(text) // Clean up the temporary directory try fileManager.removeItem(at: tempDirectoryURL) } catch { print("Failed to read DOCX file: \(error.localizedDescription)") } } func extractText(from xmlDocument: XMLDocument) -> String { let elements = xmlDocument.rootElement()?.elements(forName: "w:t") ?? [] print(xmlDocument.rootElement()) let text = elements.compactMap { $0.stringValue }.joined(separator: " ") return text }
打印根元素时可以看到XML中存在w:t元素:
... <w:t xml:space="preserve"> Developer • Information Designer</w:t> ...
问题原因
问题出在XML命名空间的处理上。DOCX的document.xml里,w是命名空间前缀,对应的正式命名空间URI是http://schemas.openxmlformats.org/wordprocessingml/2006/main。XMLDocument的elements(forName:)方法只会匹配不带命名空间前缀的本地元素名,直接传w:t是找不到对应元素的。
解决方法
方法1:使用命名空间URI+本地名递归查找
直接通过命名空间URI和元素的本地名(去掉前缀的t),递归遍历所有层级的元素来查找:
func extractText(from xmlDocument: XMLDocument) -> String { let wordNamespace = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" // 递归查找所有带目标命名空间的t元素 func findTextElements(in element: XMLElement) -> [XMLElement] { var result = [XMLElement]() // 查找当前节点下的w:t元素 if let textNodes = element.elements(forNamespaceURI: wordNamespace, qualifiedName: "t") { result.append(contentsOf: textNodes) } // 递归处理子节点 for child in element.children ?? [] { if let childElement = child as? XMLElement { result.append(contentsOf: findTextElements(in: childElement)) } } return result } guard let root = xmlDocument.rootElement() else { return "" } let textElements = findTextElements(in: root) return textElements.compactMap { $0.stringValue }.joined(separator: " ") }
方法2:使用XPath查询(更简洁)
给XMLDocument添加命名空间前缀映射,再用XPath全局查找所有w:t元素:
func extractText(from xmlDocument: XMLDocument) -> String { let wordNamespace = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" // 注册命名空间前缀映射 xmlDocument.addNamespace(withName: "w", stringValue: wordNamespace) do { // 用XPath查询所有层级的w:t元素 let nodes = try xmlDocument.nodes(forXPath: "//w:t") let textElements = nodes.compactMap { $0 as? XMLElement } return textElements.compactMap { $0.stringValue }.joined(separator: " ") } catch { print("XPath查询失败: \(error.localizedDescription)") return "" } }
内容的提问来源于stack exchange,提问作者PruitIgoe
相关产品推荐
相关产品推荐

