如何用Google Apps Script翻译Google Docs文件并保留格式、表格与图片?
批量翻译Google Docs并保留格式的解决方案
问题背景
我需要用Google Apps Script实现类似Google Docs自带“翻译文档”的功能,批量翻译指定文件夹内的文档。但使用LanguageApp.translate()提取整个文档纯文本翻译时,会丢失表格边框、字体样式等所有格式,现有代码如下:
function TranslateFunction() { //Get the files in your indicated Folder var TargetFolderID = '1VUNGtqiNbnHhIFCXmbdSwNZ-vZ5NWVTE'; //Paste the folder ID here to start var folder = DriveApp.getFolderById(TargetFolderID); var files = folder.getFiles(); //Get all the files' ID in the folder above while (files.hasNext()){ var file = files.next(); var fileID = file.getId(); //Convert each file in the folder from Docx (Word) to Docs (Google) var docx = DriveApp.getFileById(fileID); var newDoc = Drive.newFile(); var blob = docx.getBlob(); var file=Drive.Files.insert(newDoc,blob,{convert:true}); DocumentApp.openById(file.id).setName(docx.getName().slice(0,-5)); //Activate the file var doc = DocumentApp.openById(file.id); //Create a new Docs file to put the translation in + Name var newDoc = DocumentApp.create("|EN| " + docx.getName().slice(0,-5)); //Get the text in the file var bodyText = doc.getBody().getText(); //Translate the text and append it into the new Docs Translatedtext = LanguageApp.translate(bodyText,'vi','en'); newDoc.getBody().appendParagraph(Translatedtext); } }
解决思路与代码
要保留格式,必须逐元素处理文档内容,翻译每个元素的文本同时复制原格式。以下是修改后的代码,支持段落、表格、标题等常见元素的格式保留:
function batchTranslateDocsWithFormat() { const targetFolderId = '1VUNGtqiNbnHhIFCXmbdSwNZ-vZ5NWVTE'; // 目标文件夹ID const sourceLang = 'vi'; // 源语言 const targetLang = 'en'; // 目标语言 const folder = DriveApp.getFolderById(targetFolderId); const files = folder.getFiles(); while (files.hasNext()) { const file = files.next(); const fileName = file.getName(); // 仅处理docx格式文件 if (!fileName.endsWith('.docx')) continue; // 将docx转换为Google Docs格式 const blob = file.getBlob(); const convertedFile = Drive.Files.insert({}, blob, {convert: true}); const convertedDoc = DocumentApp.openById(convertedFile.id); convertedDoc.setName(fileName.slice(0, -5)); // 创建翻译后的新文档 const translatedDoc = DocumentApp.create(`|EN| ${fileName.slice(0, -5)}`); const sourceBody = convertedDoc.getBody(); const targetBody = translatedDoc.getBody(); // 遍历源文档所有元素,逐个翻译并保留格式 const numElements = sourceBody.getNumChildren(); for (let i = 0; i < numElements; i++) { const element = sourceBody.getChild(i); translateElement(element, targetBody, sourceLang, targetLang); } // 保存并关闭文档 translatedDoc.saveAndClose(); convertedDoc.saveAndClose(); } } // 递归处理不同类型的文档元素 function translateElement(element, targetBody, sourceLang, targetLang) { const elementType = element.getType(); switch(elementType) { // 处理普通段落 case DocumentApp.ElementType.PARAGRAPH: const sourcePara = element.asParagraph(); const targetPara = targetBody.appendParagraph(''); // 复制段落整体格式 targetPara.setAttributes(sourcePara.getAttributes()); // 翻译段落内的每个文本片段,保留局部样式 const numChildren = sourcePara.getNumChildren(); for (let j = 0; j < numChildren; j++) { const child = sourcePara.getChild(j); if (child.getType() === DocumentApp.ElementType.TEXT) { const text = child.asText().getText(); const translatedText = LanguageApp.translate(text, sourceLang, targetLang); const targetText = targetPara.appendText(translatedText); // 复制文本的字体、颜色等样式 targetText.setAttributes(child.asText().getAttributes()); } else { // 段落内的非文本元素(如图片)直接复制 targetPara.appendElement(child.copy()); } } break; // 处理表格 case DocumentApp.ElementType.TABLE: const sourceTable = element.asTable(); const targetTable = targetBody.appendTable(); // 复制表格整体样式(边框、对齐等) targetTable.setAttributes(sourceTable.getAttributes()); const numRows = sourceTable.getNumRows(); for (let rowIdx = 0; rowIdx < numRows; rowIdx++) { const sourceRow = sourceTable.getRow(rowIdx); const targetRow = targetTable.appendTableRow(); const numCells = sourceRow.getNumCells(); for (let cellIdx = 0; cellIdx < numCells; cellIdx++) { const sourceCell = sourceRow.getCell(cellIdx); const targetCell = targetRow.appendTableCell(); // 复制单元格格式 targetCell.setAttributes(sourceCell.getAttributes()); // 递归处理单元格内的元素 const cellChildren = sourceCell.getNumChildren(); for (let k = 0; k < cellChildren; k++) { translateElement(sourceCell.getChild(k), targetCell, sourceLang, targetLang); } } } break; // 处理标题和列表项 case DocumentApp.ElementType.HEADING_1: case DocumentApp.ElementType.HEADING_2: case DocumentApp.ElementType.HEADING_3: case DocumentApp.ElementType.LIST_ITEM: const sourceHeading = element.asParagraph(); const targetHeading = targetBody.appendParagraph(''); targetHeading.setAttributes(sourceHeading.getAttributes()); const headingText = sourceHeading.getText(); const translatedHeading = LanguageApp.translate(headingText, sourceLang, targetLang); targetHeading.setText(translatedHeading); break; // 默认处理:直接复制非文本元素(如图片、分隔线) default: targetBody.appendElement(element.copy()); break; } }
关键说明
- 逐元素处理:不再提取整个文档的纯文本,而是遍历每个元素,翻译文本内容的同时复制原格式
- 递归处理嵌套元素:表格单元格内的段落、文本会被递归处理,确保格式完整保留
- 格式复制:通过
getAttributes()和setAttributes()复制元素的样式(字体、颜色、对齐方式、表格边框等) - 文件过滤:仅处理
.docx文件,避免不必要的格式转换
内容的提问来源于stack exchange,提问作者Việt Lương
相关产品推荐
相关产品推荐

