You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在Python项目中集成docx4j的接受跟踪修订功能?

如何在Python中调用docx4j实现接受跟踪修订功能?

我完全不是Java专家,对Maven也很陌生,现在想使用docx4j实现Word文档的跟踪修订接受功能,已经成功创建了独立Maven项目,导入的跟踪修订代码能正常运行,但不知道怎么把这个功能集成到我的Python项目里,目标是在Python函数里实现docx4j的接受跟踪修订功能,特此请教。

当前Java代码

package org.example;

import java.io.File;
import org.docx4j.Docx4J;
import org.docx4j.XmlUtils;
import org.docx4j.openpackaging.parts.WordprocessingML.MainDocumentPart;
import org.docx4j.openpackaging.packages.WordprocessingMLPackage;
import org.docx4j.utils.ResourceUtils;

import javax.xml.transform.Source;
import javax.xml.transform.Templates;
import javax.xml.transform.dom.DOMResult;
import javax.xml.transform.stream.StreamSource;
import java.io.FileInputStream;
import java.io.FileOutputStream;
import java.nio.file.Files;
import java.nio.file.Paths;

public class AcceptTrackingChanges {

    public static void main(String[] args) throws Exception {
        // Load the docx
        WordprocessingMLPackage wordMLPackage = Docx4J.load(Files.newInputStream(Paths.get("Student_report.docx")));

        // Load the XLST
        Source xsltSource  = new StreamSource(
                ResourceUtils.getResource(
                        "AcceptChanges.xslt")
        );
        Templates xslt = XmlUtils.getTransformerTemplate(xsltSource);

        MainDocumentPart mdp = wordMLPackage.getMainDocumentPart();

        DOMResult contentAccepted = new DOMResult();

        // perform the transformation
        mdp.transform(xslt, null, contentAccepted);

        // replace the contents in the WordprocessingMLPackage
        org.w3c.dom.Document domDoc = (org.w3c.dom.Document)contentAccepted.getNode();
        mdp.setContents(
                mdp.unmarshal(domDoc.getDocumentElement()));

        System.out.println(mdp.getXML());

        // Save it
//        if (SAVE = true) {
//            String outputfilepath = System.getProperty("user.dir") + "/PI Onsior qrd pi en - tc -accepted.docx";
//            Docx4J.save(wordMLPackage, new File(outputfilepath), Docx4J.FLAG_NONE); //(FLAG_NONE == default == zipped docx)
//
//            System.out.println("Saved: " + outputfilepath);
//        }

        FileOutputStream out = new FileOutputStream("Accepted_changes.docx");
    }
}

当前pom.xml文件

<?xml version="1.0" encoding="UTF-8"?>
<project xmlns="http://maven.apache.org/POM/4.0.0"
         xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
         xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/xsd/maven-4.0.0.xsd">
    <modelVersion>4.0.0</modelVersion>

    <groupId>org.example</groupId>
    <artifactId>docx4j-parent</artifactId>
    <version>1.0-SNAPSHOT</version>
    <description>
        fat jar
    </description>

    <build>
        <plugins>
            <plugin>
                <groupId>org.apache.maven.plugins</groupId>
                <artifactId>maven-compiler-plugin</artifactId>
                <version>3.6.0</version>
                <configuration>
                    <source>1.8</source>
                    <target>1.8</target>
                    <encoding>UTF-8</encoding>
                    <forceJavacCompilerUse>true</forceJavacCompilerUse>
                </configuration>
            </plugin>
            <!--  mvn package

                filters avoid "Invalid signature file digest for Manifest main attributes"
                -->
            <plugin>
                <groupId>org.apache.maven.plugins</groupId>
                <artifactId>maven-shade-plugin</artifactId>
                <version>3.1.0</version>
                <executions>
                    <execution>
                        <phase>package</phase>
                        <goals>
                            <goal>shade</goal>
                        </goals>
                        <configuration>
                            <artifactSet>
                                <excludes>
                                    <exclude>junit:junit</exclude>
                                </excludes>
                            </artifactSet>
                            <shadedArtifactAttached>true</shadedArtifactAttached>
                            <shadedClassifierName>shaded</shadedClassifierName>

                            <filters>
                                <filter>
                                    <artifact>*:*</artifact>
                                    <excludes>
                                        <exclude>META-INF/*.SF</exclude>
                                        <exclude>META-INF/*.DSA</exclude>
                                        <exclude>META-INF/*.RSA</exclude>
                                    </excludes>
                                </filter>
                            </filters>
                        </configuration>
                    </execution>
                </executions>
            </plugin>

            <!--  don't deploy this jar to Maven Central -->
            <plugin>
                <groupId>org.apache.maven.plugins</groupId>
                <artifactId>maven-deploy-plugin</artifactId>
                <version>3.0.0-M1</version>
                <configuration>
                    <skip>true</skip>
                </configuration>
            </plugin>

        </plugins>
    </build>
    <dependencies>
        <dependency>
            <groupId>org.docx4j</groupId>
            <artifactId>docx4j-JAXB-ReferenceImpl</artifactId>
            <version>11.4.9</version>
        </dependency>
        <dependency>
            <groupId>org.docx4j</groupId>
            <artifactId>docx4j-core</artifactId>
            <version>11.4.9</version>
        </dependency>

    </dependencies>

    <properties>
        <maven.compiler.source>11</maven.compiler.source>
        <maven.compiler.target>11</maven.compiler.target>
        <project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
    </properties>

</project>

解决方案

方法1:打包Java代码为可执行Jar,Python通过子进程调用

这是最简单的集成方式,不需要修改Python项目的核心逻辑,只需调用外部Jar文件。

  1. 完善Java代码
    修复保存逻辑,并支持命令行参数传入输入/输出文件路径:

    public static void main(String[] args) throws Exception {
        if (args.length != 2) {
            System.err.println("用法: java -jar AcceptChanges.jar <输入docx路径> <输出docx路径>");
            System.exit(1);
        }
        String inputPath = args[0];
        String outputPath = args[1];
    
        // Load the docx
        WordprocessingMLPackage wordMLPackage = Docx4J.load(Files.newInputStream(Paths.get(inputPath)));
    
        // Load the XLST
        Source xsltSource  = new StreamSource(
                ResourceUtils.getResource("AcceptChanges.xslt")
        );
        Templates xslt = XmlUtils.getTransformerTemplate(xsltSource);
    
        MainDocumentPart mdp = wordMLPackage.getMainDocumentPart();
        DOMResult contentAccepted = new DOMResult();
        mdp.transform(xslt, null, contentAccepted);
    
        // 更新文档内容
        org.w3c.dom.Document domDoc = (org.w3c.dom.Document)contentAccepted.getNode();
        mdp.setContents(mdp.unmarshal(domDoc.getDocumentElement()));
    
        // 保存文档
        try (FileOutputStream out = new FileOutputStream(outputPath)) {
            Docx4J.save(wordMLPackage, out, Docx4J.FLAG_NONE);
        }
        System.out.println("处理完成,输出文件: " + outputPath);
    }
    
  2. 打包可执行Jar
    在Maven项目根目录执行命令:

    mvn package
    

    打包完成后,会在target目录下生成带依赖的Fat Jar(例如docx4j-parent-1.0-SNAPSHOT-shaded.jar)。

  3. Python调用Jar
    使用subprocess模块调用Jar文件:

    import subprocess
    
    def accept_docx_changes(input_docx, output_docx):
        # 检查Java环境是否存在
        if subprocess.run(["java", "-version"], capture_output=True).returncode != 0:
            raise RuntimeError("未找到Java环境,请先安装JDK 11或更高版本")
        
        jar_path = "./target/docx4j-parent-1.0-SNAPSHOT-shaded.jar"
        # 确保AcceptChanges.xslt与Jar在同一目录,或指定绝对路径
        
        result = subprocess.run(
            ["java", "-jar", jar_path, input_docx, output_docx],
            capture_output=True,
            text=True
        )
        
        if result.returncode != 0:
            raise RuntimeError(f"处理失败: {result.stderr}")
        print(result.stdout)
    
    # 使用示例
    accept_docx_changes("Student_report.docx", "Accepted_changes.docx")
    

方法2:使用JPype直接在Python中调用Java代码

这种方式可以实现更紧密的集成,无需单独调用外部Jar,但配置相对复杂。

  1. 安装JPype

    pip install jpype1
    
  2. Python调用示例

    from jpype import startJVM, shutdownJVM, JClass, JString, java
    
    def accept_docx_changes(input_docx, output_docx):
        # 启动JVM,指定docx4j依赖Jar的路径
        startJVM(
            java.getDefaultJVMPath(),
            "-ea",
            "-Djava.class.path=./target/docx4j-parent-1.0-SNAPSHOT-shaded.jar",
            convertStrings=False
        )
        
        try:
            # 加载所需Java类
            WordprocessingMLPackage = JClass("org.docx4j.openpackaging.packages.WordprocessingMLPackage")
            Docx4J = JClass("org.docx4j.Docx4J")
            XmlUtils = JClass("org.docx4j.XmlUtils")
            ResourceUtils = JClass("org.docx4j.utils.ResourceUtils")
            StreamSource = JClass("javax.xml.transform.stream.StreamSource")
            DOMResult = JClass("javax.xml.transform.dom.DOMResult")
            
            # 加载文档
            wordMLPackage = WordprocessingMLPackage.load(java.nio.file.Files.newInputStream(java.nio.file.Paths.get(JString(input_docx))))
            
            # 执行修订接受逻辑
            xsltSource = StreamSource(ResourceUtils.getResource(JString("AcceptChanges.xslt")))
            xslt = XmlUtils.getTransformerTemplate(xsltSource)
            
            mdp = wordMLPackage.getMainDocumentPart()
            contentAccepted = DOMResult()
            mdp.transform(xslt, None, contentAccepted)
            
            domDoc = contentAccepted.getNode()
            mdp.setContents(mdp.unmarshal(domDoc.getDocumentElement()))
            
            # 保存文档
            with java.io.FileOutputStream(JString(output_docx)) as out:
                Docx4J.save(wordMLPackage, out, Docx4J.FLAG_NONE)
            
            print("处理完成")
        finally:
            shutdownJVM()
    
    # 使用示例
    accept_docx_changes("Student_report.docx", "Accepted_changes.docx")
    

注意事项

  • 确保系统已安装JDK 11或更高版本(与pom.xml中指定的编译版本一致)
  • AcceptChanges.xslt文件需放置在Jar或类路径可访问的位置
  • 先单独测试Java代码运行正常,再集成到Python项目中

内容的提问来源于stack exchange,提问作者DaveLu

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.26 12:02:15