You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用docx2python保留Word转HTML时编号列表的起始索引

保留Word编号列表起始编号的HTML转换解决方案

问题概述

我正在开发Python3脚本将docx转HTML,用mammoth库时没法保留编号列表的起始编号。比如Word里的分段编号列表:

Word文档中的结构:

<ol>
<li>Test item 1</li>
<li>Test item 2</li>
</ol>
<p>break in list for some random text</p>
<ol start="3">
<li>Test item 3</li>
<li>Test item 4</li>
</ol>

转换后丢失了start="3"属性:

<ol>
<li>Test item 1</li>
<li>Test item 2</li>
</ol>
<p>break in list for some random text</p>
<ol>
<li>Test item 3</li>
<li>Test item 4</li>
</ol>

解决方案1:结合python-docx优化mammoth转换

mammoth本身不直接暴露列表起始编号,但可以用python-docx读取docx底层的编号信息,再给转换后的HTML补充start属性。

完整代码

import mammoth
import os
from bs4 import BeautifulSoup
from docx import Document

def convert_word_to_html(docx_path, output_dir):
    # 第一步:用python-docx提取所有编号列表的起始值
    doc = Document(docx_path)
    list_start_numbers = []
    for para in doc.paragraphs:
        if para.style.name in ['List Number', 'List Paragraph']:
            # 解析docx底层XML获取编号起始值
            if para._p.pPr and para._p.pPr.numPr:
                num_id = para._p.pPr.numPr.numId.val
                num_def = doc.part.numbering_part.numbering_definitions._get_num(num_id)
                start = num_def.lvlOverride[0].start.val if num_def.lvlOverride else 1
                list_start_numbers.append(start)
            else:
                list_start_numbers.append(1)
        else:
            # 非列表段落标记为分隔符
            list_start_numbers.append(None)
    
    # 整理每个ol的起始编号(只保留每个新列表的第一个起始值)
    ol_starts = []
    in_list = False
    for num in list_start_numbers:
        if num is not None:
            if not in_list:
                ol_starts.append(num)
                in_list = True
        else:
            in_list = False

    # 第二步:mammoth基础HTML转换
    style_map = """
    p[style-name='Heading 1'] => h1:fresh
    p[style-name='Heading 2'] => h2:fresh
    p[style-name='Heading 3'] => h3:fresh
    p[style-name='Heading 4'] => h4:fresh
    p[style-name='Heading 5'] => h5:fresh
    p[style-name='Heading 6'] => h6:fresh
    p[style-name='List Paragraph'] => ol > li:fresh
    p[style-name='List Bullet'] => ul > li:fresh
    p[style-name='List Number'] => ol > li:fresh
    """

    with open(docx_path, "rb") as docx_file:
        result = mammoth.convert_to_html(docx_file, style_map=style_map)
        html = result.value

    # 第三步:给ol元素添加start属性
    soup = BeautifulSoup(html, 'html.parser')
    ol_elements = soup.find_all('ol')
    for idx, ol in enumerate(ol_elements):
        if idx < len(ol_starts) and ol_starts[idx] != 1:
            ol['start'] = str(ol_starts[idx])

    # 保留原有的样式处理逻辑
    def extract_style_props(style):
        props = {}
        if style:
            for prop in style.split(';'):
                if ':' in prop:
                    key, value = prop.split(':')
                    props[key.strip()] = value.strip()
        return props

    for elem in soup.find_all(['p', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'span']):
        style = elem.get('style', '')
        props = extract_style_props(style)
        
        text_align = props.get('text-align')
        if text_align:
            elem['class'] = elem.get('class', []) + [f'align-{text_align}']
        
        font_size = props.get('font-size')
        if font_size:
            if 'pt' in font_size:
                size_pt = float(font_size.replace('pt', ''))
                size_px = int(size_pt * 1.33)
                elem['style'] = f'font-size: {size_px}px;'
            else:
                elem['style'] = f'font-size: {font_size};'

    # 添加CSS样式
    css = """
    <style>
        body { font-family: Arial, sans-serif; line-height: 1.6; color: #333; }
        .align-left { text-align: left; }
        .align-center { text-align: center; }
        .align-right { text-align: right; }
        .align-justify { text-align: justify; }
        table { border-collapse: collapse; width: 100%; }
        th, td { border: 1px solid #ddd; padding: 8px; }
        th { background-color: #f2f2f2; }
        img { max-width: 100%; height: auto; }
    </style>
    """

    full_html = f"<html><head>{css}</head><body>{soup.prettify()}</body></html>"

    # 保存转换后的HTML
    filename = os.path.splitext(os.path.basename(docx_path))[0] + '.html'
    output_path = os.path.join(output_dir, filename)
    with open(output_path, 'w', encoding='utf-8') as f:
        f.write(full_html)

    print(f"转换完成:{docx_path} -> {output_path}")

# 调用示例
docx_path = './input.docx'
output_dir = './'
convert_word_to_html(docx_path, output_dir)

解决方案2:用docx2python直接实现

docx2python可以直接读取列表的编号元数据,包括起始值,我们可以手动构建带start属性的HTML列表。

完整代码

import os
from docx2python import docx2python

def convert_docx_to_html(docx_path, output_dir):
    # 读取docx所有内容
    doc = docx2python(docx_path)
    
    html_parts = []
    current_list = None  # 记录当前列表类型(ol/ul)和起始编号
    list_items = []

    for para in doc.body[0]:
        para_style = para.style.name if hasattr(para, 'style') else ''
        text = para.text.strip()
        
        if not text:
            # 空段落,先关闭当前列表
            if current_list:
                ol_tag = f"<ol start='{current_list['start']}'>" if current_list['type'] == 'ol' else "<ul>"
                html_parts.append(ol_tag)
                html_parts.extend([f"<li>{item}</li>" for item in list_items])
                html_parts.append(f"</{current_list['type']}>")
                current_list = None
                list_items = []
            html_parts.append("<p></p>")
            continue
        
        # 处理编号列表项
        if para_style in ['List Number', 'List Paragraph']:
            num_info = para.numbering
            start_num = num_info.get('start', 1) if num_info else 1
            
            if not current_list:
                current_list = {'type': 'ol', 'start': start_num}
                list_items.append(text)
            else:
                list_items.append(text)
        # 处理无序列表项
        elif para_style == 'List Bullet':
            if not current_list or current_list['type'] != 'ul':
                # 关闭之前的列表
                if current_list:
                    ol_tag = f"<ol start='{current_list['start']}'>" if current_list['type'] == 'ol' else "<ul>"
                    html_parts.append(ol_tag)
                    html_parts.extend([f"<li>{item}</li>" for item in list_items])
                    html_parts.append(f"</{current_list['type']}>")
                current_list = {'type': 'ul', 'start': None}
                list_items.append(text)
            else:
                list_items.append(text)
        # 处理普通段落和标题
        else:
            if current_list:
                ol_tag = f"<ol start='{current_list['start']}'>" if current_list['type'] == 'ol' else "<ul>"
                html_parts.append(ol_tag)
                html_parts.extend([f"<li>{item}</li>" for item in list_items])
                html_parts.append(f"</{current_list['type']}>")
                current_list = None
                list_items = []
            
            if para_style.startswith('Heading'):
                level = para_style.split(' ')[1]
                html_parts.append(f"<h{level}>{text}</h{level}>")
            else:
                html_parts.append(f"<p>{text}</p>")
    
    # 关闭最后一个未闭合的列表
    if current_list:
        ol_tag = f"<ol start='{current_list['start']}'>" if current_list['type'] == 'ol' else "<ul>"
        html_parts.append(ol_tag)
        html_parts.extend([f"<li>{item}</li>" for item in list_items])
        html_parts.append(f"</{current_list['type']}>")
    
    # 构建完整HTML文档
    css = """
    <style>
        body { font-family: Arial, sans-serif; line-height: 1.6; color: #333; }
        .align-left { text-align: left; }
        .align-center { text-align: center; }
        .align-right { text-align: right; }
        .align-justify { text-align: justify; }
        table { border-collapse: collapse; width: 100%; }
        th, td { border: 1px solid #ddd; padding: 8px; }
        th { background-color: #f2f2f2; }
        img { max-width: 100%; height: auto; }
    </style>
    """
    html_body = '\n'.join(html_parts)
    full_html = f"<html><head>{css}</head><body>{html_body}</body></html>"
    
    # 保存文件
    filename = os.path.splitext(os.path.basename(docx_path))[0] + '.html'
    output_path = os.path.join(output_dir, filename)
    with open(output_path, 'w', encoding='utf-8') as f:
        f.write(full_html)
    
    print(f"转换完成:{docx_path} -> {output_path}")

# 调用示例
docx_path = './input.docx'
output_dir = './'
convert_docx_to_html(docx_path, output_dir)

内容的提问来源于stack exchange,提问作者Harris Charalambous

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.18 17:22:03