使用docx2python保留Word转HTML时编号列表的起始索引
保留Word编号列表起始编号的HTML转换解决方案
问题概述
我正在开发Python3脚本将docx转HTML,用mammoth库时没法保留编号列表的起始编号。比如Word里的分段编号列表:
Word文档中的结构:
<ol> <li>Test item 1</li> <li>Test item 2</li> </ol> <p>break in list for some random text</p> <ol start="3"> <li>Test item 3</li> <li>Test item 4</li> </ol>转换后丢失了
start="3"属性:<ol> <li>Test item 1</li> <li>Test item 2</li> </ol> <p>break in list for some random text</p> <ol> <li>Test item 3</li> <li>Test item 4</li> </ol>
解决方案1:结合python-docx优化mammoth转换
mammoth本身不直接暴露列表起始编号,但可以用python-docx读取docx底层的编号信息,再给转换后的HTML补充start属性。
完整代码
import mammoth import os from bs4 import BeautifulSoup from docx import Document def convert_word_to_html(docx_path, output_dir): # 第一步:用python-docx提取所有编号列表的起始值 doc = Document(docx_path) list_start_numbers = [] for para in doc.paragraphs: if para.style.name in ['List Number', 'List Paragraph']: # 解析docx底层XML获取编号起始值 if para._p.pPr and para._p.pPr.numPr: num_id = para._p.pPr.numPr.numId.val num_def = doc.part.numbering_part.numbering_definitions._get_num(num_id) start = num_def.lvlOverride[0].start.val if num_def.lvlOverride else 1 list_start_numbers.append(start) else: list_start_numbers.append(1) else: # 非列表段落标记为分隔符 list_start_numbers.append(None) # 整理每个ol的起始编号(只保留每个新列表的第一个起始值) ol_starts = [] in_list = False for num in list_start_numbers: if num is not None: if not in_list: ol_starts.append(num) in_list = True else: in_list = False # 第二步:mammoth基础HTML转换 style_map = """ p[style-name='Heading 1'] => h1:fresh p[style-name='Heading 2'] => h2:fresh p[style-name='Heading 3'] => h3:fresh p[style-name='Heading 4'] => h4:fresh p[style-name='Heading 5'] => h5:fresh p[style-name='Heading 6'] => h6:fresh p[style-name='List Paragraph'] => ol > li:fresh p[style-name='List Bullet'] => ul > li:fresh p[style-name='List Number'] => ol > li:fresh """ with open(docx_path, "rb") as docx_file: result = mammoth.convert_to_html(docx_file, style_map=style_map) html = result.value # 第三步:给ol元素添加start属性 soup = BeautifulSoup(html, 'html.parser') ol_elements = soup.find_all('ol') for idx, ol in enumerate(ol_elements): if idx < len(ol_starts) and ol_starts[idx] != 1: ol['start'] = str(ol_starts[idx]) # 保留原有的样式处理逻辑 def extract_style_props(style): props = {} if style: for prop in style.split(';'): if ':' in prop: key, value = prop.split(':') props[key.strip()] = value.strip() return props for elem in soup.find_all(['p', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'span']): style = elem.get('style', '') props = extract_style_props(style) text_align = props.get('text-align') if text_align: elem['class'] = elem.get('class', []) + [f'align-{text_align}'] font_size = props.get('font-size') if font_size: if 'pt' in font_size: size_pt = float(font_size.replace('pt', '')) size_px = int(size_pt * 1.33) elem['style'] = f'font-size: {size_px}px;' else: elem['style'] = f'font-size: {font_size};' # 添加CSS样式 css = """ <style> body { font-family: Arial, sans-serif; line-height: 1.6; color: #333; } .align-left { text-align: left; } .align-center { text-align: center; } .align-right { text-align: right; } .align-justify { text-align: justify; } table { border-collapse: collapse; width: 100%; } th, td { border: 1px solid #ddd; padding: 8px; } th { background-color: #f2f2f2; } img { max-width: 100%; height: auto; } </style> """ full_html = f"<html><head>{css}</head><body>{soup.prettify()}</body></html>" # 保存转换后的HTML filename = os.path.splitext(os.path.basename(docx_path))[0] + '.html' output_path = os.path.join(output_dir, filename) with open(output_path, 'w', encoding='utf-8') as f: f.write(full_html) print(f"转换完成:{docx_path} -> {output_path}") # 调用示例 docx_path = './input.docx' output_dir = './' convert_word_to_html(docx_path, output_dir)
解决方案2:用docx2python直接实现
docx2python可以直接读取列表的编号元数据,包括起始值,我们可以手动构建带start属性的HTML列表。
完整代码
import os from docx2python import docx2python def convert_docx_to_html(docx_path, output_dir): # 读取docx所有内容 doc = docx2python(docx_path) html_parts = [] current_list = None # 记录当前列表类型(ol/ul)和起始编号 list_items = [] for para in doc.body[0]: para_style = para.style.name if hasattr(para, 'style') else '' text = para.text.strip() if not text: # 空段落,先关闭当前列表 if current_list: ol_tag = f"<ol start='{current_list['start']}'>" if current_list['type'] == 'ol' else "<ul>" html_parts.append(ol_tag) html_parts.extend([f"<li>{item}</li>" for item in list_items]) html_parts.append(f"</{current_list['type']}>") current_list = None list_items = [] html_parts.append("<p></p>") continue # 处理编号列表项 if para_style in ['List Number', 'List Paragraph']: num_info = para.numbering start_num = num_info.get('start', 1) if num_info else 1 if not current_list: current_list = {'type': 'ol', 'start': start_num} list_items.append(text) else: list_items.append(text) # 处理无序列表项 elif para_style == 'List Bullet': if not current_list or current_list['type'] != 'ul': # 关闭之前的列表 if current_list: ol_tag = f"<ol start='{current_list['start']}'>" if current_list['type'] == 'ol' else "<ul>" html_parts.append(ol_tag) html_parts.extend([f"<li>{item}</li>" for item in list_items]) html_parts.append(f"</{current_list['type']}>") current_list = {'type': 'ul', 'start': None} list_items.append(text) else: list_items.append(text) # 处理普通段落和标题 else: if current_list: ol_tag = f"<ol start='{current_list['start']}'>" if current_list['type'] == 'ol' else "<ul>" html_parts.append(ol_tag) html_parts.extend([f"<li>{item}</li>" for item in list_items]) html_parts.append(f"</{current_list['type']}>") current_list = None list_items = [] if para_style.startswith('Heading'): level = para_style.split(' ')[1] html_parts.append(f"<h{level}>{text}</h{level}>") else: html_parts.append(f"<p>{text}</p>") # 关闭最后一个未闭合的列表 if current_list: ol_tag = f"<ol start='{current_list['start']}'>" if current_list['type'] == 'ol' else "<ul>" html_parts.append(ol_tag) html_parts.extend([f"<li>{item}</li>" for item in list_items]) html_parts.append(f"</{current_list['type']}>") # 构建完整HTML文档 css = """ <style> body { font-family: Arial, sans-serif; line-height: 1.6; color: #333; } .align-left { text-align: left; } .align-center { text-align: center; } .align-right { text-align: right; } .align-justify { text-align: justify; } table { border-collapse: collapse; width: 100%; } th, td { border: 1px solid #ddd; padding: 8px; } th { background-color: #f2f2f2; } img { max-width: 100%; height: auto; } </style> """ html_body = '\n'.join(html_parts) full_html = f"<html><head>{css}</head><body>{html_body}</body></html>" # 保存文件 filename = os.path.splitext(os.path.basename(docx_path))[0] + '.html' output_path = os.path.join(output_dir, filename) with open(output_path, 'w', encoding='utf-8') as f: f.write(full_html) print(f"转换完成:{docx_path} -> {output_path}") # 调用示例 docx_path = './input.docx' output_dir = './' convert_docx_to_html(docx_path, output_dir)
内容的提问来源于stack exchange,提问作者Harris Charalambous
相关产品推荐
相关产品推荐

