from bs4 import BeautifulSoup
text = '<figure><figcaption></figcaption><figure><img alt="dissident1.jpg" class="image-inline" src="https://example.com/dissident1.jpg" title="dissident1.jpg"/><figcaption>A file photo of XXX in prison, provided by his family. Credit: XXX</figcaption></figure><strong>Autopsy demand</strong></figure>'
# 解析HTML
soup = BeautifulSoup(text, 'html.parser')
# 定位最外层figure标签
outer_figure = soup.find('figure')
# 过滤外层figure内的空figcaption节点,保留有效内容
valid_nodes = [
node for node in outer_figure.contents
if not (node.name == 'figcaption' and not node.get_text(strip=True))
]
# 将有效节点转为HTML字符串
cleaned_html = ''.join(str(node) for node in valid_nodes)
print(cleaned_html)