You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Python endesive多页PDF签名异常:仅首个签名可验证

多页PDF签名异常:仅首个签名有效,其余签名注解不显示

问题描述

我使用Class 3 USB令牌,通过在PDF页面中查找特定文本Authorised Signatory进行数字签名,单页签名可正常工作。但在多页文档场景下,签名后用Adobe PDF Reader验证时,仅首个签名可通过验证,其余页面的签名注解无法显示。

核心代码(签名接口)

def post(self, request):
    serializer = PDFSignSerializer(data=request.data)
    if serializer.is_valid():
        data = serializer.validated_data.get('pdf_base64')

        try:
            pdf_data = base64.b64decode(data)
        except Exception as e:
            return Response({'error': f'Failed to decode base64 PDF data: {e}'}, status=status.HTTP_400_BAD_REQUEST)

        # Search for text in the PDF and retrieve coordinates
        text_to_find = 'Authorised Signatory'  # Text to search for (modify as needed)
        text_positions = self.find_text_in_pdf(pdf_data, text_to_find)

        if not text_positions:
            return Response({'error': 'No position found for signature'}, status=status.HTTP_400_BAD_REQUEST)

        # Get the current UTC time using timezone-aware objects
        current_utc_time = datetime.datetime.now(datetime.timezone.utc)

        # Calculate the Indian time by adding the UTC offset of +5:30
        indian_time = current_utc_time + datetime.timedelta(hours=5, minutes=30)

        # Format the Indian time string in 24-hour format
        indian_time_str = indian_time.strftime('%Y-%m-%d %H:%M:%S') + ' UTC+5:30'

        # Initialize the signer
        #In above code the settings.DLLPATH is the path of the file - eps2003csp11v2.dll in windows OS
        clshsm = Signer(settings.DLLPATH)  # Adjust this according to your settings

        # Prepare signing data structure
        date = indian_time - datetime.timedelta(hours=11)
        date = date.strftime('%Y%m%d%H%M%S+00\'00\'')
        dct = {
            "sigflags": 3,
            "sigbutton": True,
            "contact": f'Digitally signed\nDate: {indian_time_str}',
            "location": 'India',
            "signingdate": date.encode(),
            "reason": 'Approved',
            "text": {
                'wraptext': True,
                'fontsize': 6,
                'textalign': 'left',
                'linespacing': 1,
            },
            "signature_appearance": {
                'background': r'C:\Users\Guest\Downloads\check_mark.png',
                'labels': False,
                'display': 'contact'.split(','),
            },
        }

        for position in text_positions:
            try:
                # Reload the PDF document to avoid xref table issues
                signed_pdf_data = pdf_data
                
                dct["sigpage"] = position['page_number']
                dct["signaturebox"] = (
                    position['x0'] - 10,
                    position['page_height'] - position['y0'],
                    position['x1'] + 10,
                    position['page_height'] - position['y0'] + 60
                )

                signed_pdf_data = pdf.cms.sign(signed_pdf_data, dct, None, None, [], 'sha256', clshsm)
                pdf_data = pdf_data + signed_pdf_data
                # Return the signed PDF data after each signing operation
                
                print(base64.b64encode(pdf_data).decode())
                
            except Exception as e:
                return Response({'error': f'Failed to sign at position {position}: {e}'}, status=status.HTTP_500_INTERNAL_SERVER_ERROR)
        # Prepare response
        response_data = {
                    'message': 'PDF signed successfully.',
                    'signed_pdf_base64': base64.b64encode(pdf_data).decode()
                }
        return Response(response_data, status=status.HTTP_200_OK)
    else:
        return Response(serializer.errors, status=status.HTTP_400_BAD_REQUEST)

Signer类代码

class Signer(hsm.HSM):
    def certificate(self):
        self.login("Token", "password")
        keyid = [0x5e, 0x9a, 0x33, 0x44, 0x8b, 0xc3, 0xa1, 0x35, 0x33, 0xc7, 0xc2, 0x02, 0xf6, 0x9b, 0xde, 0x55, 0xfe, 0x83, 0x7b, 0xde]
        keyid = bytes(keyid)
        try:
            pk11objects = self.session.findObjects([(PK11.CKA_CLASS, PK11.CKO_CERTIFICATE)])
            all_attributes = [
                PK11.CKA_VALUE,
                PK11.CKA_ID,
            ]

            for pk11object in pk11objects:
                try:
                    attributes = self.session.getAttributeValue(pk11object, all_attributes)
                except PK11.PyKCS11Error as e:
                    continue

                attrDict = dict(list(zip(all_attributes, attributes)))
                cert = bytes(attrDict[PK11.CKA_VALUE])
                return bytes(attrDict[PK11.CKA_ID]), cert
        finally:
            self.logout()
        return None, None

    def sign(self, keyid, data, mech):
        self.login("Token", "password")
        try:
            privKey = self.session.findObjects([(PK11.CKA_CLASS, PK11.CKO_PRIVATE_KEY)])[0]
            mech = getattr(PK11, 'CKM_%s_RSA_PKCS' % mech.upper())
            sig = self.session.sign(privKey, data, PK11.Mechanism(mech, None))
            return bytes(sig)
        finally:
            self.logout()

错误分析与修复

核心错误点

多签名循环中的PDF数据处理逻辑完全错误:

  • 每次签名时,你基于原始PDF数据生成新签名,然后将签名后的PDF直接追加到原始数据末尾(pdf_data = pdf_data + signed_pdf_data)。PDF是结构化文档,不能通过简单拼接叠加签名,这种操作会导致后续签名的注解无法被阅读器识别。
  • 正确逻辑:每次签名必须基于上一次签名后的完整PDF进行,签名完成后更新pdf_data为最新的已签名文档。

修复后的签名循环代码

for position in text_positions:
    try:
        dct["sigpage"] = position['page_number']
        dct["signaturebox"] = (
            position['x0'] - 10,
            position['page_height'] - position['y0'],
            position['x1'] + 10,
            position['page_height'] - position['y0'] + 60
        )

        # 基于当前已签名的PDF数据进行下一次签名,直接覆盖更新pdf_data
        pdf_data = pdf.cms.sign(pdf_data, dct, None, None, [], 'sha256', clshsm)
        
        print(base64.b64encode(pdf_data).decode())
        
    except Exception as e:
        return Response({'error': f'Failed to sign at position {position}: {e}'}, status=status.HTTP_500_INTERNAL_SERVER_ERROR)

额外优化建议(非核心问题)

Signer类的certificate方法直接返回第一个找到的证书,若令牌中有多个证书可能匹配错误,建议根据预设keyid精准匹配:

def certificate(self):
    self.login("Token", "password")
    target_keyid = bytes([0x5e, 0x9a, 0x33, 0x44, 0x8b, 0xc3, 0xa1, 0x35, 0x33, 0xc7, 0xc2, 0x02, 0xf6, 0x9b, 0xde, 0x55, 0xfe, 0x83, 0x7b, 0xde])
    try:
        pk11objects = self.session.findObjects([(PK11.CKA_CLASS, PK11.CKO_CERTIFICATE)])
        all_attributes = [
            PK11.CKA_VALUE,
            PK11.CKA_ID,
        ]

        for pk11object in pk11objects:
            try:
                attributes = self.session.getAttributeValue(pk11object, all_attributes)
            except PK11.PyKCS11Error as e:
                continue

            attrDict = dict(list(zip(all_attributes, attributes)))
            cert_keyid = bytes(attrDict[PK11.CKA_ID])
            if cert_keyid == target_keyid:
                cert = bytes(attrDict[PK11.CKA_VALUE])
                return cert_keyid, cert
        # 未找到匹配证书时返回空
        return None, None
    finally:
        self.logout()

内容的提问来源于stack exchange,提问作者Yogesh Yadav

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.23 05:37:02