使用Python endesive多页PDF签名异常:仅首个签名可验证
多页PDF签名异常:仅首个签名有效,其余签名注解不显示
问题描述
我使用Class 3 USB令牌,通过在PDF页面中查找特定文本Authorised Signatory进行数字签名,单页签名可正常工作。但在多页文档场景下,签名后用Adobe PDF Reader验证时,仅首个签名可通过验证,其余页面的签名注解无法显示。
核心代码(签名接口)
def post(self, request): serializer = PDFSignSerializer(data=request.data) if serializer.is_valid(): data = serializer.validated_data.get('pdf_base64') try: pdf_data = base64.b64decode(data) except Exception as e: return Response({'error': f'Failed to decode base64 PDF data: {e}'}, status=status.HTTP_400_BAD_REQUEST) # Search for text in the PDF and retrieve coordinates text_to_find = 'Authorised Signatory' # Text to search for (modify as needed) text_positions = self.find_text_in_pdf(pdf_data, text_to_find) if not text_positions: return Response({'error': 'No position found for signature'}, status=status.HTTP_400_BAD_REQUEST) # Get the current UTC time using timezone-aware objects current_utc_time = datetime.datetime.now(datetime.timezone.utc) # Calculate the Indian time by adding the UTC offset of +5:30 indian_time = current_utc_time + datetime.timedelta(hours=5, minutes=30) # Format the Indian time string in 24-hour format indian_time_str = indian_time.strftime('%Y-%m-%d %H:%M:%S') + ' UTC+5:30' # Initialize the signer #In above code the settings.DLLPATH is the path of the file - eps2003csp11v2.dll in windows OS clshsm = Signer(settings.DLLPATH) # Adjust this according to your settings # Prepare signing data structure date = indian_time - datetime.timedelta(hours=11) date = date.strftime('%Y%m%d%H%M%S+00\'00\'') dct = { "sigflags": 3, "sigbutton": True, "contact": f'Digitally signed\nDate: {indian_time_str}', "location": 'India', "signingdate": date.encode(), "reason": 'Approved', "text": { 'wraptext': True, 'fontsize': 6, 'textalign': 'left', 'linespacing': 1, }, "signature_appearance": { 'background': r'C:\Users\Guest\Downloads\check_mark.png', 'labels': False, 'display': 'contact'.split(','), }, } for position in text_positions: try: # Reload the PDF document to avoid xref table issues signed_pdf_data = pdf_data dct["sigpage"] = position['page_number'] dct["signaturebox"] = ( position['x0'] - 10, position['page_height'] - position['y0'], position['x1'] + 10, position['page_height'] - position['y0'] + 60 ) signed_pdf_data = pdf.cms.sign(signed_pdf_data, dct, None, None, [], 'sha256', clshsm) pdf_data = pdf_data + signed_pdf_data # Return the signed PDF data after each signing operation print(base64.b64encode(pdf_data).decode()) except Exception as e: return Response({'error': f'Failed to sign at position {position}: {e}'}, status=status.HTTP_500_INTERNAL_SERVER_ERROR) # Prepare response response_data = { 'message': 'PDF signed successfully.', 'signed_pdf_base64': base64.b64encode(pdf_data).decode() } return Response(response_data, status=status.HTTP_200_OK) else: return Response(serializer.errors, status=status.HTTP_400_BAD_REQUEST)
Signer类代码
class Signer(hsm.HSM): def certificate(self): self.login("Token", "password") keyid = [0x5e, 0x9a, 0x33, 0x44, 0x8b, 0xc3, 0xa1, 0x35, 0x33, 0xc7, 0xc2, 0x02, 0xf6, 0x9b, 0xde, 0x55, 0xfe, 0x83, 0x7b, 0xde] keyid = bytes(keyid) try: pk11objects = self.session.findObjects([(PK11.CKA_CLASS, PK11.CKO_CERTIFICATE)]) all_attributes = [ PK11.CKA_VALUE, PK11.CKA_ID, ] for pk11object in pk11objects: try: attributes = self.session.getAttributeValue(pk11object, all_attributes) except PK11.PyKCS11Error as e: continue attrDict = dict(list(zip(all_attributes, attributes))) cert = bytes(attrDict[PK11.CKA_VALUE]) return bytes(attrDict[PK11.CKA_ID]), cert finally: self.logout() return None, None def sign(self, keyid, data, mech): self.login("Token", "password") try: privKey = self.session.findObjects([(PK11.CKA_CLASS, PK11.CKO_PRIVATE_KEY)])[0] mech = getattr(PK11, 'CKM_%s_RSA_PKCS' % mech.upper()) sig = self.session.sign(privKey, data, PK11.Mechanism(mech, None)) return bytes(sig) finally: self.logout()
错误分析与修复
核心错误点
多签名循环中的PDF数据处理逻辑完全错误:
- 每次签名时,你基于原始PDF数据生成新签名,然后将签名后的PDF直接追加到原始数据末尾(
pdf_data = pdf_data + signed_pdf_data)。PDF是结构化文档,不能通过简单拼接叠加签名,这种操作会导致后续签名的注解无法被阅读器识别。 - 正确逻辑:每次签名必须基于上一次签名后的完整PDF进行,签名完成后更新
pdf_data为最新的已签名文档。
修复后的签名循环代码
for position in text_positions: try: dct["sigpage"] = position['page_number'] dct["signaturebox"] = ( position['x0'] - 10, position['page_height'] - position['y0'], position['x1'] + 10, position['page_height'] - position['y0'] + 60 ) # 基于当前已签名的PDF数据进行下一次签名,直接覆盖更新pdf_data pdf_data = pdf.cms.sign(pdf_data, dct, None, None, [], 'sha256', clshsm) print(base64.b64encode(pdf_data).decode()) except Exception as e: return Response({'error': f'Failed to sign at position {position}: {e}'}, status=status.HTTP_500_INTERNAL_SERVER_ERROR)
额外优化建议(非核心问题)
Signer类的certificate方法直接返回第一个找到的证书,若令牌中有多个证书可能匹配错误,建议根据预设keyid精准匹配:
def certificate(self): self.login("Token", "password") target_keyid = bytes([0x5e, 0x9a, 0x33, 0x44, 0x8b, 0xc3, 0xa1, 0x35, 0x33, 0xc7, 0xc2, 0x02, 0xf6, 0x9b, 0xde, 0x55, 0xfe, 0x83, 0x7b, 0xde]) try: pk11objects = self.session.findObjects([(PK11.CKA_CLASS, PK11.CKO_CERTIFICATE)]) all_attributes = [ PK11.CKA_VALUE, PK11.CKA_ID, ] for pk11object in pk11objects: try: attributes = self.session.getAttributeValue(pk11object, all_attributes) except PK11.PyKCS11Error as e: continue attrDict = dict(list(zip(all_attributes, attributes))) cert_keyid = bytes(attrDict[PK11.CKA_ID]) if cert_keyid == target_keyid: cert = bytes(attrDict[PK11.CKA_VALUE]) return cert_keyid, cert # 未找到匹配证书时返回空 return None, None finally: self.logout()
内容的提问来源于stack exchange,提问作者Yogesh Yadav
相关产品推荐
相关产品推荐

