You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用GCS签名URL分片上传遇SignatureDoesNotMatch 403错误求助

GCS分片上传签名URL 403 SignatureDoesNotMatch问题排查与修复

问题描述

我需要在服务端生成GCS签名URL,传递给客户端实现大文件分片上传。目前已成功发起分片上传并获取到uploadId,但上传分片时,无论是复用发起请求的POST签名URL,还是为每个分片生成新的PUT签名URL,均返回403错误,错误码为SignatureDoesNotMatch,提示"我们计算的请求签名与您提供的签名不匹配。请检查您的Google密钥和签名方法。"。使用的sa.json服务账号已配置项目存储管理员权限,权限齐全。

原代码如下:

from google.cloud import storage
from google.oauth2 import service_account
import requests

def generate_upload_url(bucket_name, object_name, expiration, method):
    """
    Generates a signed URL to initiate a multipart upload.

    Args:
        bucket_name: Name of the bucket to upload to.
        object_name: Name of the object to upload.
        expiration: Expiration time for the URL in seconds.

    Returns:
        A signed URL for initiating the multipart upload.
    """
    # Assuming the service account JSON key is in the same directory with the name 'sa.json'
    service_account_key_path = 'sa.json'
    credentials = service_account.Credentials.from_service_account_file(service_account_key_path)
    client = storage.Client(credentials=credentials)
    bucket = client.bucket(bucket_name)
    blob = bucket.blob(object_name)
    upload_url = blob.generate_signed_url(
        version="v4",
    #   content_type="application/octet-stream",
        query_parameters={
          "uploads": "",
        },
        method=method,
        expiration=expiration,
        credentials=credentials,
    )
    return upload_url

def do_post_on_signed_url(upload_url):
    """
    Sends a POST request to the signed URL to initiate the upload.

    Args:
        upload_url: The signed URL for initiating the upload.
    """
    response = requests.post(upload_url, headers={"Content-Type": "application/octet-stream"})
    response.raise_for_status()

    print(f"Response: {response.text}")

    # parse uploadid from response example response
    #   <?xml version='1.0' encoding='UTF-8'?><InitiateMultipartUploadResult xmlns='http://s3.amazonaws.com/doc/2006-03-01/'><Bucket>pr-cache-bucket-0</Bucket><Key>randomfile.tar.gz</Key><UploadId>ABPnzm5hfCDjy44Aa2WDHPxGL9kDyJ3GwtFvToGhVYgNul2PiIOcGI8siu_HjXHeYRkbPxo</UploadId></InitiateMultipartUploadResult>
    upload_id = response.text.split("<UploadId>")[1].split("</UploadId>")[0]
    return upload_id


# Send this upload_url to your client-side application
def upload_file_in_chunks(url, upload_id, filepath, chunk_size=1024 * 1024):
    """
    Uploads a file in chunks using the provided URL.

    Args:
        url: The signed URL for uploading data chunks.
        filepath: Path to the file to upload.
        chunk_size: Size of each upload chunk in bytes (default: 1MB).
    """
    with open(filepath, "rb") as f:
        chunk_counter = 1
        for chunk in iter(lambda: f.read(chunk_size), b""):
            print(f"Uploading chunk {chunk_counter}...")
            signed_url = generate_upload_url(bucket_name, object_name, expiration, "PUT")
            url = f"{signed_url}&uploadId={upload_id}&partNumber={chunk_counter}"
            headers = {"Content-Type": "application/octet-stream"}
            if chunk:
                response = requests.put(url, data=chunk, headers=headers)
                print(f"response: {response.text}")
                response.raise_for_status()
                chunk_counter += 1

    # After all chunks uploaded, complete the upload (not shown here)

filepath = "randomfile.tar.gz"
# Example usage
bucket_name = "pr-cache-bucket-0"
object_name = "randomfile.tar.gz"
expiration = 3600  # 1 hour

upload_url = generate_upload_url(bucket_name, object_name, expiration, "POST")
upload_id = do_post_on_signed_url(upload_url)

# merge upload_id with upload_url
upload_url = f"{upload_url}&uploadId={upload_id}"

print(f"Upload URL: {upload_url}")

upload_file_in_chunks(upload_url, upload_id, filepath)

核心原因分析

  1. 签名参数不匹配:分片上传的PUT请求需要将uploadId和partNumber作为查询参数包含在签名计算中,但原代码是先生成不带这两个参数的签名URL,再手动拼接,导致实际请求的参数与签名时使用的参数不一致,触发SignatureDoesNotMatch错误。
  2. Content-Type一致性问题:发起初始化POST请求时,签名过程未指定content_type,但请求中添加了Content-Type header,部分场景下会导致签名验证失败。

修复方案

  1. 生成分片PUT签名URL时,将uploadId和partNumber作为query_parameters传入generate_signed_url方法,确保签名计算包含这些参数。
  2. 保持签名时指定的content_type与实际请求的Content-Type完全一致。
  3. 补充分片上传完成后的合并逻辑,否则上传的分片无法形成完整文件。

修正后的完整代码

from google.cloud import storage
from google.oauth2 import service_account
import requests
import xml.etree.ElementTree as ET

def generate_upload_url(bucket_name, object_name, expiration, method, extra_query_params=None, content_type=None):
    """
    Generates a signed URL for multipart upload operations (initiate or upload chunk).

    Args:
        bucket_name: Name of the bucket to upload to.
        object_name: Name of the object to upload.
        expiration: Expiration time for the URL in seconds.
        method: HTTP method (POST for init, PUT for chunks).
        extra_query_params: Additional query parameters to include in the signature.
        content_type: Content-Type header to include in the signature.

    Returns:
        A signed URL for the operation.
    """
    service_account_key_path = 'sa.json'
    credentials = service_account.Credentials.from_service_account_file(service_account_key_path)
    client = storage.Client(credentials=credentials)
    bucket = client.bucket(bucket_name)
    blob = bucket.blob(object_name)
    
    query_parameters = {"uploads": ""}
    if extra_query_params:
        query_parameters.update(extra_query_params)
    
    upload_url = blob.generate_signed_url(
        version="v4",
        content_type=content_type,
        query_parameters=query_parameters,
        method=method,
        expiration=expiration,
        credentials=credentials,
    )
    return upload_url

def do_post_on_signed_url(upload_url):
    """
    Sends a POST request to initiate multipart upload and get uploadId.
    """
    # 与签名时指定的content_type保持一致
    headers = {"Content-Type": "application/octet-stream"}
    response = requests.post(upload_url, headers=headers)
    response.raise_for_status()

    # 用XML解析代替字符串分割,更可靠
    root = ET.fromstring(response.text)
    upload_id = root.find("{http://s3.amazonaws.com/doc/2006-03-01/}UploadId").text
    return upload_id

def upload_file_in_chunks(bucket_name, object_name, expiration, upload_id, filepath, chunk_size=1024 * 1024):
    """
    Uploads file chunks with properly signed URLs.
    """
    part_etags = []
    with open(filepath, "rb") as f:
        chunk_counter = 1
        for chunk in iter(lambda: f.read(chunk_size), b""):
            print(f"Uploading chunk {chunk_counter}...")
            # 生成包含uploadId和partNumber的签名URL
            signed_url = generate_upload_url(
                bucket_name,
                object_name,
                expiration,
                "PUT",
                extra_query_params={
                    "uploadId": upload_id,
                    "partNumber": chunk_counter
                },
                content_type="application/octet-stream"
            )
            headers = {"Content-Type": "application/octet-stream"}
            response = requests.put(signed_url, data=chunk, headers=headers)
            response.raise_for_status()
            # 保存每个分片的ETag,用于完成上传
            part_etags.append({
                "PartNumber": chunk_counter,
                "ETag": response.headers.get("ETag")
            })
            chunk_counter += 1
    return part_etags

def complete_multipart_upload(bucket_name, object_name, expiration, upload_id, part_etags):
    """
    Completes the multipart upload by merging all chunks.
    """
    # 生成完成上传的签名URL
    complete_url = generate_upload_url(
        bucket_name,
        object_name,
        expiration,
        "POST",
        extra_query_params={
            "uploadId": upload_id
        }
    )
    # 构造完成上传的XML请求体
    xml_body = '<CompleteMultipartUpload>'
    for part in part_etags:
        xml_body += f'<Part><PartNumber>{part["PartNumber"]}</PartNumber><ETag>{part["ETag"]}</ETag></Part>'
    xml_body += '</CompleteMultipartUpload>'
    
    headers = {"Content-Type": "application/xml"}
    response = requests.post(complete_url, data=xml_body, headers=headers)
    response.raise_for_status()
    print("Multipart upload completed successfully!")

# Example usage
filepath = "randomfile.tar.gz"
bucket_name = "pr-cache-bucket-0"
object_name = "randomfile.tar.gz"
expiration = 3600  # 1 hour

# 1. 初始化分片上传,获取uploadId
init_url = generate_upload_url(
    bucket_name,
    object_name,
    expiration,
    "POST",
    content_type="application/octet-stream"
)
upload_id = do_post_on_signed_url(init_url)
print(f"Upload ID: {upload_id}")

# 2. 上传所有分片
part_etags = upload_file_in_chunks(bucket_name, object_name, expiration, upload_id, filepath)

# 3. 完成分片上传,合并所有分片
complete_multipart_upload(bucket_name, object_name, expiration, upload_id, part_etags)

关键修改说明

  • 扩展generate_upload_url函数,支持传入额外查询参数和指定content_type,确保签名包含所有必要参数。
  • 使用XML解析提取uploadId,比字符串分割更可靠。
  • 上传分片时直接生成包含uploadId和partNumber的签名URL,避免手动拼接导致的签名不匹配。
  • 补充complete_multipart_upload函数,完成分片合并,确保文件可用。

内容的提问来源于stack exchange,提问作者Prashant Raj

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.26 14:00:36