You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何检测File/Blob对象是否为有效UTF-8?现有方法存误判

检测File/Blob是否为有效UTF-8的正确实现

你之前的实现存在逻辑缺陷:blob.text()会将无效的UTF-8字节序列自动替换为�(U+FFFD),但�本身是合法的UTF-8字符,因此无法通过检查该字符的存在来判断原始数据是否为有效UTF-8。

正确的做法是直接读取Blob的原始二进制数据,手动验证字节序列是否符合UTF-8的编码规范:

async function isUTF8(blob) {
  const buffer = await blob.arrayBuffer();
  const uint8 = new Uint8Array(buffer);
  let i = 0;
  const length = uint8.length;

  while (i < length) {
    const byte = uint8[i];

    // 单字节字符(0xxxxxxx)
    if ((byte & 0x80) === 0) {
      i += 1;
      continue;
    }

    // 判断多字节字符的起始字节类型
    let bytesNeeded;
    if ((byte & 0xE0) === 0xC0) {
      bytesNeeded = 1; // 双字节字符,需额外1个后续字节
    } else if ((byte & 0xF0) === 0xE0) {
      bytesNeeded = 2; // 三字节字符,需额外2个后续字节
    } else if ((byte & 0xF8) === 0xF0) {
      bytesNeeded = 3; // 四字节字符,需额外3个后续字节
    } else {
      // 无效的起始字节格式
      return false;
    }

    // 检查剩余字节是否足够组成完整字符
    if (i + bytesNeeded >= length) {
      return false;
    }

    // 验证后续字节是否符合10xxxxxx的格式
    for (let j = 1; j <= bytesNeeded; j++) {
      const nextByte = uint8[i + j];
      if ((nextByte & 0xC0) !== 0x80) {
        return false;
      }
    }

    // 验证编码范围合法性,避免冗余编码或无效码点
    let codePoint;
    if (bytesNeeded === 1) {
      codePoint = ((byte & 0x1F) << 6) | (uint8[i+1] & 0x3F);
      if (codePoint < 0x80) return false; // 双字节不能表示单字节可覆盖的字符
    } else if (bytesNeeded === 2) {
      codePoint = ((byte & 0x0F) << 12) | ((uint8[i+1] & 0x3F) << 6) | (uint8[i+2] & 0x3F);
      if (codePoint < 0x800 || (codePoint >= 0xD800 && codePoint <= 0xDFFF)) return false;
      // 三字节不能覆盖双字节范围,且不能是UTF-16代理对
    } else if (bytesNeeded === 3) {
      codePoint = ((byte & 0x07) << 18) | ((uint8[i+1] & 0x3F) << 12) | ((uint8[i+2] & 0x3F) << 6) | (uint8[i+3] & 0x3F);
      if (codePoint < 0x10000 || codePoint > 0x10FFFF) return false;
      // 四字节必须在U+10000到U+10FFFF之间
    }

    i += bytesNeeded + 1;
  }

  return true;
}

// 测试用例
isUTF8(new Blob(["�"])).then(console.log); // 合法�,返回true
isUTF8(new Blob(["example"])).then(console.log); // 正常文本,返回true
isUTF8(new Blob([new Uint8Array([0xFF])])).then(console.log); // 无效字节,返回false

实现说明

  • 直接读取原始二进制字节,避免了blob.text()的自动替换逻辑,能准确判断原始数据的UTF-8有效性。
  • 严格遵循UTF-8编码规则:
    1. 单字节字符最高位为0;
    2. 多字节字符的起始字节以110、1110、11110开头,对应2、3、4字节长度;
    3. 后续字节必须以10开头;
    4. 额外验证码点范围,排除冗余编码、代理对、超出Unicode范围的无效编码。

内容的提问来源于stack exchange,提问作者luek baja

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.22 10:54:14