You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用mp4box.js在浏览器中提取MP4视频的评论等元数据?

如何用mp4box.js在浏览器中提取MP4的评论等元数据

我使用mp4box.js在浏览器中读取MP4视频,想要提取评论、作者这类元数据(重点是评论,可通过VLC的Ctrl-I快捷键添加到视频中),但找不到相关指引。目前仅能关联moov > udta > meta > ilst块的大小与评论大小,但二进制内容不匹配,猜测需要进一步解析步骤。mp4boxFile.onReady无法提供用户元数据,但手动读取udta盒子能找到内容,不过里面包含其他信息,需要了解编码才能读取。

比如找到的结构示例:

--> "hdlr" --> "mdir" --> "appl" --> "ilst" --> ©cmt.data:
Hey. I am a comment that I'd like to print in the browser. 
I might have json structure inside like ("Hey": [1,2.3]). 
R (or byte val: 0x52)
©nam
J (or byte val: 0x4A)
.dataBBH gravitational lensing of gw150914_comment_from_vlc.mp4% 

--> "©too" data: Lavf58.76.108 

--> "free"

我当前使用的代码:

<!DOCTYPE html>
<html>
  <head>
    <title>MP4 Metadata Extraction</title>
  </head>
  <body>
    <input type="file" id="fileInput" accept=".mp4">
    <div id="metadataDisplay"></div>
    <script src="https://gpac.github.io/mp4box.js/dist/mp4box.all.js"></script>
    <script>
      document.getElementById("fileInput").addEventListener("change", function (e) {
        const file = e.target.files[0];
        if (file) {
          const reader = new FileReader();

          reader.onload = function (e) {
            const arrayBuffer = e.target.result;

            // Initialize mp4box
            const mp4boxFile = MP4Box.createFile();
            mp4boxFile.onReady = (info) => {
              console.log("READY");
              console.log(info);
            };
            mp4boxFile.onError = () => {
              console.log("Error");
            };
            // Create a Blob from the ArrayBuffer
            /* const blob = new Blob([arrayBuffer]);
             */
            console.log("created blob");

            // Read the Blob using mp4box
            arrayBuffer.fileStart = 0;
            mp4boxFile.appendBuffer(arrayBuffer);

            // Get metadata
            const udtaBoxes = mp4boxFile.getBoxes("udta");

            const decoder = new TextDecoder('utf-8');
            udtaBoxes.forEach((box) => {
              console.log("Box ", box);
              box.boxes.forEach((subbox) => {
                console.log("Subbox", subbox);
                if (subbox.data) {
                  console.log(decoder.decode(subbox.data));
                }
              });
            });
          };

          reader.readAsArrayBuffer(file);
        }
      });
    </script>
  </body>
</html>

解决方案

要提取评论(©cmt)、作者等元数据,需要深入遍历udta下的嵌套盒子,找到ilst容器,再解析其中的元数据原子盒子(比如©cmt、©nam、©too等)。这些原子盒子的data字段前通常有4字节的标识,需要跳过这部分再解码文本。

修改后的代码如下:

<!DOCTYPE html>
<html>
  <head>
    <title>MP4 Metadata Extraction</title>
  </head>
  <body>
    <input type="file" id="fileInput" accept=".mp4">
    <div id="metadataDisplay"></div>
    <script src="https://gpac.github.io/mp4box.js/dist/mp4box.all.js"></script>
    <script>
      document.getElementById("fileInput").addEventListener("change", function (e) {
        const file = e.target.files[0];
        if (!file) return;

        const reader = new FileReader();
        reader.onload = function (e) {
          const arrayBuffer = e.target.result;
          const mp4boxFile = MP4Box.createFile();
          
          mp4boxFile.onReady = (info) => {
            // 遍历udta盒子,深入解析元数据
            extractMetadata(mp4boxFile);
          };

          mp4boxFile.onError = () => console.error("解析MP4文件出错");
          
          arrayBuffer.fileStart = 0;
          mp4boxFile.appendBuffer(arrayBuffer);
          mp4boxFile.flush(); // 确保文件解析完成
        };

        reader.readAsArrayBuffer(file);
      });

      function extractMetadata(mp4boxFile) {
        const metadataDisplay = document.getElementById("metadataDisplay");
        const decoder = new TextDecoder('utf-8');
        const metadata = {};

        // 获取所有udta盒子
        const udtaBoxes = mp4boxFile.getBoxes("udta");
        udtaBoxes.forEach(udta => {
          // 在udta中找meta盒子
          const metaBoxes = udta.boxes.filter(box => box.type === "meta");
          metaBoxes.forEach(meta => {
            // 在meta中找ilst盒子
            const ilstBoxes = meta.boxes.filter(box => box.type === "ilst");
            ilstBoxes.forEach(ilst => {
              // 遍历ilst下的所有元数据原子盒子
              ilst.boxes.forEach(atomBox => {
                const boxType = atomBox.type;
                // 只处理带data的元数据盒子(比如©cmt、©nam等)
                if (atomBox.data) {
                  // 跳过前4字节的标识,解码剩余内容
                  const text = decoder.decode(atomBox.data.slice(4));
                  metadata[boxType] = text;
                }
              });
            });
          });
        });

        // 展示提取到的元数据
        if (Object.keys(metadata).length > 0) {
          let html = "<h3>提取到的元数据:</h3>";
          for (const [key, value] of Object.entries(metadata)) {
            html += `<p><strong>${key}:</strong> ${value.replace(/\n/g, "<br>")}</p>`;
          }
          metadataDisplay.innerHTML = html;
        } else {
          metadataDisplay.innerHTML = "<p>未找到用户元数据</p>";
        }
        console.log("提取到的元数据:", metadata);
      }
    </script>
  </body>
</html>

关键说明

  • 盒子遍历逻辑:MP4的用户元数据嵌套结构为moov > udta > meta > ilst > [元数据原子盒子],需要逐层定位到ilst容器。
  • 数据解码处理:元数据原子盒子的data字段前4字节为类型标识,需跳过这部分再用UTF-8解码,才能得到正确的文本内容。
  • flush()调用:添加mp4boxFile.flush()确保文件完全解析,避免因数据未处理完导致盒子获取不全。

内容的提问来源于stack exchange,提问作者tobiasBora

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.10 17:40:09