You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何将HTML摄像头视频当前帧转为JS数组并裁剪为64x64x3?

实现摄像头帧转数组并裁剪为64x64x3尺寸的方案

getFrameFromVideo 函数实现

要获取视频当前帧的像素数组,借助Canvas捕获视频帧,再通过getImageData提取RGB像素数据:

function getFrameFromVideo(video) {
  // 创建临时Canvas元素
  const canvas = document.createElement('canvas');
  canvas.width = video.videoWidth;
  canvas.height = video.videoHeight;
  const ctx = canvas.getContext('2d');
  
  // 将当前视频帧绘制到Canvas
  ctx.drawImage(video, 0, 0, canvas.width, canvas.height);
  
  // 直接生成[width][height][3]的RGB二维数组
  const imageData = ctx.getImageData(0, 0, canvas.width, canvas.height);
  const frameArray = [];
  for (let y = 0; y < canvas.height; y++) {
    const row = [];
    for (let x = 0; x < canvas.width; x++) {
      const idx = (y * canvas.width + x) * 4;
      row.push([
        imageData.data[idx],     // R通道
        imageData.data[idx+1],   // G通道
        imageData.data[idx+2]    // B通道
      ]);
    }
    frameArray.push(row);
  }
  return frameArray;
}

crop_to_size 函数实现

要得到尽可能大的64x64图像,先从原始帧中居中裁剪出最大正方形(避免缩放变形),再缩放到目标尺寸:

function crop_to_size(img, targetWidth, targetHeight) {
  const originalW = img[0].length;
  const originalH = img.length;
  
  // 计算最大正方形边长,居中裁剪
  const squareSize = Math.min(originalW, originalH);
  const startX = Math.floor((originalW - squareSize) / 2);
  const startY = Math.floor((originalH - squareSize) / 2);
  
  // 裁剪出正方形区域
  const croppedSquare = [];
  for (let y = startY; y < startY + squareSize; y++) {
    croppedSquare.push(img[y].slice(startX, startX + squareSize));
  }
  
  // 将正方形缩放到目标尺寸
  const scaledImg = [];
  const scale = squareSize / targetWidth; // 目标为正方形,宽高缩放比例一致
  
  for (let y = 0; y < targetHeight; y++) {
    const row = [];
    for (let x = 0; x < targetWidth; x++) {
      // 取对应原始像素(追求精度可替换为双线性插值)
      const srcX = Math.floor(x * scale);
      const srcY = Math.floor(y * scale);
      row.push(croppedSquare[srcY][srcX]);
    }
    scaledImg.push(row);
  }
  
  return scaledImg;
}

完整整合代码

将上述函数与现有逻辑结合,补充摄像头初始化逻辑,最终代码如下:

<video id="cam" autoplay muted></video>
<button onclick="predict_img()">预测当前帧</button>

<script src="https://cdn.jsdelivr.net/npm/@tensorflow/tfjs@4.14.0/dist/tf.min.js"></script>
<script>
const video = document.getElementById("cam");
var arr_frame;
var cropped_frame;
var tensor_frame;

// 初始化摄像头
async function initCam() {
  try {
    const stream = await navigator.mediaDevices.getUserMedia({ video: true });
    video.srcObject = stream;
  } catch (err) {
    console.error("无法访问摄像头:", err);
  }
}

// 页面加载完成后启动摄像头
window.onload = initCam;

function getFrameFromVideo(video) {
  const canvas = document.createElement('canvas');
  canvas.width = video.videoWidth;
  canvas.height = video.videoHeight;
  const ctx = canvas.getContext('2d');
  
  ctx.drawImage(video, 0, 0, canvas.width, canvas.height);
  
  const imageData = ctx.getImageData(0, 0, canvas.width, canvas.height);
  const frameArray = [];
  for (let y = 0; y < canvas.height; y++) {
    const row = [];
    for (let x = 0; x < canvas.width; x++) {
      const idx = (y * canvas.width + x) * 4;
      row.push([
        imageData.data[idx],
        imageData.data[idx+1],
        imageData.data[idx+2]
      ]);
    }
    frameArray.push(row);
  }
  return frameArray;
}

function crop_to_size(img, targetWidth, targetHeight) {
  const originalW = img[0].length;
  const originalH = img.length;
  
  const squareSize = Math.min(originalW, originalH);
  const startX = Math.floor((originalW - squareSize) / 2);
  const startY = Math.floor((originalH - squareSize) / 2);
  
  const croppedSquare = [];
  for (let y = startY; y < startY + squareSize; y++) {
    croppedSquare.push(img[y].slice(startX, startX + squareSize));
  }
  
  const scaledImg = [];
  const scale = squareSize / targetWidth;
  
  for (let y = 0; y < targetHeight; y++) {
    const row = [];
    for (let x = 0; x < targetWidth; x++) {
      const srcX = Math.floor(x * scale);
      const srcY = Math.floor(y * scale);
      row.push(croppedSquare[srcY][srcX]);
    }
    scaledImg.push(row);
  }
  
  return scaledImg;
}

function predict_img() {
  // 确保视频已加载完成
  if (video.videoWidth === 0 || video.videoHeight === 0) {
    console.warn("视频未加载完成,请稍后重试");
    return;
  }
  
  arr_frame = getFrameFromVideo(video);
  cropped_frame = crop_to_size(arr_frame, 64, 64);
  // 转换为TFJS张量,形状为[64,64,3]
  tensor_frame = tf.tensor3d(cropped_frame);
  
  // 此处添加张量处理逻辑
  console.log("转换完成的张量形状:", tensor_frame.shape);
}
</script>

内容的提问来源于stack exchange,提问作者personion

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.26 14:42:04