如何将HTML摄像头视频当前帧转为JS数组并裁剪为64x64x3?
实现摄像头帧转数组并裁剪为64x64x3尺寸的方案
getFrameFromVideo 函数实现
要获取视频当前帧的像素数组,借助Canvas捕获视频帧,再通过getImageData提取RGB像素数据:
function getFrameFromVideo(video) { // 创建临时Canvas元素 const canvas = document.createElement('canvas'); canvas.width = video.videoWidth; canvas.height = video.videoHeight; const ctx = canvas.getContext('2d'); // 将当前视频帧绘制到Canvas ctx.drawImage(video, 0, 0, canvas.width, canvas.height); // 直接生成[width][height][3]的RGB二维数组 const imageData = ctx.getImageData(0, 0, canvas.width, canvas.height); const frameArray = []; for (let y = 0; y < canvas.height; y++) { const row = []; for (let x = 0; x < canvas.width; x++) { const idx = (y * canvas.width + x) * 4; row.push([ imageData.data[idx], // R通道 imageData.data[idx+1], // G通道 imageData.data[idx+2] // B通道 ]); } frameArray.push(row); } return frameArray; }
crop_to_size 函数实现
要得到尽可能大的64x64图像,先从原始帧中居中裁剪出最大正方形(避免缩放变形),再缩放到目标尺寸:
function crop_to_size(img, targetWidth, targetHeight) { const originalW = img[0].length; const originalH = img.length; // 计算最大正方形边长,居中裁剪 const squareSize = Math.min(originalW, originalH); const startX = Math.floor((originalW - squareSize) / 2); const startY = Math.floor((originalH - squareSize) / 2); // 裁剪出正方形区域 const croppedSquare = []; for (let y = startY; y < startY + squareSize; y++) { croppedSquare.push(img[y].slice(startX, startX + squareSize)); } // 将正方形缩放到目标尺寸 const scaledImg = []; const scale = squareSize / targetWidth; // 目标为正方形,宽高缩放比例一致 for (let y = 0; y < targetHeight; y++) { const row = []; for (let x = 0; x < targetWidth; x++) { // 取对应原始像素(追求精度可替换为双线性插值) const srcX = Math.floor(x * scale); const srcY = Math.floor(y * scale); row.push(croppedSquare[srcY][srcX]); } scaledImg.push(row); } return scaledImg; }
完整整合代码
将上述函数与现有逻辑结合,补充摄像头初始化逻辑,最终代码如下:
<video id="cam" autoplay muted></video> <button onclick="predict_img()">预测当前帧</button> <script src="https://cdn.jsdelivr.net/npm/@tensorflow/tfjs@4.14.0/dist/tf.min.js"></script> <script> const video = document.getElementById("cam"); var arr_frame; var cropped_frame; var tensor_frame; // 初始化摄像头 async function initCam() { try { const stream = await navigator.mediaDevices.getUserMedia({ video: true }); video.srcObject = stream; } catch (err) { console.error("无法访问摄像头:", err); } } // 页面加载完成后启动摄像头 window.onload = initCam; function getFrameFromVideo(video) { const canvas = document.createElement('canvas'); canvas.width = video.videoWidth; canvas.height = video.videoHeight; const ctx = canvas.getContext('2d'); ctx.drawImage(video, 0, 0, canvas.width, canvas.height); const imageData = ctx.getImageData(0, 0, canvas.width, canvas.height); const frameArray = []; for (let y = 0; y < canvas.height; y++) { const row = []; for (let x = 0; x < canvas.width; x++) { const idx = (y * canvas.width + x) * 4; row.push([ imageData.data[idx], imageData.data[idx+1], imageData.data[idx+2] ]); } frameArray.push(row); } return frameArray; } function crop_to_size(img, targetWidth, targetHeight) { const originalW = img[0].length; const originalH = img.length; const squareSize = Math.min(originalW, originalH); const startX = Math.floor((originalW - squareSize) / 2); const startY = Math.floor((originalH - squareSize) / 2); const croppedSquare = []; for (let y = startY; y < startY + squareSize; y++) { croppedSquare.push(img[y].slice(startX, startX + squareSize)); } const scaledImg = []; const scale = squareSize / targetWidth; for (let y = 0; y < targetHeight; y++) { const row = []; for (let x = 0; x < targetWidth; x++) { const srcX = Math.floor(x * scale); const srcY = Math.floor(y * scale); row.push(croppedSquare[srcY][srcX]); } scaledImg.push(row); } return scaledImg; } function predict_img() { // 确保视频已加载完成 if (video.videoWidth === 0 || video.videoHeight === 0) { console.warn("视频未加载完成,请稍后重试"); return; } arr_frame = getFrameFromVideo(video); cropped_frame = crop_to_size(arr_frame, 64, 64); // 转换为TFJS张量,形状为[64,64,3] tensor_frame = tf.tensor3d(cropped_frame); // 此处添加张量处理逻辑 console.log("转换完成的张量形状:", tensor_frame.shape); } </script>
内容的提问来源于stack exchange,提问作者personion
相关产品推荐
相关产品推荐

