You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用JavaScript实现3D透视图像转2D正交平面的四点校正裁剪

How to Convert Perspective 3D Images to Orthographic 2D View with JavaScript (4-Point Cropping)

I totally get where you're coming from—turning a skewed, perspective shot of a document into a flat, top-down view is a classic problem, and translating the math into working JS code can feel daunting. Let's break this down step by step, with a complete, runnable example.

1. Basic Setup: HTML Structure

First, we'll need a simple UI to upload images, let users select four points, and display the result. Here's the markup:

<input type="file" id="imageUpload" accept="image/*">
<canvas id="sourceCanvas"></canvas>
<canvas id="resultCanvas"></canvas>
<p>Click on the 4 corners of your document (top-left → top-right → bottom-right → bottom-left)</p>

2. Track User's 4-Point Selection

We need to let users click the source image to pick the four corners, and store those coordinates. We'll also handle image upload to load it into the source canvas:

const uploadInput = document.getElementById('imageUpload');
const sourceCanvas = document.getElementById('sourceCanvas');
const resultCanvas = document.getElementById('resultCanvas');
const ctxSource = sourceCanvas.getContext('2d');
const ctxResult = resultCanvas.getContext('2d');

let selectedPoints = [];

// Handle image upload
uploadInput.addEventListener('change', (e) => {
  const file = e.target.files[0];
  const reader = new FileReader();
  reader.onload = (event) => {
    const img = new Image();
    img.onload = () => {
      // Set canvas size to match image
      sourceCanvas.width = img.width;
      sourceCanvas.height = img.height;
      ctxSource.drawImage(img, 0, 0);
      selectedPoints = []; // Reset points when new image is loaded
    };
    img.src = event.target.result;
  };
  reader.readAsDataURL(file);
});

// Handle canvas clicks to select points
sourceCanvas.addEventListener('click', (e) => {
  if (selectedPoints.length >= 4) return; // Stop after 4 points
  
  const rect = sourceCanvas.getBoundingClientRect();
  const x = e.clientX - rect.left;
  const y = e.clientY - rect.top;
  
  selectedPoints.push({x, y});
  
  // Draw a small red circle to mark the selected point
  ctxSource.beginPath();
  ctxSource.arc(x, y, 5, 0, Math.PI * 2);
  ctxSource.fillStyle = 'red';
  ctxSource.fill();
  
  // Once 4 points are selected, run the transformation
  if (selectedPoints.length === 4) {
    performPerspectiveTransform();
  }
});

3. The Core: Calculate the Perspective Transform Matrix

This is the math heavy-lifting. We need a 3x3 homography matrix that maps your four selected points to a flat rectangle (our orthographic view). Here's a function to compute that matrix:

function getPerspectiveTransform(srcPoints, dstPoints) {
  // srcPoints: 4 points from the skewed source image
  // dstPoints: 4 points defining the target flat rectangle
  
  const matrix = [];
  for (let i = 0; i < 4; i++) {
    const [x1, y1] = [srcPoints[i].x, srcPoints[i].y];
    const [x2, y2] = [dstPoints[i].x, dstPoints[i].y];
    matrix.push([x1, y1, 1, 0, 0, 0, -x2 * x1, -x2 * y1]);
    matrix.push([0, 0, 0, x1, y1, 1, -y2 * x1, -y2 * y1]);
  }
  
  // Solve the linear system using Gaussian elimination
  const result = solveLinearSystem(matrix);
  return [
    result[0], result[1], result[2],
    result[3], result[4], result[5],
    result[6], result[7], 1
  ];
}

function solveLinearSystem(matrix) {
  const n = matrix.length;
  const m = matrix[0].length;
  
  for (let i = 0; i < n; i++) {
    // Find the pivot row (largest absolute value in current column)
    let pivot = i;
    for (let j = i; j < n; j++) {
      if (Math.abs(matrix[j][i]) > Math.abs(matrix[pivot][i])) {
        pivot = j;
      }
    }
    
    // Swap pivot row with current row
    [matrix[i], matrix[pivot]] = [matrix[pivot], matrix[i]];
    
    // Normalize the pivot row
    const div = matrix[i][i];
    for (let j = i; j < m; j++) {
      matrix[i][j] /= div;
    }
    
    // Eliminate other rows
    for (let j = 0; j < n; j++) {
      if (j !== i && matrix[j][i] !== 0) {
        const factor = matrix[j][i];
        for (let k = i; k < m; k++) {
          matrix[j][k] -= factor * matrix[i][k];
        }
      }
    }
  }
  
  // Extract the solution values
  const solution = [];
  for (let i = 0; i < n; i++) {
    solution.push(matrix[i][m - 1]);
  }
  return solution;
}

4. Apply the Transform to the Image

Once we have the matrix, we'll map each pixel from the target flat canvas back to the source image, then copy the color data. We'll use bilinear interpolation to avoid pixelated results:

function performPerspectiveTransform() {
  // Define target dimensions (preserve the document's aspect ratio)
  const targetWidth = 800;
  const sourceWidth = selectedPoints[1].x - selectedPoints[0].x;
  const sourceHeight = selectedPoints[3].y - selectedPoints[0].y;
  const targetHeight = Math.round(targetWidth * (sourceHeight / sourceWidth));
  
  resultCanvas.width = targetWidth;
  resultCanvas.height = targetHeight;
  
  // Define the four corners of our target flat rectangle
  const dstPoints = [
    {x: 0, y: 0},
    {x: targetWidth, y: 0},
    {x: targetWidth, y: targetHeight},
    {x: 0, y: targetHeight}
  ];
  
  // Get the transformation matrix
  const transformMatrix = getPerspectiveTransform(selectedPoints, dstPoints);
  
  // Get pixel data from source image
  const sourceImageData = ctxSource.getImageData(0, 0, sourceCanvas.width, sourceCanvas.height);
  const sourceData = sourceImageData.data;
  
  // Create empty pixel data for the result
  const resultImageData = ctxResult.createImageData(targetWidth, targetHeight);
  const resultData = resultImageData.data;
  
  // Iterate over every pixel in the result canvas
  for (let y = 0; y < targetHeight; y++) {
    for (let x = 0; x < targetWidth; x++) {
      // Calculate corresponding position in source image using inverse transform
      const invDenom = transformMatrix[6] * x + transformMatrix[7] * y + 1;
      const srcX = (transformMatrix[0] * x + transformMatrix[1] * y + transformMatrix[2]) / invDenom;
      const srcY = (transformMatrix[3] * x + transformMatrix[4] * y + transformMatrix[5]) / invDenom;
      
      // Only process pixels within the source image bounds
      if (srcX >= 0 && srcX < sourceCanvas.width && srcY >= 0 && srcY < sourceCanvas.height) {
        // Bilinear interpolation for smoother edges
        const xFloor = Math.floor(srcX);
        const yFloor = Math.floor(srcY);
        const xFrac = srcX - xFloor;
        const yFrac = srcY - yFloor;
        
        const idx1 = (yFloor * sourceCanvas.width + xFloor) * 4;
        const idx2 = (yFloor * sourceCanvas.width + xFloor + 1) * 4;
        const idx3 = ((yFloor + 1) * sourceCanvas.width + xFloor) * 4;
        const idx4 = ((yFloor + 1) * sourceCanvas.width + xFloor + 1) * 4;
        
        // Interpolate each color channel (RGBA)
        for (let c = 0; c < 4; c++) {
          const val1 = sourceData[idx1 + c];
          const val2 = sourceData[idx2 + c];
          const val3 = sourceData[idx3 + c];
          const val4 = sourceData[idx4 + c];
          
          const top = val1 * (1 - xFrac) + val2 * xFrac;
          const bottom = val3 * (1 - xFrac) + val4 * xFrac;
          resultData[(y * targetWidth + x) * 4 + c] = top * (1 - yFrac) + bottom * yFrac;
        }
      }
    }
  }
  
  // Draw the final flat image to the result canvas
  ctxResult.putImageData(resultImageData, 0, 0);
}

Quick Tips for Better Results

  • Stick to the Point Order: Make sure users select corners in top-left → top-right → bottom-right → bottom-left order—this correspondence is critical for the transform to work correctly.
  • Skip Interpolation for Speed: If you need faster performance (e.g., mobile devices), you can replace the bilinear interpolation with simple nearest-neighbor sampling (just round srcX and srcY to integers).
  • Adjust Target Size: Tweak the targetWidth value to get larger/smaller output images as needed.

内容的提问来源于stack exchange,提问作者P Lau

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.26 10:34:58