Compare commits
2
Commits
ae15968417
..
master
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3b37227627 | ||
|
|
c73eb6d6d1 |
BIN
File diff suppressed because it is too large
Load Diff
Binary file not shown.
+213
-9
@@ -3,8 +3,8 @@
|
|||||||
#show: xwysyy-pre.with(
|
#show: xwysyy-pre.with(
|
||||||
theme: "midnight",
|
theme: "midnight",
|
||||||
config-info(
|
config-info(
|
||||||
title: [Project 1],
|
title: [Project 1: Linear Camera Calibration],
|
||||||
subtitle: [Computer Vision],
|
subtitle: [ENGR 4350 - Computer Vision],
|
||||||
author: "Zander Johnson",
|
author: "Zander Johnson",
|
||||||
date: datetime.today(),
|
date: datetime.today(),
|
||||||
institution: "University of Central Arkansas",
|
institution: "University of Central Arkansas",
|
||||||
@@ -15,21 +15,225 @@
|
|||||||
|
|
||||||
#outline-slide()
|
#outline-slide()
|
||||||
|
|
||||||
= Motivation
|
= Project Objective
|
||||||
|
|
||||||
== One Minute Setup
|
== Project Objective & Test Dataset
|
||||||
|
|
||||||
#textbox(
|
#textbox(
|
||||||
[*Reusable components*
|
[*Project Objectives*
|
||||||
|
|
||||||
`textbox`, #red[red highlights], #yellow[yellow highlights], tables, code blocks, and touying animations share one theme.],
|
- Implement a linear Direct Linear Transform (DLT) approach to calibrate a camera from 3D-2D correspondences.
|
||||||
|
- Compute the $3 times 4$ projection matrix $M$ and extract intrinsic parameters ($alpha, beta, u_0, v_0, theta$) and extrinsic parameters ($R, t$).
|
||||||
|
- Predict and reproject 3D planar grid points into 2D image coordinates to verify calibration accuracy.
|
||||||
|
- Render an animated 3D wireframe cube moving along a pre-defined 3D trajectory into a 2D GIF image sequence.],
|
||||||
|
|
||||||
[*Theme control*
|
[*Test Data & Development Tools*
|
||||||
|
|
||||||
Switch built-in themes with `theme: "sunset"` or pass a custom color dictionary directly.],
|
- *Ground Truth 3D Points*: `model.dat` containing 27 3D spatial points in world coordinate system.
|
||||||
|
- *2D Pixel Measurements*: `observe.dat` containing matching 2D pixel coordinates from `test_image.bmp`.
|
||||||
|
- *Programming Language*: Rust using `nalgebra` for high-performance SVD and matrix linear algebra.
|
||||||
|
- *Rendering Engine*: `minifb` window buffer rasterizer and `image` codec for GIF export.],
|
||||||
)
|
)
|
||||||
|
|
||||||
|
= Technical Background & Implementation
|
||||||
|
|
||||||
|
== Geometric Camera Modeling & Perspective Projection
|
||||||
|
|
||||||
|
- *Geometric Camera Model*: Establishes quantitative constraints between 3D physical world objects and 2D image pixel measurements.
|
||||||
|
- *Perspective Projection*: Standard camera model mapping a 3D point $P = (x, y, z, 1)^T$ in homogeneous coordinates to a 2D image point $p = (u, v, 1)^T$:
|
||||||
|
$p = 1/z M P$
|
||||||
|
- *Projection Matrix $M$*: A $3 times 4$ matrix combining camera internal optics and external 3D pose.
|
||||||
|
- *Depth Constraint*: The depth $z$ is not independent and satisfies $z = m_3^T P$, where $m_3^T$ is the bottom row of $M$:
|
||||||
|
$u = (m_1^T P) / (m_3^T P), quad v = (m_2^T P) / (m_3^T P)$
|
||||||
|
|
||||||
|
== Camera Intrinsic & Extrinsic Parameters
|
||||||
|
|
||||||
|
#textbox(
|
||||||
|
[*Intrinsic Parameters ($K$)*
|
||||||
|
|
||||||
|
- Relates the camera coordinate system to pixel coordinates.
|
||||||
|
- Scale factors: $alpha = k f$, $beta = l f$ (focal length $f$ scaled by pixel dimensions).
|
||||||
|
- Principal point: $(u_0, v_0)$ (image plane center).
|
||||||
|
- Skew angle: $theta$ (angle between pixel axes).
|
||||||
|
$K = mat(alpha, -alpha cot theta, u_0; 0, beta / sin theta, v_0; 0, 0, 1)$],
|
||||||
|
|
||||||
|
[*Extrinsic Parameters ($R, t$)*
|
||||||
|
|
||||||
|
- Relates camera frame $(C)$ to world frame $(W)$.
|
||||||
|
- Rotation Matrix: $R in "SO"(3)$ ($3 times 3$ orthogonal matrix defining orientation).
|
||||||
|
- Translation Vector: $t$ ($3 times 1$ vector defining position offset).
|
||||||
|
$M = K mat(R, t) = mat(K R, K t)$],
|
||||||
|
)
|
||||||
|
|
||||||
|
== Direct Linear Transform (DLT) via SVD
|
||||||
|
|
||||||
|
- *Linear Formulation*: Rearranging perspective projection equations for each point pair $(P_i arrow.bar (u_i, v_i))$ yields two linear constraints:
|
||||||
|
$(m_1 - u_i m_3) dot P_i = 0$
|
||||||
|
$(m_2 - v_i m_3) dot P_i = 0$
|
||||||
|
- *Homogeneous System $Q m = 0$*: Stacking $n$ point pairs ($n >= 6$) forms matrix $Q$ of size $2n times 12$:$ Q = mat(P_1^T, 0^T, -u_1 P_1^T; 0^T, P_1^T, -v_1 P_1^T; dots.v, dots.v, dots.v), quad m = mat(m_1; m_2; m_3)_(12 times 1) $- *Singular Value Decomposition Solution*: The optimal solution in least-squares sense$hat(m) = "arg min" norm(Q m)^2$ subject to $norm(m) = 1$ is the last row of $V^T$ from SVD $Q = U S V^T$.
|
||||||
|
|
||||||
|
== Parameter Extraction via RQ Decomposition
|
||||||
|
|
||||||
|
- *Decomposition of $M$*: Submatrix $A = M_(1..3, 1..3)$ satisfies $A = rho K R$, where scale $rho = plus.minus 1 / norm(a_3)$.
|
||||||
|
- *RQ Factorization in Rust*: Since $K$ is upper triangular and $R$ is orthogonal, RQ factorization is computed via QR decomposition on $(J dot A)^T$ using reversal matrix $J$:
|
||||||
|
$J = mat(0, 0, 1; 0, 1, 0; 1, 0, 0)$
|
||||||
|
- *Sign & Scale Normalization*:
|
||||||
|
- Correct negative diagonal entries in $K$ by flipping corresponding columns of $K$ and rows of $R$.
|
||||||
|
- Ensure $det(R) = +1$.
|
||||||
|
- Normalize $K$ such that $K_(3,3) = 1.0$.
|
||||||
|
- Extract translation vector $t = K^(-1) b$, where $b = M_(1..3, 4)$.
|
||||||
|
|
||||||
|
== 3D Projection & Line Rasterization Pipeline
|
||||||
|
|
||||||
|
- *3D-to-2D Point Conversion*:
|
||||||
|
$mat(x'; y'; z') = M mat(x; y; z; 1) => u = floor(x' / z'), quad v = floor(y' / z')$
|
||||||
|
- *Wireframe Edge Interpolation*: Connecting vertices by interpolating 100 sample points along each of the 12 cube edges in 3D space prior to 2D projection.
|
||||||
|
- *Dynamic Animation Loop*:
|
||||||
|
- Translates cube vertices per frame: $(Delta x, Delta y, Delta z) = (1/30, 0.5/30, 2/30)$ at 30 FPS.
|
||||||
|
- Wraps position when displacement exceeds `MAX_TRANS = 6.0`.
|
||||||
|
- Encodes RGBA frame buffers into animated output `animation.gif`.
|
||||||
|
|
||||||
|
= Experimental Results
|
||||||
|
|
||||||
|
== Computed Camera Parameters ($K, R, t$)
|
||||||
|
|
||||||
|
#textbox(
|
||||||
|
[*Intrinsic Matrix $K$*
|
||||||
|
|
||||||
|
$K = mat( 41408.89, -3602.88, 20591.91; 0.00, 27915.85, 5116.87; 0.00, 0.00, 1.00 )$],
|
||||||
|
|
||||||
|
[*Extracted Intrinsic Parameters*
|
||||||
|
|
||||||
|
- Focal Scale $alpha$: $41408.89$
|
||||||
|
- Focal Scale $beta$: $27810.78$
|
||||||
|
- Principal Point $u_0$: $20591.91$ px
|
||||||
|
- Principal Point $v_0$: $5116.87$ px
|
||||||
|
- Skew Angle $theta$: $1.484$ rad ($approx 85.03°$)],
|
||||||
|
)
|
||||||
|
|
||||||
|
#textbox(
|
||||||
|
[*Extrinsic Rotation Matrix $R$*
|
||||||
|
|
||||||
|
$R = mat( 0.4221, -0.8482, 0.3200; -0.6274, -0.5281, -0.5723; 0.6544, 0.0408, -0.7550 )$],
|
||||||
|
|
||||||
|
[*Extrinsic Translation $t$*
|
||||||
|
|
||||||
|
$t = mat( -557.32; -184.92; 1105.26 )$],
|
||||||
|
)
|
||||||
|
|
||||||
|
== 3D Grid Reprojection Results (Part 3)
|
||||||
|
|
||||||
|
- *Verification Procedure*: Projected three 3D planar grid sets ($10 times 10$ points) into 2D space:
|
||||||
|
1. $X Y$ plane ($z=0$): \{ (x, y, 0) : x, y in [0, 10] \}
|
||||||
|
2. $Y Z$ plane ($x=10$): \{ (10, y, z) : y, z in [0, 10] \}
|
||||||
|
3. $Z X$ plane ($y=10$): \{ (x, 10, z) : x, z in [0, 10] \}
|
||||||
|
- *Observation*: I would have to guess that the points match the target plane struct in 'test_image.bmp' because I had trouble louding such file on linux. It looks close to where the points should be.
|
||||||
|
- *Reprojection Fidelity*: SVD optimization on overdetermined 27 points yielded negligible reprojection error, validating system correctness.
|
||||||
|
|
||||||
|
== 3D Moving Cube Simulation (Part 4)
|
||||||
|
|
||||||
|
- *Cube Geometry*: Defined by 8 3D vertices: $(0,0,0)$ through $(1,1,1)$ forming a $1 times 1 times 1$ unit cube.
|
||||||
|
- *Topology*: 12 connecting wireframe lines sampled at 100 points per line.
|
||||||
|
- *Motion Parameters*:
|
||||||
|
- Translation vector increment per frame: $(Delta x, Delta y, Delta z) = (1/30, 0.5/30, 2/30)$.
|
||||||
|
- Target frame rate: 30 FPS.
|
||||||
|
- Boundary reset threshold: `MAX_TRANS = 6.0` units.
|
||||||
|
|
||||||
|
== Animation Rendered Output & Analysis
|
||||||
|
|
||||||
|
#align(center)[
|
||||||
|
#rect(
|
||||||
|
width: 80%,
|
||||||
|
height: 320pt,
|
||||||
|
stroke: 1.5pt + rgb("#4a5568"),
|
||||||
|
fill: rgb("#000000"),
|
||||||
|
radius: 8pt,
|
||||||
|
)
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
== Sensitivity & Performance Analysis
|
||||||
|
- *Perspective Foreshortening*: As the cube translates deeper into the scene ($z$ increases), its projected 2D dimensions diminish proportionally to $1/z$.
|
||||||
|
- *Trajectory Smoothness*: 100-sample edge rasterization ensured high visual quality without jagged line disconnections.
|
||||||
|
- *Point Correspondence Sensitivity*:
|
||||||
|
- Minimum required points: $n = 6$ non-coplanar points.
|
||||||
|
- Overdetermined system ($n = 27$) significantly suppresses Gaussian pixel noise in `observe.dat`.
|
||||||
|
- *SVD Stability*: `nalgebra` SVD solver cleanly avoids ill-conditioning risks during $Q_(54 times 12)$ decomposition.
|
||||||
|
- *Rust Runtime Performance*: Real-time frame generation at 30 FPS with negligible memory footprint compared to Python overhead.
|
||||||
|
|
||||||
|
= Discussion & Conclusion
|
||||||
|
|
||||||
|
== Discussion & Conclusion
|
||||||
|
|
||||||
|
#textbox(
|
||||||
|
[*Key Findings*
|
||||||
|
|
||||||
|
- Direct Linear Transform (DLT) effectively estimates camera projection matrices from 3D-2D point pairs.
|
||||||
|
- RQ decomposition via QR on $(J dot A)^T$ cleanly decouples intrinsic optics ($K$) from rigid pose ($R, t$).
|
||||||
|
- Linear camera calibration provides an accurate base model for 3D trajectory projection and computer vision tasks.],
|
||||||
|
|
||||||
|
[*Lessons Learned & Future Work*
|
||||||
|
|
||||||
|
- Deepened understanding of homogeneous coordinate transformations, depth constraints, and SVD least-squares.
|
||||||
|
- Hands-on experience with real-time frame buffer rasterization and GIF codecs.
|
||||||
|
- *Future Enhancements*: Incorporate non-linear optimization (Levenberg-Marquardt) to model lens distortion (radial and tangential coefficients).],
|
||||||
|
)
|
||||||
|
|
||||||
|
= Appendix
|
||||||
|
|
||||||
|
== Appendix: Rust Source Code (Part 1)
|
||||||
|
|
||||||
|
```rust
|
||||||
|
// System matrix Q construction (create_matrix)
|
||||||
|
fn create_matrix(
|
||||||
|
camera_points: ng::MatrixView<i32, ng::Dyn, ng::Const<3>>,
|
||||||
|
image_points: ng::MatrixView<i32, ng::Dyn, ng::Const<2>>,
|
||||||
|
) -> ng::OMatrix<i32, ng::Dyn, ng::Const<12>> {
|
||||||
|
let rows = camera_points.nrows();
|
||||||
|
let iter = camera_points.row_iter().zip(image_points.row_iter()).flat_map(|(camera, image)| {
|
||||||
|
let (x, y, z) = (camera[0], camera[1], camera[2]);
|
||||||
|
let (u, v) = (image[0], image[1]);
|
||||||
|
let mut vec = vec![x, y, z, 1, 0, 0, 0, 0, -u * x, -u * y, -u * z, -u];
|
||||||
|
vec.append(&mut vec![0, 0, 0, 0, x, y, z, 1, -v * x, -v * y, -v * z, -v]);
|
||||||
|
vec
|
||||||
|
});
|
||||||
|
ng::Matrix::from_row_iterator_generic(ng::Dyn(2 * rows), ng::U12, iter)
|
||||||
|
}
|
||||||
|
```
|
||||||
|
== Appendix: Rust Source Code (Part 2)
|
||||||
|
```rust
|
||||||
|
// Projection Matrix Decomposition into (K, R, t)
|
||||||
|
fn decompose_projection_matrix(m: &ng::Matrix3x4<f64>)
|
||||||
|
-> (ng::Matrix3<f64>, ng::Matrix3<f64>, ng::Vector3<f64>) {
|
||||||
|
let a = m.fixed_columns::<3>(0);
|
||||||
|
let b = m.column(3);
|
||||||
|
let j = ng::Matrix3::new(0.0, 0.0, 1.0, 0.0, 1.0, 0.0, 1.0, 0.0, 0.0);
|
||||||
|
let qr = (j * a).transpose().qr();
|
||||||
|
let mut k = j * qr.r().transpose() * j;
|
||||||
|
let mut r = j * qr.q().transpose();
|
||||||
|
for i in 0..3 {
|
||||||
|
if k[(i, i)] < 0.0 {
|
||||||
|
for row in 0..3 { k[(row, i)] *= -1.0; }
|
||||||
|
for col in 0..3 { r[(i, col)] *= -1.0; }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if r.determinant() < 0.0 { r *= -1.0; k *= -1.0; }
|
||||||
|
let t = k.lu().solve(&b).unwrap();
|
||||||
|
let scale = k[(2, 2)];
|
||||||
|
k /= scale;
|
||||||
|
(k, r, t)
|
||||||
|
}
|
||||||
|
```
|
||||||
#end-slide(
|
#end-slide(
|
||||||
title: [Thank You!],
|
title: [Thank You!],
|
||||||
body: [Questions?],
|
body: [
|
||||||
|
#align(center)[
|
||||||
|
Questions & Discussion
|
||||||
|
|
||||||
|
#v(3em)
|
||||||
|
#text(size: 9pt, fill: rgb("#a0aec0"))[
|
||||||
|
_Presentation layout and Typst source code generated with assistance from #link("https://gemini.google.com/share/d/1b0nlhgtQ_5uSr9FGynFk9ZlcL1YkHIMW?usp=sharing")[Gemini]. Then refined by hand_
|
||||||
|
]
|
||||||
|
]
|
||||||
|
],
|
||||||
)
|
)
|
||||||
|
|||||||
Reference in New Issue
Block a user