diff --git a/CMakeLists.txt b/CMakeLists.txt index c4f6270..62bd4f4 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -4,7 +4,7 @@ project(Vector3D) add_subdirectory(src) add_subdirectory(unit-tests) -set(CMAKE_CXX_STANDARD 11) +set(CMAKE_CXX_STANDARD 17) add_compile_options(-Wall -Wextra -Wpedantic) add_compile_options (-fdiagnostics-color=always) diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt index a1dcb05..920e8d3 100644 --- a/src/CMakeLists.txt +++ b/src/CMakeLists.txt @@ -41,6 +41,40 @@ target_link_libraries(vector-3d PRIVATE ) +# SVD +add_library(svd + STATIC + SVD.cpp +) + +target_link_libraries(svd + PUBLIC + vector-3d-intf + PRIVATE +) + +set_target_properties(svd + PROPERTIES + LINKER_LANGUAGE CXX +) + +# QR (eigenvalues/eigenvectors via implicit shifted QR iteration) +add_library(qr + STATIC + QR.cpp +) + +target_link_libraries(qr + PUBLIC + vector-3d-intf + PRIVATE +) + +set_target_properties(qr + PROPERTIES + LINKER_LANGUAGE CXX +) + # Matrix add_library(matrix STATIC @@ -51,6 +85,8 @@ target_link_libraries(matrix PUBLIC vector-3d-intf PRIVATE + svd + qr ) set_target_properties(matrix diff --git a/src/Matrix.cpp b/src/Matrix.cpp index 0449e8d..b1b6621 100644 --- a/src/Matrix.cpp +++ b/src/Matrix.cpp @@ -5,6 +5,35 @@ #include "Matrix.hpp" #endif +// Forward-declare QR::EigenQR so the Matrix::EigenQR implementation below can +// call it even when Matrix.cpp is pulled in through QR.hpp's own include chain +// (QR.cpp -> QR.hpp -> Matrix.hpp -> Matrix.cpp), where the QR namespace has +// not been declared yet at this point. If we are not already inside that +// chain, pull in the full QR library so its template definition is available. +namespace QR { +template +void EigenQR(Matrix &matrixToDecompose, Matrix &eigenVectors, + Matrix &eigenValues, uint32_t maxIterations, + float tolerance); +} +#ifndef QR_H_ +#include "QR.hpp" +#endif + +// Forward-declare SVD::SVD so the Matrix::SVD implementation below can call +// it even when Matrix.cpp is pulled in through SVD.hpp's own include chain +// (SVD.hpp -> Matrix.hpp -> Matrix.cpp), where the SVD namespace has not +// been declared yet at this point. If we are not already inside that chain, +// pull in the full SVD library so its template definition is available. +namespace SVD { +template +void SVD(Matrix &matrixToDecompose, Matrix &U, + Matrix &sigma, Matrix &Vt); +} +#ifndef SVD_H_ +#include "SVD.hpp" +#endif + #ifdef MATRIX_H_ // since the .cpp file has to be included by the .hpp file this // will evaluate to true #include "Matrix.hpp" @@ -20,7 +49,8 @@ Matrix::Matrix(const std::array &array) { } template -template +template && ...), int>> Matrix::Matrix(Args... args) { constexpr uint16_t arraySize{static_cast(rows) * static_cast(columns)}; @@ -531,7 +561,7 @@ void Matrix::QRDecomposition(Matrix &Q, Q.Fill(0); R.Fill(0); Matrix a_col, e, u, Q_column_k{}; - Matrix<1, rows> a_T, e_T{}; + Matrix<1, rows> e_T{}; for (uint8_t column = 0; column < columns; column++) { this->GetColumn(column, a_col); @@ -570,37 +600,24 @@ void Matrix::EigenQR(Matrix &eigenVectors, static_assert(rows > 1, "Matrix size must be > 1 for QR iteration"); static_assert(rows == columns, "Matrix size must be square for QR iteration"); - Matrix Ak = *this; // Copy original matrix - Matrix QQ{Matrix::Identity()}; - Matrix shift{0}; + // Delegate to the QR library: implicit shifted QR iteration with + // Wilkinson shift (see src/QR.hpp for the algorithm and conventions). + Matrix A = *this; // QR::EigenQR does not modify its input + QR::EigenQR(A, eigenVectors, eigenValues, maxIterations, tolerance); +} - for (uint32_t iter = 0; iter < maxIterations; ++iter) { - Matrix Q, R; - - // // QR shift lets us "attack" the first diagonal to speed up the algorithm - // shift = Matrix::Identity() * Ak[rows - 1][rows - 1]; - (Ak - shift).QRDecomposition(Q, R); - Ak = R * Q + shift; - QQ = QQ * Q; - - // Check convergence: off-diagonal norm - float offDiagSum = 0.0f; - for (uint32_t row = 1; row < rows; row++) { - for (uint32_t column = 0; column < row; column++) { - offDiagSum += fabs(Ak[row][column]); - } - } - - if (offDiagSum < tolerance) { - break; - } - } - - // Diagonal elements are the eigenvalues - for (uint8_t i = 0; i < rows; i++) { - eigenValues[i][0] = Ak[i][i]; - } - eigenVectors = QQ; +template +void Matrix::SVD(Matrix &U, + Matrix &sigma, + Matrix &Vt) const { + // Delegate to the SVD library (see src/SVD.hpp for the algorithm and + // conventions). NB: the fully-qualified ::SVD is required here — inside + // this member the unqualified name SVD refers to this method, which + // would shadow the namespace in a qualified lookup. SVD::SVD takes its + // input by non-const reference but does not modify it; pass a copy so + // the const-ness of *this is preserved. + Matrix A = *this; + ::SVD::SVD(A, U, sigma, Vt); } #endif // MATRIX_H_ \ No newline at end of file diff --git a/src/Matrix.hpp b/src/Matrix.hpp index d4b1798..aa0e909 100644 --- a/src/Matrix.hpp +++ b/src/Matrix.hpp @@ -3,10 +3,9 @@ #include #include #include +#include -// TODO: Add a function to calculate eigenvalues/vectors // TODO: Add a function to compute RREF -// TODO: Add a function for SVD decomposition // TODO: Add a function for LQ decomposition template class Matrix { @@ -29,9 +28,12 @@ public: Matrix(const Matrix &other); /** - * @brief Initialize a matrix directly with any number of arguments + * @brief Initialize a matrix directly with scalar values + * Uses SFINAE to only accept arithmetic types (int, float, double, etc.) */ - template Matrix(Args... args); + template && ...), int> = 0> + Matrix(Args... args); /** * @brief Create an identity matrix @@ -230,12 +232,19 @@ public: Matrix &R) const; /** - * @brief Uses QR decomposition to efficiently calculate the eigenvectors - * and values of this matrix - * @param eigenVectors a buffer that will contain the eigenvectors fo this - * matrix - * @param eigenValues a buffer that will contain the eigenValues fo this - * matrix + * @brief Calculates the eigenvectors and values of this matrix using the + * implicit shifted QR iteration (Wilkinson shift, Givens bulge chasing); + * see src/QR.hpp in the QR library for the full algorithm. + * @note For a matrix larger than 2x2 the matrix MUST be symmetric. + * A general (nonsymmetric) 2x2 is handled via the closed-form + * solution. + * @note The eigenvalues come out sorted DESCENDING (largest first); the + * eigenvector columns are swapped to match. Eigenvector signs are + * arbitrary. + * @param eigenVectors a buffer that will contain the eigenvectors of this + * matrix in its columns (column i pairs with eigenValues[i]) + * @param eigenValues a buffer that will contain the eigenvalues of this + * matrix, sorted descending * @param maxIterations the number of iterations to perform before giving * up on reaching the given tolerance * @param tolerance the level of accuracy to obtain before stopping. @@ -243,6 +252,24 @@ public: void EigenQR(Matrix &eigenVectors, Matrix &eigenValues, uint32_t maxIterations = 1000, float tolerance = 1e-6f) const; + /** + * @brief Compute the Singular Value Decomposition (SVD) of this matrix. + * + * Wrapper around SVD::SVD (see SVD.hpp for the full algorithm + * description, output storage conventions, and stack-usage notes). + * Decomposes A = U · Σ · Vᵀ where U is rows×columns, Σ is the vector + * of singular values (columns×1, sorted descending), and Vᵀ is + * columns×columns. Works for any shape (wide matrices are handled + * internally by computing SVD(Aᵀ) and swapping the factors back). + * This matrix is not modified. + * + * @param U Output: left singular vectors (rows×columns) + * @param sigma Output: singular values in descending order (columns×1) + * @param Vt Output: right singular vectors, transposed (columns×columns) + */ + void SVD(Matrix &U, Matrix &sigma, + Matrix &Vt) const; + protected: std::array matrix; diff --git a/src/QR.cpp b/src/QR.cpp new file mode 100644 index 0000000..f08713a --- /dev/null +++ b/src/QR.cpp @@ -0,0 +1,422 @@ +// This #ifndef section makes clangd happy so that it can properly do type hints +// in this file +#ifndef QR_H_ +#define QR_H_ +#include "QR.hpp" +#endif + +#ifdef QR_H_ // since the .cpp file has to be included by the .hpp file this + // will evaluate to true +#include "QR.hpp" +#include +#include + +namespace QR { + +// ============================================================================ +// QR Building Block Implementations (fully templated, heap-free) +// ============================================================================ + +/** + * GivensRotation: R * (a, b)^T = (r, 0)^T with R = [[c, s], [-s, c]], + * r = +hypot(a, b), c = a/r, s = b/r. + */ +// [[maybe_unused]]: this helper is only referenced from template +// (EigenQR/Tridiagonalize), so in translation units that include this file +// but never instantiate those templates, the definition is legitimately +// unused. The attribute silences -Wunused-function there without hiding +// real dead code in TUs that do use the algorithm. +[[maybe_unused]] static void GivensRotation(float a, float b, float &c, + float &s) { + float r = sqrtf(a * a + b * b); + if (r == 0.0f) { + c = 1.0f; + s = 0.0f; + return; + } + c = a / r; + s = b / r; +} + +/** + * ApplyRotationBothSides: A <- G A G^T (similarity transform) with + * G = [[c, s], [-s, c]] on the (i, i+1) block, i.e. G is the ZEROING + * rotation G*(x, y)^T = (r, 0)^T (the orientation used by the implicit QR + * chase: A = Q R with Q = G^T gives the next iterate R Q = G A G^T). + * With (c, s) = GivensRotation(A[i][i], A[i+1][i]) this zeroes + * A[i+1][i] after the LEFT multiplication; the right multiplication then + * chases the bulge along the superdiagonal (tridiagonal chase). + * + * A must be symmetric on entry; the result stays symmetric, so both + * triangles are written. + * + * Block updates (with a00 = A[i][i], a01 = A[i][i+1], a11 = A[i+1][i+1]): + * A[i][i] = c^2 a00 + 2 c s a01 + s^2 a11 + * A[i][i+1] = (c^2 - s^2) a01 + c s (a11 - a00) + * A[i+1][i+1] = s^2 a00 - 2 c s a01 + c^2 a11 + * Off-block updates (uniform for both sides, since the left factor G and + * the right factor G^T mix each side with the pattern (a, b) -> (c a + s b, + * -s a + c b) after transposition): + * for j not in {i, i+1}: + * A[i][j] = A[j][i] = c A[i][j] + s A[i+1][j] + * A[i+1][j] = A[j][i+1] = -s A[i][j] + c A[i+1][j] + */ +template +static void ApplyRotationBothSides(Matrix &A, uint8_t i, float c, + float s) { + float a00 = A.Get(i, i); + float a01 = A.Get(i, i + 1); + float a11 = A.Get(i + 1, i + 1); + float c2 = c * c; + float s2 = s * s; + float cs = c * s; + + A[i][i] = c2 * a00 + 2.0f * cs * a01 + s2 * a11; + A[i][i + 1] = (c2 - s2) * a01 + cs * (a11 - a00); + A[i + 1][i + 1] = s2 * a00 - 2.0f * cs * a01 + c2 * a11; + A[i + 1][i] = A[i][i + 1]; // keep both triangles in sync + + for (uint8_t j = 0; j < N; ++j) { + if (j == i || j == i + 1) + continue; + float x = A.Get(i, j); + float y = A.Get(i + 1, j); + A[i][j] = c * x + s * y; + A[j][i] = A[i][j]; + A[i + 1][j] = -s * x + c * y; + A[j][i + 1] = A[i + 1][j]; + } +} + +/** + * ApplyRotationToVectors: V <- V G^T with G = [[c, s], [-s, c]] on columns + * (i, i+1), applied to every row. G^T = [[c, -s], [s, c]], so + * V[r][i] <- c V[r][i] + s V[r][i+1] + * V[r][i+1] <- -s V[r][i] + c V[r][i+1] + * + * Convention pairing: if A evolves as A <- G A G^T (ApplyRotationBothSides + * with the SAME c, s), then V accumulates V <- V G^T. With V0 = I the + * invariant A0 = V A V^T is preserved at every step, so at convergence + * A0 = V D V^T and the columns of V are the eigenvectors. (Rationale: + * each chase step is A <- R Q with R = G A the upper-triangular factor and + * Q = G^T the orthogonal factor of A = Q R, so A = G^T A' G and the + * orthogonal factors multiply as G1^T G2^T ... in application order.) + */ +template +static void ApplyRotationToVectors(Matrix &V, uint8_t i, float c, + float s) { + for (uint8_t r = 0; r < N; ++r) { + float x = V.Get(r, i); + float y = V.Get(r, i + 1); + V[r][i] = c * x + s * y; + V[r][i + 1] = -s * x + c * y; + } +} + +/** + * WilkinsonShift: eigenvalue of [[a, b], [b, d]] closest to d. + * mu = (a+d)/2 - sign(a-d) * sqrt(((a-d)/2)^2 + b^2), sign(0) = +1. + */ +[[maybe_unused]] static float WilkinsonShift(float a, float b, float d) { + float delta = 0.5f * (a - d); + float spread = sqrtf(delta * delta + b * b); + return 0.5f * (a + d) - (delta >= 0.0f ? spread : -spread); +} + +/** + * Solve2x2Eigen: closed-form eigen-decomposition of the 2x2 block at + * (lo, lo+1). Works for symmetric blocks and for general 2x2 blocks with + * real eigenvalues (used by the N == 2 entry point). + * + * lambdaHi/lambdaLo come from the characteristic polynomial + * lambda^2 - trace*lambda + det = 0. + * The eigenvector for lambdaHi is v = (b, lambdaHi - a) (from the first + * row of (A - lambda*I)v = 0), normalized to unit length. If b == 0 the + * block is triangular and the eigenvectors are coordinate vectors: + * e1 for the larger of {a, d}, e2 for the other. + */ +template +static void Solve2x2Eigen(const Matrix &A, uint8_t lo, float &lambdaHi, + float &lambdaLo, float &c, float &s) { + float a = A.Get(lo, lo); + float b = A.Get(lo, lo + 1); + float e = A.Get(lo + 1, lo); + float d = A.Get(lo + 1, lo + 1); + + float trace = a + d; + float det = a * d - b * e; + float disc = trace * trace - 4.0f * det; + if (disc < 0.0f) + disc = 0.0f; // round-off clamp: real 2x2 blocks have disc >= 0 + float sqrtDisc = sqrtf(disc); + lambdaHi = 0.5f * (trace + sqrtDisc); + lambdaLo = 0.5f * (trace - sqrtDisc); + + if (b != 0.0f) { + float v1 = lambdaHi - a; + float n = sqrtf(b * b + v1 * v1); + c = b / n; + s = v1 / n; + } else if (a >= d) { + c = 1.0f; // e1 is the eigenvector of a = lambdaHi + s = 0.0f; + } else { + c = 0.0f; // e2 is the eigenvector of d = lambdaHi + s = 1.0f; + } +} + +/** + * Deflate: zero subdiagonal entries i in [lo, hi) whose magnitude is at or + * below tolerance * (|A[i][i]| + |A[i+1][i+1]|). + */ +template +static void Deflate(Matrix &A, uint8_t lo, uint8_t hi, float tolerance) { + for (uint8_t i = lo; i < hi; ++i) { + float t = A.Get(i + 1, i); + float scale = fabsf(A.Get(i, i)) + fabsf(A.Get(i + 1, i + 1)); + if (fabsf(t) <= tolerance * scale) { + A[i + 1][i] = 0.0f; + A[i][i + 1] = 0.0f; + } + } +} + +// ============================================================================ +// QR::EigenQR driver (implicit Wilkinson-shifted QR, bulge chasing) +// ============================================================================ + +/** + * Tridiagonalize: Givens tridiagonalization (Golub & Van Loan 8.3.1). + * + * For column k = 0..N-3 the entries A[k+2..N-1, k] are eliminated by + * rotations on (i, i+1) applied BOTTOM-UP, i = N-2 down to k+1, each + * formed from the CURRENT (already-updated) pair (A[i][k], A[i+1][k]). + * Bottom-up is essential: a top-down pass zeros A[i+1][k] with a rotation + * that would later be undone when the next rotation (i+1, i+2) is formed + * from an entry below, reviving A[i][k]. Each bottom-up rotation zeros the + * bottom of the remaining nonzero pair and the entries below stay zero + * (they are not mixed again, only rows i-1/i are mixed next). + * + * Already-tridiagonalized leading columns j < k are untouched: the mixed + * rows are both >= k+1 > j+1, so A[i][j] and A[i+1][j] are both zero there. + * The rotation on (i, i+1) also keeps column k+1..k+2 structure intact and + * does not destroy earlier columns, so after column k is done the leading + * (k+1)x(k+1) block is tridiagonal forever. + * + * On return: A is symmetric tridiagonal and A_orig = U A U^T (U = product + * of every rotation applied, in application order, as U <- U G^T). + */ +template +static void Tridiagonalize(Matrix &A, Matrix &U) { + U = Matrix{0}; + for (uint8_t i = 0; i < N; ++i) { + U[i][i] = 1.0f; + } + float c = 0.0f, s = 0.0f; + for (uint8_t k = 0; k + 2 < N; ++k) { + for (int i = (int)N - 2; i >= (int)k + 1; --i) { + GivensRotation(A.Get(i, k), A.Get(i + 1, k), c, s); + ApplyRotationBothSides(A, (uint8_t)i, c, s); + ApplyRotationToVectors(U, (uint8_t)i, c, s); + } + } +} + +/** + * See QR.hpp for the full contract. Implementation sketch: + * + * Phase 0 (N >= 3): Tridiagonalize(A, U) // A_orig = U A U^T + * V = I. + * while (hi > 0): + * Deflate(A, 0, hi, tol); peel exact-zero trailing subdiagonals (hi--) + * lo = top of the trailing unreduced block (scan down, stop at first + * exact zero subdiagonal) + * if lo == hi - 1: closed-form 2x2 eigen-solve; fold Vblock into V + * else: one implicit Wilkinson-shifted QR step: + * mu = WilkinsonShift(A[hi-1][hi-1], A[hi][hi-1], A[hi][hi]) + * A[lo..hi diagonal] -= mu // whole block! + * G1 = Givens(A[lo][lo], A[lo+1][lo]) + * for i = lo..hi-1: + * (i > lo: Gi = Givens(A[i][i], A[i+1][i])) + * ApplyRotationBothSides(A, i, Gi) // A <- Gi A Gi^T + * ApplyRotationToVectors(V, i, Gi) // V <- V Gi^T + * A[lo..hi diagonal] += mu + * eigenvalues = diag(A), sorted descending with matching V column swaps. + * eigenvectors = U * V. + * + * Invariant maintained for N >= 3 (symmetric input): A is symmetric + * tridiagonal (up to deflated zeros and ~1e-7 float roundoff in the + * off-tridiagonal corners) at the top of every loop iteration, and + * A_orig = U A U^T = (U V) A (U V)^T throughout (V = product of every + * rotation applied so far, in application order, as V <- V Gi^T). At + * convergence A = V D V^T and therefore A_orig = (U V) D (U V)^T. + * + * Orientation note: each chase rotation Gi is the ZEROING rotation + * (Gi * (x, y)^T = (r, 0)^T). The step A <- Gi A Gi^T equals R Q with + * R = Gi A upper-triangular (on the block) and Q = Gi^T -- i.e. it IS the + * standard QR update Q(A - mu I)Q^T with Q the orthogonal QR factor. The + * eigenvector accumulator therefore collects the Q factors: V <- V Gi^T. + */ +template +void EigenQR(Matrix &matrixToDecompose, Matrix &eigenVectors, + Matrix &eigenValues, uint32_t maxIterations, float tolerance) { + static_assert(N >= 2, "QR::EigenQR requires N >= 2 (N = 1 is trivial)"); + + Matrix A = matrixToDecompose; // input is not modified + Matrix V{0}; + // NB: Matrix::Identity() is a static factory that returns by value; a + // bare call would be a no-op. Set the diagonal explicitly. + for (uint8_t i = 0; i < N; ++i) { + V[i][i] = 1.0f; + } + + // ------------------------------------------------------------------ + // N == 2: closed-form solution (works for nonsymmetric input too) + // ------------------------------------------------------------------ + if (N == 2) { + float l1 = 0.0f, l2 = 0.0f, c = 0.0f, s = 0.0f; + Solve2x2Eigen(A, 0, l1, l2, c, s); + // V = I * Vblock = [[c, -s], [s, c]] + V[0][0] = c; + V[0][1] = -s; + V[1][0] = s; + V[1][1] = c; + eigenValues[0][0] = l1; + eigenValues[1][0] = l2; + for (uint8_t r = 0; r < N; ++r) + for (uint8_t col = 0; col < N; ++col) + eigenVectors[r][col] = V.Get(r, col); + return; + } + + // ------------------------------------------------------------------ + // N >= 3: implicit shifted QR iteration (symmetric input required) + // ------------------------------------------------------------------ + // Phase 0: general symmetric -> symmetric tridiagonal. The implicit + // QR bulge chase only preserves a tridiagonal structure, so the input + // must be reduced first: A_orig = U A U^T with A tridiagonal. + Matrix U{}; + Tridiagonalize(A, U); + + uint32_t iter = 0; + uint8_t hi = N - 1; + while (hi > 0) { + Deflate(A, 0, hi, tolerance); + // Peel trailing rows whose subdiagonal is exactly zero (deflated or + // already solved). Must be re-done every iteration: a peel is only + // meaningful once the subdiagonal beneath it has converged. + while (hi > 0 && A.Get(hi, hi - 1) == 0.0f) { + --hi; + } + if (hi == 0) { + break; // fully diagonal (within tolerance) + } + + // Find the top of the trailing unreduced block: scan down from hi-1 + // and stop at the first exact zero subdiagonal. A[hi][hi-1] != 0 here + // (just peeled), so lo < hi. + uint8_t lo = hi; + for (int i = (int)hi - 1; i >= 0; --i) { + if (A.Get(i + 1, i) == 0.0f) { + break; + } + lo = (uint8_t)i; + } + + if (lo + 1 == hi) { + // Trailing unreduced block is 2x2: solve in closed form. + float l1 = 0.0f, l2 = 0.0f, c = 0.0f, s = 0.0f; + Solve2x2Eigen(A, lo, l1, l2, c, s); + A[lo][lo] = l1; + A[lo + 1][lo + 1] = l2; + A[lo][lo + 1] = 0.0f; + A[lo + 1][lo] = 0.0f; + // Fold Vblock = [[c, -s], [s, c]] into V: V <- V * Vblock on + // columns (lo, lo+1). NOTE the sign convention differs from + // ApplyRotationToVectors (which applies [[c, s], [-s, c]]): + // here column 0 of Vblock is (c, s)^T, column 1 is (-s, c)^T. + for (uint8_t r = 0; r < N; ++r) { + float x = V.Get(r, lo); + float y = V.Get(r, lo + 1); + V[r][lo] = c * x + s * y; + V[r][lo + 1] = -s * x + c * y; + } + if (lo == 0) { + break; // block reached the top: matrix is fully solved + } + hi = (uint8_t)(lo - 1); + continue; + } + + // One implicit Wilkinson-shifted QR step on block [lo, hi]. + float mu = WilkinsonShift(A.Get(hi - 1, hi - 1), A.Get(hi, hi - 1), + A.Get(hi, hi)); + + // The shift applies to the ENTIRE active block: bulge chasing + // triangularizes (A - mu*I), and the first Givens rotation is formed + // from (A[lo][lo] - mu, A[lo+1][lo]). + for (uint8_t i = lo; i <= hi; ++i) { + A[i][i] -= mu; + } + + float c = 0.0f, s = 0.0f; + for (uint8_t i = lo; i < hi; ++i) { + if (i == lo) { + GivensRotation(A.Get(lo, lo), A.Get(lo + 1, lo), c, s); + } else { + GivensRotation(A.Get(i, i), A.Get(i + 1, i), c, s); + } + ApplyRotationBothSides(A, i, c, s); + ApplyRotationToVectors(V, i, c, s); + } + + for (uint8_t i = lo; i <= hi; ++i) { + A[i][i] += mu; + } + + if (++iter >= maxIterations) { + // Best-effort: fall through with the partially diagonalized A. + break; + } + } + + // ------------------------------------------------------------------ + // Collect eigenvalues and sort DESCENDING (swap eigenvectors to match) + // ------------------------------------------------------------------ + for (uint8_t i = 0; i < N; ++i) { + eigenValues[i][0] = A.Get(i, i); + } + for (uint8_t i = 0; i < N - 1; ++i) { + uint8_t k = i; + for (uint8_t j = i + 1; j < N; ++j) { + if (eigenValues.Get(j, 0) > eigenValues.Get(k, 0)) { + k = j; + } + } + if (k != i) { + float t = eigenValues[i][0]; + eigenValues[i][0] = eigenValues[k][0]; + eigenValues[k][0] = t; + for (uint8_t r = 0; r < N; ++r) { + float x = V.Get(r, i); + V[r][i] = V.Get(r, k); + V[r][k] = x; + } + } + } + + // True eigenvectors of the original matrix: U * V. Reuse the A buffer + // (its diagonal has already been collected into eigenValues). + U.Mult(V, A); + + for (uint8_t r = 0; r < N; ++r) { + for (uint8_t col = 0; col < N; ++col) { + eigenVectors[r][col] = A.Get(r, col); + } + } +} + +} // namespace QR + +#endif // QR_H_ diff --git a/src/QR.hpp b/src/QR.hpp new file mode 100644 index 0000000..02d5b6d --- /dev/null +++ b/src/QR.hpp @@ -0,0 +1,196 @@ +#pragma once +#include "Matrix.hpp" + +/** + * @brief Library that uses Matrix.hpp and computes the eigenvalues and + * eigenvectors of a square matrix with the implicit shifted QR iteration + * (Wilkinson shift, Givens bulge chasing). + * + * @note Fully templated: QR::EigenQR works for ANY Matrix with N in + * 2..255 (the uint8_t range of Matrix). There is no 5x5 limit. + * + * @note N >= 3: the input matrix MUST be symmetric (A[i][j] == A[j][i]). + * The implicit QR bulge chase maintains a symmetric tridiagonal + * structure, which only exists for symmetric input. N = 2 handles + * a general (nonsymmetric) 2x2 via the closed-form solution, so + * nonsymmetric 2x2 inputs also work. + * + * @note The input matrix is NOT modified (the iteration runs on a local + * copy), mirroring the SVD::SVD convention. + * + * @note EMBEDDED CONSTRAINT -- no heap. All working storage is stack + * allocated as templated Matrix buffers. Peak stack usage per + * call is 3 * N^2 floats (A working copy + U and V accumulators) = + * 12 * N^2 bytes: + * N = 5 -> ~0.3 KB + * N = 10 -> ~1.2 KB + * N = 20 -> ~4.8 KB + * N = 50 -> ~30 KB + * N = 100 -> ~120 KB + * N = 255 -> ~783 KB + * Instantiate only the sizes that fit your call-stack budget. + * + * @note Conventions: + * - Eigenvalues come out sorted DESCENDING (largest first); the + * eigenvector columns are swapped to match. + * - Eigenvector signs are arbitrary (v and -v are both valid); + * tests must be sign-invariant. + * - Wilkinson shift: the eigenvalue of the trailing 2x2 block + * closest to the bottom-right corner (Trefethen & Bau 13.4.1). + * + * @note Algorithm (Trefethen & Bau 13.4, Golub & Van Loan 8.4.3): + * Phase 0 (N >= 3): Givens tridiagonalization. A general symmetric + * matrix is NOT suitable for implicit QR (the bulge chase only + * preserves the tridiagonal structure), so first reduce A with + * adjacent Givens similarities A <- G A G^T (rotations applied + * BOTTOM-UP, i = N-2 down to k+1, per column k), accumulating + * U <- U G^T, until A is symmetric tridiagonal and + * A_orig = U A U^T. (N = 2 needs no reduction.) + * Phase 1: iterate until A is diagonal: + * 1. Deflate: zero out subdiagonal entries at/under the tolerance + * (scaled by the adjacent diagonal magnitudes). + * 2. Scan for the trailing unreduced block [lo, hi]. + * - block of size 1: A[hi][hi] is a converged eigenvalue, done. + * - block of size 2: solve the 2x2 eigenproblem in closed form + * and fold its eigenvector matrix into V. + * - block larger: one implicit Wilkinson-shifted QR step + * (bulge chasing with Givens rotations; the shift is applied + * to the ENTIRE active block [lo, hi], not just the trailing + * 2x2 -- the first Givens rotation must be formed from + * (A[lo][lo] - mu, A[lo+1][lo])). Every rotation is folded + * into V. + * Phase 2: eigenvalues = diag(A), sorted DESCENDING (eigenvector + * columns swapped to match), and the true eigenvectors of the + * ORIGINAL matrix are U * V. + * + * @note If maxIterations is exhausted before convergence the best-effort + * (partially diagonalized) values on the diagonal are returned. + */ +namespace QR { + +/** + * @brief Compute the eigenvalues and eigenvectors of a square matrix + * + * @param matrixToDecompose The matrix to take eigenvalues of (not + * modified). MUST be symmetric for N >= 3. + * @param eigenVectors a buffer that will contain the eigenvectors in its + * COLUMNS, sorted by descending eigenvalue (column i is the + * eigenvector for eigenValues[i]). + * @param eigenValues a buffer that will contain the eigenvalues sorted + * DESCENDING (largest first). + * @param maxIterations the number of QR steps to perform before giving up + * on reaching the given tolerance + * @param tolerance the level of accuracy to obtain before stopping; a + * subdiagonal entry is deflated when |A[i+1][i]| <= tolerance * + * (|A[i][i]| + |A[i+1][i+1]|). For float32 arithmetic, values + * around 1e-6 are a sensible choice (single-precision epsilon is + * ~1.2e-7). + */ +template +void EigenQR(Matrix &matrixToDecompose, Matrix &eigenVectors, + Matrix &eigenValues, uint32_t maxIterations, float tolerance); + +/** + * @brief Apply the similarity transform A <- G A G^T on rows/cols (i, i+1) + * + * G = [ c s ] on the (i, i+1) block, identity elsewhere, where G is the + * [ -s c ] + * ZEROING rotation (G * (x, y)^T = (r, 0)^T) -- the orientation used by + * the implicit QR chase: A = Q R with Q = G^T gives the next iterate + * R Q = G A G^T. With (c, s) = GivensRotation(A[i][i], A[i+1][i]) the + * (i+1, i) entry is zeroed by the left multiplication and the bulge is + * chased along the superdiagonal by the right one. The matrix must be + * symmetric on entry (guaranteed by construction in the QR iteration: + * symmetric input stays symmetric under similarity by an orthogonal + * matrix). Updates the full matrix, not just the tridiagonal structure. + */ +template +static void ApplyRotationBothSides(Matrix &A, uint8_t i, float c, + float s); + +/** + * @brief Accumulate eigenvectors: V <- V G^T on columns (i, i+1) + * + * G^T = [ c -s ] on columns (i, i+1), identity elsewhere, where G = + * [ s c ] + * [ c, s ] / [ -s, c ] is the zeroing rotation paired with + * ApplyRotationBothSides. Applied to all rows: + * V[r][i] -> c V[r][i] + s V[r][i+1] + * V[r][i+1] -> -s V[r][i] + c V[r][i+1] + * + * Every QR step's rotation is folded into V this way so that, together + * with A <- G A G^T, the invariant A_orig = V A V^T is preserved at every + * step (each step is A <- R Q with Q = G^T the orthogonal factor, and + * the orthogonal factors multiply as G1^T G2^T ... in application order). + * At convergence A_orig = V D V^T and the columns of V are the + * eigenvectors. + */ +template +static void ApplyRotationToVectors(Matrix &V, uint8_t i, float c, + float s); + +/** + * @brief Solve the 2x2 eigenproblem of block rows/cols (lo, lo+1) + * + * Solves the (possibly nonsymmetric) 2x2 block + * [ A[lo][lo] A[lo][lo+1] ] + * [ A[lo+1][lo] A[lo+1][lo+1] ] + * in closed form (characteristic polynomial + eigenvector back-substitution). + * + * @param A the matrix containing the block (not modified) + * @param lo the row/col index of the top-left corner of the block + * @param lambdaHi (out) the LARGER eigenvalue + * @param lambdaLo (out) the smaller eigenvalue + * @param c (out), s (out) eigenvector pair as an orthogonal matrix + * Vblock = [ c -s ] whose columns are the eigenvectors: column 0 + * [ s c ] + * (c, s) is the unit eigenvector for lambdaHi, column 1 (-s, c) is + * the unit eigenvector for lambdaLo. + * + * Note: the caller applies Vblock to its eigenvector accumulator with + * V <- V * Vblock (i.e. V[r][lo] = c*x + s*y, + * V[r][lo+1] = -s*x + c*y). Vblock has the + * SAME [ c -s; s c ] form as the G^T factor used by + * ApplyRotationToVectors, so both folding operations follow one uniform + * convention. + */ +template +static void Solve2x2Eigen(const Matrix &A, uint8_t lo, float &lambdaHi, + float &lambdaLo, float &c, float &s); + +/** + * @brief Deflate (zero out) subdiagonal entries that are at/under tolerance + * + * For each i in [lo, hi): if |A[i+1][i]| <= tolerance * + * (|A[i][i]| + |A[i+1][i+1]|), sets A[i+1][i] = A[i][i+1] = 0, splitting + * the matrix into smaller independent blocks. + */ +template +static void Deflate(Matrix &A, uint8_t lo, uint8_t hi, float tolerance); + +/** + * @brief Reduce a symmetric matrix to symmetric tridiagonal form + * + * Chases each column's entries below the subdiagonal to zero with + * adjacent Givens similarities (Golub & Van Loan 8.3.1, Givens variant): + * for column k = 0..N-3, rotations on (N-2, N-1), (N-3, N-2), ... + * (k+1, k+2) -- BOTTOM-UP, each formed from the current (A[i][k], + * A[i+1][k]) -- zero A[k+2..N-1, k] one by one. A top-down pass would not + * work: the rotation that zeros A[i+1][k] would be undone by the later + * rotation on (i+1, i+2) forming a new nonzero at A[i][k]. Each rotation + * is applied to A as a similarity (A <- G A G^T) and accumulated into U + * (U <- U G^T), so on return: + * - A is symmetric tridiagonal (off-tridiagonal entries EXACTLY zero), + * - A_orig = U A U^T (i.e. U^T A_orig U = A). + * + * U is initialized to the identity internally (its input contents are + * ignored). + */ +template +static void Tridiagonalize(Matrix &A, Matrix &U); + +} // namespace QR + +#ifndef QR_H_ +#include "QR.cpp" +#endif diff --git a/src/SVD.cpp b/src/SVD.cpp new file mode 100644 index 0000000..aea037c --- /dev/null +++ b/src/SVD.cpp @@ -0,0 +1,1004 @@ +// This #ifndef section makes clangd happy so that it can properly do type hints +// in this file +#ifndef SVD_H_ +#define SVD_H_ +#include "SVD.hpp" +#endif + +#ifdef SVD_H_ // since the .cpp file has to be included by the .hpp file this + // will evaluate to true +#include "SVD.hpp" +#include +#include + +// ============================================================================ +// SVD Building Block Implementations (fully templated, heap-free) +// +// All block operations work on Matrix working buffers with runtime +// bounds. N is the maximum matrix dimension (max(rows, columns) of the +// SVD input). No dynamic allocation is performed anywhere in this file — +// all temporaries are fixed-size arrays whose bounds derive from the +// template parameter N (a compile-time constant per instantiation). +// ============================================================================ + +[[maybe_unused]] float SVD::ComputeHouseholder(const float *x, uint8_t len, + float *v, float &alpha) { + // Compute ||x|| + float norm = 0.0f; + for (uint8_t i = 0; i < len; i++) { + norm += x[i] * x[i]; + } + norm = sqrtf(norm); + + if (norm < 1e-30f) { + alpha = 0.0f; + for (uint8_t i = 0; i < len; i++) { + v[i] = 0.0f; + } + return 0.0f; + } + + // Choose sign to avoid cancellation: alpha has opposite sign of x[0] + alpha = (x[0] >= 0.0f) ? -norm : norm; + + // v = x - alpha * e1, then normalize + float v0 = x[0] - alpha; + + // Compute ||v||² directly: v0² + x₁² + ... + xₙ₋₁² + float vv = v0 * v0; + for (uint8_t i = 1; i < len; i++) { + vv += x[i] * x[i]; + } + + if (vv < 1e-30f) { + // Already aligned with e1 + for (uint8_t i = 0; i < len; i++) { + v[i] = (i == 0) ? 1.0f : 0.0f; + } + return norm; + } + + float scale = 1.0f / sqrtf(vv); + for (uint8_t i = 0; i < len; i++) { + v[i] = (i == 0) ? v0 * scale : x[i] * scale; + } + + return norm; +} + +template +void SVD::ApplyHouseholderLeft(Matrix &W, const float *v, + uint8_t startRow, uint8_t endRow) { + uint8_t len = endRow - startRow + 1; + + // Compute vᵀv (should be 2.0 for our normalized vectors, but compute + // explicitly) + float vv = 0.0f; + for (uint8_t i = 0; i < len; i++) { + vv += v[i] * v[i]; + } + if (vv < 1e-30f) + return; + + float twoOverVv = 2.0f / vv; + + // W = (I - 2vvᵀ) · W — applied across all N columns; zero-padded + // columns map to zero under the reflection, so this is a no-op there. + for (uint8_t col = 0; col < N; col++) { + float dot = 0.0f; + for (uint8_t i = 0; i < len; i++) { + dot += v[i] * W[startRow + i][col]; + } + dot *= twoOverVv; + for (uint8_t i = 0; i < len; i++) { + W[startRow + i][col] -= dot * v[i]; + } + } +} + +template +void SVD::ApplyHouseholderRight(Matrix &W, const float *v, + uint8_t startCol, uint8_t endCol) { + uint8_t len = endCol - startCol + 1; + + float vv = 0.0f; + for (uint8_t i = 0; i < len; i++) { + vv += v[i] * v[i]; + } + if (vv < 1e-30f) + return; + + float twoOverVv = 2.0f / vv; + + // W = W · (I - 2vvᵀ) — applied across all N rows; zero-padded rows + // map to zero under the reflection, so this is a no-op there. + for (uint8_t row = 0; row < N; row++) { + float dot = 0.0f; + for (uint8_t i = 0; i < len; i++) { + dot += W[row][startCol + i] * v[i]; + } + dot *= twoOverVv; + for (uint8_t i = 0; i < len; i++) { + W[row][startCol + i] -= dot * v[i]; + } + } +} + +[[gnu::unused]] void SVD::ComputeGivens(float x, float y, float &c, float &s) { + float r = sqrtf(x * x + y * y); + + if (r < 1e-30f) { + c = 1.0f; + s = 0.0f; + return; + } + + c = x / r; + s = y / r; +} + +template +[[gnu::unused]] +void SVD::ApplyGivensLeft(Matrix &W, uint8_t i, uint8_t j, float c, + float s, uint8_t startCol, uint8_t endCol) { + // [c s] [row_i] = [new_row_i] + // [-s c] [row_j] [new_row_j] + for (uint8_t col = startCol; col <= endCol && col < N; col++) { + float t1 = W[i][col]; + float t2 = W[j][col]; + W[i][col] = c * t1 + s * t2; + W[j][col] = -s * t1 + c * t2; + } +} + +template +[[gnu::unused]] +void SVD::ApplyGivensRight(Matrix &W, uint8_t i, uint8_t j, float c, + float s, uint8_t startRow, uint8_t endRow) { + // [col_i col_j] · [c -s] = [new_col_i new_col_j] + // [s c] + for (uint8_t row = startRow; row <= endRow && row < N; row++) { + float t1 = W[row][i]; + float t2 = W[row][j]; + W[row][i] = c * t1 + s * t2; + W[row][j] = -s * t1 + c * t2; + } +} + +// ============================================================================ +// Phase 1: Householder Bidiagonalization +// ============================================================================ + +template +void SVD::Bidiagonalize(Matrix &W, uint8_t m, uint8_t q, uint8_t p, + Matrix &QL, Matrix &QR) { + // Working matrix W is m×q (padded to N×N). + // QL and QR are initialized to identity by the caller. + // We reduce W to upper bidiagonal form B using Householder reflections. + + float hhVec[N]; // Householder vector storage (N ≥ any len) + + for (uint8_t k = 0; k < p; k++) { + // --- Left Householder on column k, rows k..m-1 --- + // Zero out subdiagonal elements below B[k+1][k] + { + uint8_t len = m - k; + if (len <= 1) + continue; + + // Extract the column segment W[k..k+len-1][k] + float x[N]; + for (uint8_t i = 0; i < len; i++) { + x[i] = W[k + i][k]; + } + + // Compute Householder reflector + float alpha; + SVD::ComputeHouseholder(x, len, hhVec, alpha); + + if (alpha == 0.0f) + continue; + + // Apply H from left to W: W = H·W (all columns; padding is no-op) + SVD::ApplyHouseholderLeft(W, hhVec, k, k + len - 1); + + // Apply H from right to QL: QL = QL · H + SVD::ApplyHouseholderRight(QL, hhVec, k, k + len - 1); + } + + // --- Right Householder on row k, columns k+1..q-1 --- + // Zero out elements above the first superdiagonal in row k. + // The Householder maps [W[k][k+1], ..., W[k][q-1]] to [gamma, 0, ..., 0], + // preserving the first superdiagonal element (now gamma) and zeroing + // the rest. + { + int len = static_cast(q) - 1 - k; + if (len <= 1) + continue; // Need at least 2 elements to zero something out + + // Extract the row segment starting from column k+1 + float x[N]; + for (uint8_t i = 0; i < len; i++) { + x[i] = W[k][k + 1 + i]; + } + + // Compute Householder reflector + float alpha; + SVD::ComputeHouseholder(x, len, hhVec, alpha); + + if (alpha == 0.0f) + continue; + + // Apply H from right to W: W = W·H (columns k+1..k+len-1) + SVD::ApplyHouseholderRight(W, hhVec, k + 1, k + len); + + // Apply H from right to QR: QR = QR · H + SVD::ApplyHouseholderRight(QR, hhVec, k + 1, k + len); + } + } +} + +// ============================================================================ +// Phase 2 helpers: block solving of the bidiagonal matrix +// ============================================================================ + +template +void SVD::DeflateBidiagonal(Matrix &W, uint8_t p, float tol) { + // Zero out superdiagonal elements that are negligible relative to the + // local diagonal scale. This deflates the bidiagonal matrix into + // independent unreduced blocks, each of which can be solved on its own. + if (p < 2) + return; + for (uint8_t i = 0; i < p - 1; i++) { + float test = fabsf(W[i][i + 1]); + float scale = fabsf(W[i][i]) + fabsf(W[i + 1][i + 1]); + // Use absolute threshold for small scales to avoid division issues + if (test < tol * fmaxf(scale, 1e-10f)) { + W[i][i + 1] = 0; + } + } +} + +template +bool SVD::BidiagonalIsDiagonal(const Matrix &W, uint8_t p, float tol) { + // True when every superdiagonal element of the p×p bidiagonal matrix + // has been reduced to (numerically) zero, i.e. the diagonal holds the + // singular values and no unreduced blocks remain. + if (p < 2) + return true; + for (uint8_t i = 0; i < p - 1; i++) { + if (fabsf(W.Get(i, i + 1)) > tol * 1e-30f) { + return false; + } + } + return true; +} + +void SVD::SolveBidiagonalBlock2x2(float a, float b, float d, float Ublock[2][2], + float Vblock[2][2], float sigma[2]) { + // Full SVD of the 2×2 upper-bidiagonal block B = [[a, b], [0, d]]: + // B = Ublock · diag(sigma[0], sigma[1]) · Vblockᵀ + // where: + // - sigma[0] ≥ sigma[1] ≥ 0 + // - columns of Ublock are the left singular vectors + // - columns of Vblock are the right singular vectors (Vblock = scipy Vᵀᵀ) + // + // Uses eigen-decomposition of BᵀB = [[a², ab], [ab, b²+d²]] (symmetric + // 2×2, closed form), then uᵢ = B·vᵢ/σᵢ. + + // Singular values = sqrt of eigenvalues of BᵀB (trace/det closed form) + float trace = a * a + b * b + d * d; + float det = a * a * d * d; + float disc = trace * trace - 4.0f * det; + if (disc < 0) + disc = 0; + float sqrtDisc = sqrtf(disc); + float hi = sqrtf((trace + sqrtDisc) / 2.0f); + float lo = sqrtf((trace - sqrtDisc) / 2.0f); + if (lo > hi) { + float tmp = hi; + hi = lo; + lo = tmp; + } + sigma[0] = hi; + sigma[1] = lo; + + // Right singular vector v1: eigenvector of BᵀB for λ1 = hi². + // Null-space vector of (BᵀB − λ1·I) is [ab, λ1 − a²]. + float a2 = a * a; + float ab_val = a * b; + float e1x = ab_val; + float e1y = hi * hi - a2; + float normE1 = sqrtf(e1x * e1x + e1y * e1y); + float v1x, v1y; + if (normE1 > 1e-30f) { + v1x = e1x / normE1; + v1y = e1y / normE1; + } else { + // Degenerate (e.g. b = 0 and |a| ≥ |d|): e₁ is already an eigenvector + v1x = 1.0f; + v1y = 0.0f; + } + + // v2 is the unit vector orthogonal to v1 (completes the 2D basis) + float v2x = -v1y; + float v2y = v1x; + + // Vblock columns = right singular vectors + Vblock[0][0] = v1x; + Vblock[1][0] = v1y; + Vblock[0][1] = v2x; + Vblock[1][1] = v2y; + + // Ublock columns: uᵢ = B·vᵢ / σᵢ, with a rank-deficiency guard. + // When σᵢ ≈ 0, dividing produces inf/NaN; instead fill the U column with + // the signed orthogonal complement of the other U column (keeps Ublock + // orthogonal, and B·vᵢ ≈ 0 so any unit complement satisfies the SVD). + float u1x, u1y, u2x, u2y; + if (hi > 1e-30f) { + u1x = (a * v1x + b * v1y) / hi; + u1y = d * v1y / hi; + } else { + u1x = 1.0f; + u1y = 0.0f; + } + if (lo > 1e-30f) { + u2x = (a * v2x + b * v2y) / lo; + u2y = d * v2y / lo; + } else { + u2x = -u1y; + u2y = u1x; + } + + Ublock[0][0] = u1x; + Ublock[1][0] = u1y; + Ublock[0][1] = u2x; + Ublock[1][1] = u2y; +} + +template +void SVD::JacobiEigenSymmetric(Matrix &T, uint8_t n, float *evals, + Matrix &V) { + // Cyclic Jacobi eigenvalue algorithm on symmetric n×n matrix T (in place). + // On return: + // - T is (near-)diagonal; its diagonal entries are the eigenvalues + // - evals[i] = T[i][i] (unsorted) + // - columns of V are the corresponding eigenvectors (V is accumulated + // as V ← V·J so that T·V = V·Λ) + float jacTol = 1e-10f; + + // V starts as the identity (first n×n): eigenvector accumulator + for (uint8_t i = 0; i < n; i++) + for (uint8_t j = 0; j < n; j++) + V[i][j] = (i == j) ? 1.0f : 0.0f; + + for (uint32_t jacIter = 0; jacIter < 100; jacIter++) { + // Check convergence over ALL off-diagonal entries of the n×n part, + // not just the tridiagonal band: cyclic Jacobi on a 3x3+ block fills + // non-band entries (e.g. T[0][2]) during sweeps, so a band-only test + // can declare convergence too early. + bool converged = true; + for (uint8_t i = 0; i < n - 1 && converged; i++) { + for (uint8_t j = i + 1; j < n; j++) { + float scale = fabsf(T[i][i]) + fabsf(T[j][j]); + if (fabsf(T[i][j]) > jacTol * fmaxf(scale, 1e-30f)) { + converged = false; + break; + } + } + } + if (converged) + break; + + // Cyclic Jacobi: zero out T[p][q] for p < q + for (uint8_t p = 0; p < n - 1; p++) { + for (uint8_t q = p + 1; q < n; q++) { + float tPQ = T[p][q]; + if (fabsf(tPQ) < jacTol * 1e-30f) + continue; + + float tPP = T[p][p]; + float tQQ = T[q][q]; + float theta = (tQQ - tPP) / (2.0f * tPQ); + float t; + if (theta >= 0.0f) + t = 1.0f / (theta + sqrtf(1.0f + theta * theta)); + else + t = -1.0f / (-theta + sqrtf(1.0f + theta * theta)); + + float c = 1.0f / sqrtf(1.0f + t * t); + float s = t * c; + + // Update T + T[p][p] = tPP - t * tPQ; + T[q][q] = tQQ + t * tPQ; + T[p][q] = 0.0f; + T[q][p] = 0.0f; + + // Update other elements + for (uint8_t k = 0; k < n; k++) { + if (k == p || k == q) + continue; + float tPK = T[k][p]; + float tQK = T[k][q]; + T[k][p] = c * tPK - s * tQK; + T[p][k] = T[k][p]; + T[k][q] = s * tPK + c * tQK; + T[q][k] = T[k][q]; + } + + // Accumulate eigenvectors + for (uint8_t k = 0; k < n; k++) { + float vKP = V[k][p]; + float vKQ = V[k][q]; + V[k][p] = c * vKP - s * vKQ; + V[k][q] = s * vKP + c * vKQ; + } + } + } + } + + for (uint8_t i = 0; i < n; i++) { + // NOTE: no fabsf — the eigenvalues keep their sign (this is a + // general symmetric eigen solver, not just for PSD matrices like + // T = BᵀB). Callers needing magnitudes take them. + evals[i] = T[i][i]; + } +} + +template +void SVD::ApplyBlockFactorsToAccumulators(uint8_t blockStart, + uint8_t blockSize, + const Matrix &Ublock, + const Matrix &Vblock, + uint8_t rowsQL, uint8_t rowsQR, + Matrix &QL, + Matrix &QR) { + // Fold the block SVD factors into the accumulated Householder + // transformation matrices: + // QL[:, blockStart..blockStart+blockSize-1] ← QL[:, ...] · Ublock + // (over rows 0..rowsQL−1) + // QR[:, blockStart..blockStart+blockSize-1] ← QR[:, ...] · Vblock + // (over rows 0..rowsQR−1) + // + // rowsQL / rowsQR are the meaningful row extents of the accumulators: + // for a transposed (wide) problem W = Aᵀ has n rows, so QL carries n + // meaningful rows while in the normal case it carries m. + // + // BUG FIX (preserved): Use temporary buffers to avoid in-place + // corruption. The old code updated QL[j][blockStart+i] while still + // reading from QL[j][blockStart+k] for later i values, corrupting + // subsequent columns. + // + // Only the block columns of QL/QR change; temps hold the full N×N + // worst case (compile-time sized, heap-free). + + float newQL[N][N] = {{0}}; + for (uint8_t j = 0; j < rowsQL; j++) { + for (uint8_t i = 0; i < blockSize; i++) { + float sum = 0.0f; + for (uint8_t k = 0; k < blockSize; k++) { + sum += QL[j][blockStart + k] * Ublock.Get(k, i); + } + newQL[j][blockStart + i] = sum; + } + } + for (uint8_t j = 0; j < rowsQL; j++) + for (uint8_t i = 0; i < blockSize; i++) + QL[j][blockStart + i] = newQL[j][blockStart + i]; + + float newQR[N][N] = {{0}}; + for (uint8_t j = 0; j < rowsQR; j++) { + for (uint8_t i = 0; i < blockSize; i++) { + float sum = 0.0f; + for (uint8_t k = 0; k < blockSize; k++) { + sum += QR[j][blockStart + k] * Vblock.Get(k, i); + } + newQR[j][blockStart + i] = sum; + } + } + for (uint8_t j = 0; j < rowsQR; j++) + for (uint8_t i = 0; i < blockSize; i++) + QR[j][blockStart + i] = newQR[j][blockStart + i]; +} + +template +void SVD::SolveBidiagonalBlockJacobi(Matrix &W, uint8_t blockStart, + uint8_t blockSize, uint8_t rowsQL, + uint8_t rowsQR, Matrix &QL, + Matrix &QR) { + // Full SVD of an unreduced upper-bidiagonal block of size > 2, computed + // as the eigen-decomposition of the symmetric tridiagonal T = BᵀB: + // + // 1. Snapshot the ORIGINAL block diagonal/superdiagonal from W + // 2. Form T = BᵀB (tridiagonal symmetric) + // 3. JacobiEigenSymmetric → eigenvalues + eigenvector matrix V + // 4. Sort eigenvalues descending, reordering V columns + // 5. Compute RESIDUAL singular values: σᵢ = ‖B_orig · vᵢ‖ + // (NOT sqrt(eigenvalue) — forming BᵀB squares the condition + // number, causing float noise to swamp true tiny eigenvalues for + // rank-deficient blocks) + // 6. Re-sort σ descending, keeping V and B·v consistent + // 7. Build Ublock: uᵢ = B_orig · vᵢ / σᵢ (unit norm); for σᵢ ≈ 0, + // use Gram-Schmidt orthogonal completion against prior U columns + // 8. Fold Ublock/Vblock into QL/QR via ApplyBlockFactorsToAccumulators + // 9. Write residual norms into W's diagonal and zero the block's + // superdiagonals + + // Step 1: snapshot the original bidiagonal block from W + float d[N]; // block diagonal + float e[N - 1]; // block superdiagonal + for (uint8_t i = 0; i < blockSize; i++) { + d[i] = W[blockStart + i][blockStart + i]; + } + for (uint8_t i = 0; i < blockSize - 1; i++) { + e[i] = W[blockStart + i][blockStart + i + 1]; + } + + // Step 2: form T = BᵀB (tridiagonal) + // T[i][i] = d[i]² + e[i−1]² (e[−1] = 0) + // T[i][i+1] = d[i] · e[i] + // (e[i] lives in column i+1 of B, so it contributes to T[i+1][i+1], + // NOT T[i][i] — do not add e[i]² here.) + Matrix T{0}; + for (uint8_t i = 0; i < blockSize; i++) { + float val = d[i] * d[i]; + if (i > 0) + val += e[i - 1] * e[i - 1]; + T[i][i] = val; + if (i < blockSize - 1) { + float off = d[i] * e[i]; + T[i][i + 1] = off; + T[i + 1][i] = off; + } + } + + // Step 3: Jacobi eigen decomposition of T + float evals[N]; + Matrix V{0}; + SVD::JacobiEigenSymmetric(T, blockSize, evals, V); + // After: evals[i] = T[i][i] (unsorted), V columns are eigenvectors. + // NOTE: T is overwritten in place, so the snapshot d/e from Step 1 is + // used to form B·v later, not the (destroyed) T. + + // Step 4: sort eigenvalues descending, reordering V columns. + // Stable: swap V columns as a whole so vᵢ stays paired with evals[i]. + for (uint8_t i = 0; i < blockSize; i++) { + for (uint8_t j = i + 1; j < blockSize; j++) { + if (evals[j] > evals[i]) { + float tmpE = evals[i]; + evals[i] = evals[j]; + evals[j] = tmpE; + for (uint8_t r = 0; r < blockSize; r++) { + float tmpV = V[r][i]; + V[r][i] = V[r][j]; + V[r][j] = tmpV; + } + } + } + } + + // Step 5: residual singular values σᵢ = ‖B_orig · vᵢ‖. + // B is upper bidiagonal with diagonal d and superdiagonal e, so: + // (B·v)[k] = d[k]·v[k] + (k < bs-1 ? e[k]·v[k+1] : 0) + // Computing from the ORIGINAL block avoids the double conditioning of + // sqrt(eigenvalue-of-BᵀB), which destroys rank-deficient blocks. + float Bv[N][N]; // Bv[k][i] = (B·vᵢ)[k] + for (uint8_t i = 0; i < blockSize; i++) { + for (uint8_t k = 0; k < blockSize; k++) { + float val = d[k] * V[k][i]; + if (k < blockSize - 1) + val += e[k] * V[k + 1][i]; + Bv[k][i] = val; + } + } + + float sigma[N]; + for (uint8_t i = 0; i < blockSize; i++) { + float s2 = 0.0f; + for (uint8_t k = 0; k < blockSize; k++) + s2 += Bv[k][i] * Bv[k][i]; + sigma[i] = sqrtf(s2); + } + + // Step 6: re-sort by σ descending (σ and evals may differ in order due + // to residual computation), moving V and B·v columns together. + for (uint8_t i = 0; i < blockSize; i++) { + for (uint8_t j = i + 1; j < blockSize; j++) { + if (sigma[j] > sigma[i]) { + float ts = sigma[i]; + sigma[i] = sigma[j]; + sigma[j] = ts; + for (uint8_t r = 0; r < blockSize; r++) { + float tv = V[r][i]; + V[r][i] = V[r][j]; + V[r][j] = tv; + float tb = Bv[r][i]; + Bv[r][i] = Bv[r][j]; + Bv[r][j] = tb; + } + } + } + } + + // Step 7: build Ublock (left singular vectors). + // For non-zero σᵢ, uᵢ = B·vᵢ / σᵢ is already unit norm (up to float + // error). For zero σᵢ (rank-deficient block), B·vᵢ ≈ 0 and any unit + // vector orthogonal to the other U columns completes the SVD: + // ||B·Uᵢ|| = ||B·vᵢ|| = σᵢ = 0, and uᵢ·uᵢ = 1. + // We use Gram-Schmidt orthogonal completion (with a basis-vector seed) + // to guarantee Ublock stays orthogonal even when σᵢ is tiny-but- + // non-zero (noise-dominated) — the 2×2 solver uses the simple swap + // trick (only one complement to worry about), but the Jacobi path can + // have many small σᵢ so we re-orthogonalize against ALL prior columns. + Matrix Ublock{0}; + for (uint8_t i = 0; i < blockSize; i++) { + if (sigma[i] > 1e-30f) { + for (uint8_t k = 0; k < blockSize; k++) { + Ublock[k][i] = Bv[k][i] / sigma[i]; + } + } else { + // Seed with the first basis vector that has meaningful alignment + // with the null-space direction: prefer eᵢ (natural for a + // rank-deficient bidiagonal block), fall back to any eₖ. + float g[N] = {0}; + for (uint8_t k = 0; k < blockSize; k++) { + g[k] = (k == i % blockSize) ? 1.0f : 0.0f; + } + // Re-orthogonalize against all prior U columns (twice for float + // robustness) + for (uint8_t pass = 0; pass < 2; pass++) { + for (uint8_t j = 0; j < i; j++) { + float dot = 0.0f; + for (uint8_t k = 0; k < blockSize; k++) + dot += g[k] * Ublock[k][j]; + for (uint8_t k = 0; k < blockSize; k++) + g[k] -= dot * Ublock[k][j]; + } + } + float norm = 0.0f; + for (uint8_t k = 0; k < blockSize; k++) + norm += g[k] * g[k]; + norm = sqrtf(norm); + if (norm < 1e-30f) { + // Degenerate: re-orthogonalization collapsed; force a unit vector + g[i % blockSize] = 1.0f; + norm = 1.0f; + } + for (uint8_t k = 0; k < blockSize; k++) + Ublock[k][i] = g[k] / norm; + } + } + + // Second pass: full Gram-Schmidt re-orthogonalization of ALL Ublock + // columns against each other. This kills accumulated error from the + // Jacobi eigensolver AND from the residual-σ normalization (when two + // columns are both noise-dominated they can end up nearly parallel). + // Only touches the i-th column using columns 0..i-1 which are already + // finalized, so in-place is safe here (unlike the QL/QR fold below). + for (uint8_t i = 0; i < blockSize; i++) { + for (uint8_t pass = 0; pass < 2; pass++) { + for (uint8_t j = 0; j < i; j++) { + float dot = 0.0f; + for (uint8_t k = 0; k < blockSize; k++) + dot += Ublock[k][i] * Ublock[k][j]; + for (uint8_t k = 0; k < blockSize; k++) + Ublock[k][i] -= dot * Ublock[k][j]; + } + } + // Re-normalize after subtraction (can shrink slightly) + float norm = 0.0f; + for (uint8_t k = 0; k < blockSize; k++) + norm += Ublock[k][i] * Ublock[k][i]; + norm = sqrtf(norm); + if (norm > 1e-30f) { + for (uint8_t k = 0; k < blockSize; k++) + Ublock[k][i] /= norm; + } else { + // Collapse: pick any basis direction not already used + for (uint8_t k = 0; k < blockSize; k++) + Ublock[k][i] = (k == i) ? 1.0f : 0.0f; + } + } + + // Step 8: fold block factors into the accumulated QL/QR + SVD::ApplyBlockFactorsToAccumulators(blockStart, blockSize, Ublock, V, + rowsQL, rowsQR, QL, QR); + + // Step 9: write back the (sorted) residual σᵢ into W's diagonal and + // zero the block's superdiagonal to mark the block as fully reduced. + // ExtractAndSortSingularValues reads the diagonal of W to get σᵢ; + // writing the residual (not sqrt(evals)) keeps the diagonal consistent + // with the uᵢ/vᵢ we just installed, and is the numerically robust + // choice for rank-deficient blocks. + for (uint8_t i = 0; i < blockSize; i++) { + W[blockStart + i][blockStart + i] = sigma[i]; + if (i < blockSize - 1) + W[blockStart + i][blockStart + i + 1] = 0.0f; + } +} + +// ============================================================================ +// Phase 3: Extract singular values and assemble U, Σ, Vt +// ============================================================================ + +template +void SVD::ExtractAndSortSingularValues(Matrix &W, Matrix &sigma, + uint8_t p, Matrix &QL, + Matrix &QR) { + // Extract singular values as absolute values of diagonal elements. + // If a diagonal element is negative, flip the sign of the + // corresponding column in QL to maintain U · Σ · Vᵀ = A. This fires + // for diagonal entries that never went through a block solver (the + // solvers always write non-negative σ, so their blocks are no-ops). + for (uint8_t i = 0; i < p; i++) { + if (W[i][i] < 0.0f) { + for (uint8_t k = 0; k < N; k++) { + QL[k][i] = -QL[k][i]; + } + } + sigma[i][0] = fabsf(W[i][i]); + } + + // Sort in descending order, reordering U and V columns to match. + // Selection sort: find the max, swap it into position i. + for (uint8_t i = 0; i < p - 1; i++) { + uint8_t maxIdx = i; + float maxVal = sigma.Get(i, 0); + for (uint8_t j = i + 1; j < p; j++) { + float val = sigma.Get(j, 0); + if (val > maxVal) { + maxVal = val; + maxIdx = j; + } + } + if (maxIdx != i) { + // Swap sigma entries + float tmp = sigma.Get(i, 0); + sigma[i][0] = sigma.Get(maxIdx, 0); + sigma[maxIdx][0] = tmp; + + // Swap corresponding columns of QL (left singular vectors) + for (uint8_t k = 0; k < N; k++) { + float tmpQL = QL.Get(k, i); + QL[k][i] = QL.Get(k, maxIdx); + QL[k][maxIdx] = tmpQL; + } + + // Swap corresponding columns of QR (right singular vectors) + for (uint8_t k = 0; k < N; k++) { + float tmpQR = QR.Get(k, i); + QR[k][i] = QR.Get(k, maxIdx); + QR[k][maxIdx] = tmpQR; + } + } + } + + } + +template +void SVD::AssembleUAndVt(uint8_t m, uint8_t n, uint8_t p, + bool transposeNeeded, const Matrix &QL, + const Matrix &QR, Matrix &U, + Matrix &Vt) { + // Assemble the final U (m×k) and Vt (k×n) from the accumulated + // Householder transformations. + // + // For the non-transpose case (m >= n): + // U = first k columns of QL + // Vt = transpose of first k rows of QR (i.e. Vt[i][j] = QR[j][i]) + // + // For the transpose case (m < n, we computed SVD of Aᵀ = V·Σ·Uᵀ): + // U = first k columns of QR (right transforms of Aᵀ = left of A) + // Vt = transpose of QL (full QLᵀ, all n rows) + // + // NOTE: QL and QR were accumulated with the convention + // B = QLᵀ · A · QR (see Bidiagonalize) + // so the left singular vectors of A are columns of QL (not QLᵀ), and + // the right singular vectors of A are columns of QR. + + if (!transposeNeeded) { + // U = QL[:, 0:p] + for (uint8_t i = 0; i < m; i++) { + for (uint8_t j = 0; j < p; j++) { + U[i][j] = QL.Get(i, j); + } + for (uint8_t j = p; j < n; j++) { + U[i][j] = 0.0f; + } + } + // Vt = QRᵀ [0:p, :] + for (uint8_t i = 0; i < n; i++) { + for (uint8_t j = 0; j < n; j++) { + if (i < p) { + Vt[i][j] = QR.Get(j, i); + } else { + Vt[i][j] = 0.0f; + } + } + } + } else { + // U = QR[:, 0:p] (U is rows×columns = n×m; only first p columns + // meaningful, remaining columns zero) + for (uint8_t i = 0; i < n; i++) { + for (uint8_t j = 0; j < p; j++) { + U[i][j] = QR.Get(i, j); + } + for (uint8_t j = p; j < m; j++) { + U[i][j] = 0.0f; + } + } + // Vt = QLᵀ — the FULL m×m transpose. QL is the left factor of Aᵀ and + // has m = rows(Aᵀ) = columns(A) = N meaningful rows, so Vt (N×N) + // needs ALL m rows, not just n or p. Using a smaller bound here + // leaves trailing Vt rows zero and breaks both orthogonality and the + // U·Σ·Vᵀ = A reconstruction of the last columns of A. + for (uint8_t i = 0; i < m; i++) { + for (uint8_t j = 0; j < m; j++) { + Vt[i][j] = QL.Get(j, i); + } + } + } +} + +// ============================================================================ +// Main SVD Function +// ============================================================================ + +template +void SVD::SVD(Matrix &matrixToDecompose, + Matrix &U, Matrix &sigma, + Matrix &Vt) { + // N = max(rows, columns): the working-buffer dimension. All internal + // temporaries are Matrix on the stack (heap-free). See the + // header for the stack-usage estimate (≈ 11·N² floats at peak). + constexpr uint8_t N = (rows > columns) ? rows : columns; + + // For wide matrices (rows < columns), compute SVD of Aᵀ (which is + // tall), then swap back. This keeps all Householder logic in the + // tall case. + const bool transposeNeeded = rows < columns; + const uint8_t m = transposeNeeded ? columns : rows; + const uint8_t n = transposeNeeded ? rows : columns; + const uint8_t p = m < n ? m : n; // rank (min of m and n) + + // Working matrices (N×N, heap-free stack storage) + Matrix W{0}; + Matrix QL{0}; + Matrix QR{0}; + Matrix UInternal{0}; + Matrix VtInternal{0}; + Matrix sigmaInternal{0}; + + // Fill W with the working matrix (A or Aᵀ) + if (transposeNeeded) { + for (uint8_t i = 0; i < m; i++) { + for (uint8_t j = 0; j < n; j++) { + W[i][j] = matrixToDecompose.Get(j, i); + } + } + } else { + for (uint8_t i = 0; i < m; i++) { + for (uint8_t j = 0; j < n; j++) { + W[i][j] = matrixToDecompose.Get(i, j); + } + } + } + + // Initialize QL and QR to identity (N×N). NOTE: Matrix::Identity() is + // a static factory returning by value — a bare call would be a no-op. + for (uint8_t i = 0; i < N; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + // Phase 1: Householder bidiagonalization + SVD::Bidiagonalize(W, m, n, p, QL, QR); + + // Phase 2: reduce the bidiagonal matrix to diagonal form by solving + // each unreduced block independently. + // + // Deflation zeros out negligible superdiagonals, splitting the + // bidiagonal matrix into independent blocks. Each block of size 2 is + // solved in closed form; blocks larger than 2 use a cyclic Jacobi + // eigen-solve of BᵀB. + SVD::DeflateBidiagonal(W, p, 1e-8f); + if (!SVD::BidiagonalIsDiagonal(W, p, 1e-10f)) { + uint8_t blockStart = 0; + bool processedAny = false; + while (blockStart < p) { + // Find the end of the current unreduced block + uint8_t blockEnd = blockStart; + while (blockEnd < p - 1 && + fabsf(W.Get(blockEnd, blockEnd + 1)) > 0.0f) { + blockEnd++; + } + uint8_t blockSize = blockEnd - blockStart + 1; + + if (blockSize == 2) { + // 2×2 block: closed-form SVD + float a = W.Get(blockStart, blockStart); + float b = W.Get(blockStart, blockStart + 1); + float d = W.Get(blockStart + 1, blockStart + 1); + + float Ublock2[2][2], Vblock2[2][2], sigma2[2]; + SVD::SolveBidiagonalBlock2x2(a, b, d, Ublock2, Vblock2, sigma2); + + // Expand the 2×2 block factors into N×N working buffers, then + // fold them into QL and QR (the snapshot-before-compute pattern + // in ApplyBlockFactorsToAccumulators avoids the in-place + // corruption of the original code). + Matrix Ublock{0}; + Matrix Vblock{0}; + for (uint8_t i = 0; i < 2; i++) { + for (uint8_t j = 0; j < 2; j++) { + Ublock[i][j] = Ublock2[i][j]; + Vblock[i][j] = Vblock2[i][j]; + } + } + + SVD::ApplyBlockFactorsToAccumulators(blockStart, 2, Ublock, Vblock, + m, n, QL, QR); + + // Write the sorted singular values back into W's diagonal and + // zero the block's superdiagonal. + W[blockStart][blockStart] = sigma2[0]; + W[blockStart + 1][blockStart + 1] = sigma2[1]; + W[blockStart][blockStart + 1] = 0.0f; + } else if (blockSize > 2) { + // Larger block: cyclic Jacobi eigen-solve of BᵀB + SVD::SolveBidiagonalBlockJacobi(W, blockStart, blockSize, m, n, QL, + QR); + } + + // Move to the next block + blockStart = blockEnd + 1; + processedAny = true; + } + (void)processedAny; + } + + // Phase 3: extract and sort singular values, assemble U and Vt + SVD::ExtractAndSortSingularValues(W, sigmaInternal, p, QL, QR); + SVD::AssembleUAndVt(m, n, p, transposeNeeded, QL, QR, UInternal, + VtInternal); + + // Copy results into the output matrices + if (transposeNeeded) { + // U (m×n = rows×columns) from UInternal (N×N) + for (uint8_t i = 0; i < rows; i++) { + for (uint8_t j = 0; j < columns; j++) { + U[i][j] = UInternal.Get(i, j); + } + } + // sigma (n×1 = columns×1) from sigmaInternal + for (uint8_t i = 0; i < columns; i++) { + sigma[i][0] = sigmaInternal.Get(i, 0); + } + // Vt (n×n = columns×columns) from VtInternal + for (uint8_t i = 0; i < columns; i++) { + for (uint8_t j = 0; j < columns; j++) { + Vt[i][j] = VtInternal.Get(i, j); + } + } + } else { + // U (m×n = rows×columns) from UInternal + for (uint8_t i = 0; i < rows; i++) { + for (uint8_t j = 0; j < columns; j++) { + U[i][j] = UInternal.Get(i, j); + } + } + // sigma (n×1 = columns×1) from sigmaInternal + for (uint8_t i = 0; i < columns; i++) { + sigma[i][0] = sigmaInternal.Get(i, 0); + } + // Vt (n×n = columns×columns) from VtInternal + for (uint8_t i = 0; i < columns; i++) { + for (uint8_t j = 0; j < columns; j++) { + Vt[i][j] = VtInternal.Get(i, j); + } + } + } +} + +#endif diff --git a/src/SVD.hpp b/src/SVD.hpp new file mode 100644 index 0000000..cf9024b --- /dev/null +++ b/src/SVD.hpp @@ -0,0 +1,411 @@ +#pragma once +#include "Matrix.hpp" + +/** + * @brief library that uses Matrix.hpp and performs SVD on a matrix + * + * @note Fully templated: SVD works for ANY Matrix with R, C in + * 1..255 (the uint8_t range of Matrix). There is no 5×5 limit. + * + * @note EMBEDDED CONSTRAINT — no heap. All working storage is stack + * allocated as templated Matrix buffers where + * N = max(R, C). Peak stack usage per SVD call is + * ≈ 11·N² floats (≈ 44·N² bytes): + * N = 5 → ~1.1 KB + * N = 10 → ~4.4 KB + * N = 20 → ~18 KB + * N = 50 → ~110 KB + * N = 100 → ~440 KB + * N = 255 → ~2.9 MB + * Instantiate only the sizes that fit your call-stack budget. + */ +namespace SVD { +/** + * @brief Compute the Singular Value Decomposition (SVD) of this matrix. + * + * Decomposes A into U × Σ × Vᵀ where: + * - U is an m×k orthogonal matrix (left singular vectors) + * - Σ is a k×k diagonal matrix with non-negative singular values + * (stored as a k×1 column vector) + * - Vᵀ is a k×n orthogonal matrix (right singular vectors, transposed) + * - k = min(m, n) + * + * The decomposition satisfies: A ≈ U × diag(Σ) × Vᵀ + * Singular values are returned in descending order. + * + * Output storage conventions: + * - U: Matrix — first k columns are meaningful + * (rows k..columns−1 are zero in the wide case) + * - sigma: Matrix — first k entries are the singular + * values; entries beyond k (wide matrices only) are zero + * - Vt: Matrix — first k rows are meaningful + * (zero-padded in the tall case) + * + * For wide matrices (rows < columns) the SVD is computed on Aᵀ and the + * factors are swapped back. + * + * @tparam rows Number of rows in A (1..255) + * @tparam columns Number of columns in A (1..255) + * @param matrixToDecompose Input: the matrix A + * @param U Output: left singular vectors (rows×columns matrix) + * @param sigma Output: singular values (columns×1 vector, sorted descending) + * @param Vt Output: right singular vectors transposed (columns×columns) + * + * @note This implementation uses Householder bidiagonalization followed + * by block reduction: 2×2 blocks via closed form, larger blocks + * via cyclic Jacobi eigen-decomposition of BᵀB with residual + * singular values σᵢ = ‖B·vᵢ‖ (see docs/svd-refactor.md). + */ +template +void SVD(Matrix &matrixToDecompose, Matrix &U, + Matrix &sigma, Matrix &Vt); + +// ======================================================================== +// SVD Building Block Functions (for unit testing) +// +// Templated on the working-buffer size N. All block operations work on +// N×N matrices with runtime bounds (m, n, p, blockSize, ...) — the +// regions beyond the bounds are zero-padded working space. +// +// N is deduced from the Matrix arguments at the call site, e.g. +// Matrix<8, 8> W, QL, QR; +// SVD::Bidiagonalize(W, 6, 8, 6, QL, QR); // N = 8 deduced +// ======================================================================== + +/** + * @brief Compute a Householder reflector vector. + * + * Given input vector x, computes normalized v and scalar alpha such that: + * (I - 2·v·vᵀ) · x = [alpha, 0, 0, ...]ᵀ + * + * @param x Input vector (up to len elements) + * @param len Number of valid elements in x + * @param v Output: normalized Householder vector (length ≥ len) + * @param alpha Output: the resulting first element after reflection + * @return The norm of the input vector x + */ +static float ComputeHouseholder(const float *x, uint8_t len, float *v, + float &alpha); + +/** + * @brief Apply a Householder reflection from the left. + * + * Transforms W = (I - 2·v·vᵀ) · W where v operates on rows [startRow..endRow] + * and is applied across all N columns (zero-padded columns are a no-op). + * + * @tparam N Working buffer size + * @param W Input/output: matrix to transform + * @param v Householder vector (length = endRow - startRow + 1) + * @param startRow First row index + * @param endRow Last row index + */ +template +static void ApplyHouseholderLeft(Matrix &W, const float *v, + uint8_t startRow, uint8_t endRow); + +/** + * @brief Apply a Householder reflection from the right. + * + * Transforms W = W · (I - 2·v·vᵀ) where v operates on columns + * [startCol..endCol] and is applied across all N rows (zero-padded rows + * are a no-op). + * + * @tparam N Working buffer size + * @param W Input/output: matrix to transform + * @param v Householder vector (length = endCol - startCol + 1) + * @param startCol First column index + * @param endCol Last column index + */ +template +static void ApplyHouseholderRight(Matrix &W, const float *v, + uint8_t startCol, uint8_t endCol); + +/** + * @brief Reduce a matrix to upper bidiagonal form using Householder reflections. + * + * Applies a sequence of Householder reflections to reduce the input matrix + * W (m×q, where q ≥ p, stored in N×N working space) to upper bidiagonal + * form B (p×q), accumulating the left and right transformation matrices + * in QL and QR respectively. + * + * Algorithm (Golub-Kahan bidiagonalization): + * For k = 0 to p-1: + * 1. Left HH on column k, rows k..m-1: zero out subdiagonal below B[k+1][k] + * 2. Right HH on row k, cols k+2..q-1: zero out superdiagonal above B[k][k+1] + * + * The accumulated transformations satisfy: + * QLᵀ · W_original · QR = B (upper bidiagonal) + * + * @tparam N Working buffer size (≥ m and ≥ q) + * @param W Input/output: matrix to bidiagonalize (first m×q used) + * @param m Number of rows in the working matrix + * @param q Number of columns in the working matrix (q ≥ p) + * @param p Rank = min(m, original_columns) — number of bidiagonalization steps + * @param QL Input/output: left Householder accumulation (initialized to identity) + * @param QR Input/output: right Householder accumulation (initialized to identity) + */ +template +static void Bidiagonalize(Matrix &W, uint8_t m, uint8_t q, uint8_t p, + Matrix &QL, Matrix &QR); + +/** + * @brief Deflate a bidiagonal matrix by zeroing negligible superdiagonals. + * + * Scans the p×p upper-bidiagonal matrix stored in W and zeros out any + * superdiagonal element W[i][i+1] whose magnitude is negligible relative + * to the local diagonal scale (|W[i][i]| + |W[i+1][i+1]|). Deflating + * splits the matrix into independent unreduced blocks that can each be + * solved separately. + * + * @tparam N Working buffer size + * @param W Input/output: bidiagonal matrix (first p×p used) + * @param p Size of the bidiagonal matrix (min(rows, columns)) + * @param tol Relative deflation tolerance (e.g. 1e-8f) + */ +template +static void DeflateBidiagonal(Matrix &W, uint8_t p, float tol); + +/** + * @brief Check whether a bidiagonal matrix has fully reduced to diagonal. + * + * Returns true when every superdiagonal element of the p×p bidiagonal + * matrix in W is (numerically) zero, i.e. the diagonal entries are the + * (unsorted) singular values and no unreduced blocks remain. + * + * @tparam N Working buffer size + * @param W Input: bidiagonal matrix (first p×p used) + * @param p Size of the bidiagonal matrix (min(rows, columns)) + * @param tol Numerical zero threshold multiplier + * @return true when all superdiagonal elements are ~0 + */ +template +static bool BidiagonalIsDiagonal(const Matrix &W, uint8_t p, float tol); + +/** + * @brief Compute the full SVD of a 2×2 upper-bidiagonal block (pure). + * + * Decomposes B = [[a, b], [0, d]] as: + * B = Ublock · diag(sigma[0], sigma[1]) · Vblockᵀ + * + * Guarantees: + * - sigma[0] ≥ sigma[1] ≥ 0 (singular values, from eigenvalues of BᵀB) + * - Ublock and Vblock are orthogonal (columns are the left/right + * singular vectors respectively; Vblock = scipy's Vᵀᵀ) + * - Ublock · diag(sigma) · Vblockᵀ == B (within float tolerance) + * + * Math: eigenvectors of BᵀB = [[a², ab], [ab, b²+d²]] give the right + * singular vectors (v1 = normalize(ab, σ1²−a²) with a safe fallback when + * that vector is ~0; v2 = (−v1y, v1x)); left singular vectors are + * uᵢ = B·vᵢ/σᵢ with a rank-deficiency guard: when σᵢ ≈ 0 (i.e. ~1e-30), + * that U column is filled with the signed orthogonal complement of the + * other U column instead of dividing by ~0. + * + * @param a B[0][0] (first diagonal element) + * @param b B[0][1] (superdiagonal element) + * @param d B[1][1] (second diagonal element) + * @param Ublock Output: 2×2 left singular vectors (columns) + * @param Vblock Output: 2×2 right singular vectors (columns) + * @param sigma Output: singular values, sigma[0] ≥ sigma[1] ≥ 0 + */ +static void SolveBidiagonalBlock2x2(float a, float b, float d, + float Ublock[2][2], float Vblock[2][2], + float sigma[2]); + +/** + * @brief Cyclic Jacobi eigenvalue algorithm for a symmetric matrix (pure). + * + * Reduces symmetric n×n matrix T to (near-)diagonal form IN PLACE using + * cyclic Jacobi rotations, accumulating the eigenvectors in V. + * + * On return: + * - T's diagonal entries are the eigenvalues (off-diagonals ~0) + * - evals[i] = T[i][i], UNSORTED, SIGNED (this is a general symmetric + * eigen solver, not just for PSD matrices like T = BᵀB) + * - columns of V are the corresponding eigenvectors (T·V = V·Λ) + * + * Convergence: relative off-diagonal tolerance 1e-10, hard-capped at + * 100 sweeps. + * + * @tparam N Working buffer size (≥ n) + * @param T Input/output: symmetric matrix (first n×n used, destroyed in place) + * @param n Matrix size + * @param evals Output: eigenvalues, unsorted, length ≥ n + * @param V Output: eigenvector matrix (first n×n used), columns are eigenvectors + */ +template +static void JacobiEigenSymmetric(Matrix &T, uint8_t n, float *evals, + Matrix &V); + +/** + * @brief Fold a block SVD's factors into the QL/QR accumulators. + * + * Given the block SVD of a bidiagonal block, B = Ublock·Σ·Vblockᵀ, the + * accumulated Householder matrices must absorb the block factors: + * QL[:, blockStart..blockStart+blockSize−1] ← QL[:, ...] · Ublock + * (rows 0..rowsQL−1) + * QR[:, blockStart..blockStart+blockSize−1] ← QR[:, ...] · Vblock + * (rows 0..rowsQR−1) + * + * rowsQL / rowsQR are the meaningful row extents of the accumulators + * (e.g. for a wide matrix W = Aᵀ, QL carries n = rows(W) meaningful + * rows while QR is read back over its first m rows). + * + * In-place update is done through temporary buffers (updating QL's block + * columns while still reading them corrupts the result). + * + * @tparam N Working buffer size + * @param blockStart First column/row index of the block in W + * @param blockSize Size of the block (2, or > 2 for the Jacobi path) + * @param Ublock Left singular-vector factor of the block (first blockSize×blockSize used) + * @param Vblock Right singular-vector factor of the block (first blockSize×blockSize used) + * @param rowsQL Number of meaningful rows of QL + * @param rowsQR Number of meaningful rows of QR + * @param QL Input/output: left transformation accumulator + * @param QR Input/output: right transformation accumulator + */ +template +static void ApplyBlockFactorsToAccumulators(uint8_t blockStart, + uint8_t blockSize, + const Matrix &Ublock, + const Matrix &Vblock, + uint8_t rowsQL, uint8_t rowsQR, + Matrix &QL, + Matrix &QR); + +/** + * @brief Solve a bidiagonal block larger than 2×2 via Jacobi eigen of BᵀB. + * + * Computes the full SVD of the unreduced upper-bidiagonal block + * W[blockStart..blockStart+blockSize−1] via eigendecomposition of the + * tridiagonal T = BᵀB: + * 1. Snapshot the ORIGINAL block diagonal/superdiagonal from W + * 2. Form T = BᵀB (tridiagonal symmetric) + * 3. JacobiEigenSymmetric on T → eigenvalues (unsorted) + V + * 4. Sort eigenvalues descending, reordering V columns + * 5. Compute RESIDUAL singular values: σᵢ = ‖B_orig · vᵢ‖ + * (NOT sqrt(eigenvalue) — forming BᵀB squares the condition number, + * causing float noise to swamp true tiny eigenvalues for + * rank-deficient blocks) + * 6. Re-sort σ descending, keeping V and B·v consistent + * 7. Build Ublock: uᵢ = B_orig · vᵢ / σᵢ (unit norm); for σᵢ ≈ 0, + * use Gram-Schmidt orthogonal completion against prior U columns + * 8. Fold Ublock/Vblock into QL/QR via ApplyBlockFactorsToAccumulators + * 9. Write residual norms into W's diagonal and zero the block's + * superdiagonals + * + * @tparam N Working buffer size (≥ blockSize) + * @param W Input/output: bidiagonal matrix; the block's diagonal holds + * the singular values and its superdiagonals are zeroed on return + * @param blockStart First column/row index of the block + * @param blockSize Size of the block (> 2) + * @param rowsQL Number of meaningful rows of QL + * @param rowsQR Number of meaningful rows of QR + * @param QL Input/output: left transformation accumulator + * @param QR Input/output: right transformation accumulator + */ +template +static void SolveBidiagonalBlockJacobi(Matrix &W, uint8_t blockStart, + uint8_t blockSize, uint8_t rowsQL, + uint8_t rowsQR, Matrix &QL, + Matrix &QR); + +/** + * @brief Extract singular values from bidiagonal matrix diagonal and sort. + * + * Extracts absolute values of diagonal elements of W as singular values, + * then sorts them in descending order while reordering columns of QL + * and QR to maintain consistency. A negative diagonal element flips the + * sign of the corresponding QL column to keep A = U·Σ·Vᵀ. + * + * @tparam N Working buffer size + * @param W Input: bidiagonal matrix (first p×p used) + * @param sigma Output: sorted singular values (N×1 column vector, only first p used) + * @param p Number of singular values (min(rows, columns)) + * @param QL Input/output: left transformation matrix (modified during sort) + * @param QR Input/output: right transformation matrix (modified during sort) + */ +template +static void ExtractAndSortSingularValues(Matrix &W, Matrix &sigma, + uint8_t p, Matrix &QL, + Matrix &QR); + +/** + * @brief Assemble final U and Vt matrices from QL/QR. + * + * Computes the final left singular vectors (U) and right singular vectors + * transposed (Vt) from the accumulated Householder transformations. + * + * For non-transpose case: U = QL[:,0:p], Vt = QR[:,0:p]ᵀ + * For transpose case: U = QR[:,0:p], Vt = full QLᵀ (all n rows) + * + * @tparam N Working buffer size (≥ m and ≥ n) + * @param m Number of rows in original matrix + * @param n Number of columns in original matrix + * @param p Rank = min(m, n) + * @param transposeNeeded True if we computed SVD(Aᵀ) instead of SVD(A) + * @param QL Left Householder accumulation (N×N) + * @param QR Right Householder accumulation (N×N) + * @param U Output: left singular vectors (N×N, first m×p used) + * @param Vt Output: right singular vectors transposed (N×N, first p×n used) + */ +template +static void AssembleUAndVt(uint8_t m, uint8_t n, uint8_t p, + bool transposeNeeded, const Matrix &QL, + const Matrix &QR, Matrix &U, + Matrix &Vt); + +/** + * @brief Compute a Givens rotation that zeros out y. + * + * Computes c, s such that: + * [c s] [x] = [r] + * [-s c] [y] [0] + * where r = sqrt(x² + y²). + * + * @param x First element + * @param y Second element (to be zeroed) + * @param c Output: cosine of rotation angle + * @param s Output: sine of rotation angle + */ +static void ComputeGivens(float x, float y, float &c, float &s); + +/** + * @brief Apply a Givens rotation from the left to rows i and j. + * + * Applies [c s; -s c] to rows i, j of W (columns startCol..endCol). + * + * @tparam N Working buffer size + * @param W Input/output: matrix to transform + * @param i First row index + * @param j Second row index + * @param c Cosine of rotation angle + * @param s Sine of rotation angle + * @param startCol First column to transform + * @param endCol Last column to transform + */ +template +static void ApplyGivensLeft(Matrix &W, uint8_t i, uint8_t j, float c, + float s, uint8_t startCol, uint8_t endCol); + +/** + * @brief Apply a Givens rotation from the right to columns i and j. + * + * Applies [c -s; s c]ᵀ to columns i, j of W (rows startRow..endRow). + * + * @tparam N Working buffer size + * @param W Input/output: matrix to transform + * @param i First column index + * @param j Second column index + * @param c Cosine of rotation angle + * @param s Sine of rotation angle + * @param startRow First row to transform + * @param endRow Last row to transform + */ +template +static void ApplyGivensRight(Matrix &W, uint8_t i, uint8_t j, float c, + float s, uint8_t startRow, uint8_t endRow); +} // namespace SVD + +#ifndef SVD_H_ +#include "SVD.cpp" +#endif diff --git a/unit-tests/CMakeLists.txt b/unit-tests/CMakeLists.txt index a57babd..e348fa6 100644 --- a/unit-tests/CMakeLists.txt +++ b/unit-tests/CMakeLists.txt @@ -13,6 +13,7 @@ add_executable(matrix-tests matrix-tests.cpp) target_link_libraries(matrix-tests PRIVATE matrix + qr Catch2::Catch2WithMain ) @@ -32,4 +33,34 @@ target_link_libraries(vector-3d-tests PRIVATE vector-3d Catch2::Catch2WithMain +) + +# SVD building block tests +add_executable(svd-build-blocks-tests svd-build-blocks-tests.cpp) + +target_link_libraries(svd-build-blocks-tests + PRIVATE + matrix + svd + Catch2::Catch2WithMain +) + +# SVD integration tests +add_executable(svd-integration-test svd-integration-test.cpp) + +target_link_libraries(svd-integration-test + PRIVATE + matrix + svd + Catch2::Catch2WithMain +) + +# QR building block tests +add_executable(qr-build-blocks-tests qr-build-blocks-tests.cpp) + +target_link_libraries(qr-build-blocks-tests + PRIVATE + matrix + qr + Catch2::Catch2WithMain ) \ No newline at end of file diff --git a/unit-tests/matrix-tests.cpp b/unit-tests/matrix-tests.cpp index d9d67d4..c2eb205 100644 --- a/unit-tests/matrix-tests.cpp +++ b/unit-tests/matrix-tests.cpp @@ -4,6 +4,8 @@ // include the module you're going to test next #include "Matrix.hpp" +#include "QR.hpp" +#include "SVD.hpp" // any other libraries #include @@ -389,7 +391,7 @@ TEST_CASE("Identity Matrix", "Matrix") { if (oneColumnIndex == column) { REQUIRE_THAT(value, Catch::Matchers::WithinRel(1.0f, 1e-6f)); } else { - REQUIRE_THAT(value, Catch::Matchers::WithinRel(0.0f, 1e-6f)); + REQUIRE_THAT(value, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); } } oneColumnIndex++; @@ -406,7 +408,7 @@ TEST_CASE("Identity Matrix", "Matrix") { if (oneColumnIndex == column && row < 3) { REQUIRE_THAT(value, Catch::Matchers::WithinRel(1.0f, 1e-6f)); } else { - REQUIRE_THAT(value, Catch::Matchers::WithinRel(0.0f, 1e-6f)); + REQUIRE_THAT(value, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); } } oneColumnIndex++; @@ -422,7 +424,7 @@ TEST_CASE("Identity Matrix", "Matrix") { if (oneColumnIndex == column) { REQUIRE_THAT(value, Catch::Matchers::WithinRel(1.0f, 1e-6f)); } else { - REQUIRE_THAT(value, Catch::Matchers::WithinRel(0.0f, 1e-6f)); + REQUIRE_THAT(value, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); } } oneColumnIndex++; @@ -518,7 +520,7 @@ TEST_CASE("QR Decompositions", "Matrix") { // check that all R values are correct REQUIRE_THAT(R[0][0], Catch::Matchers::WithinRel(3.16228f, 1e-4f)); REQUIRE_THAT(R[0][1], Catch::Matchers::WithinRel(4.42719f, 1e-4f)); - REQUIRE_THAT(R[1][0], Catch::Matchers::WithinRel(0.0f, 1e-4f)); + REQUIRE_THAT(R[1][0], Catch::Matchers::WithinAbs(0.0f, 1e-4f)); REQUIRE_THAT(R[1][1], Catch::Matchers::WithinRel(0.63246f, 1e-4f)); } @@ -600,8 +602,78 @@ TEST_CASE("QR Decompositions", "Matrix") { } } +// ============================================================================ +// Eigen QR Helpers (scipy references; eigenvector checks are sign-invariant) +// ============================================================================ + +/** + * @brief Normalized eigenpair residual ||A v - lambda v|| / (||A||_F + |lambda|) + */ +template +static float eigenResidual(const Matrix &A, float lambda, + const Matrix &v) { + Matrix Av{}; + A.Mult(v, Av); + float sum = 0.0f; + float frob = 0.0f; + for (uint8_t i = 0; i < N; i++) { + float d = Av.Get(i, 0) - lambda * v.Get(i, 0); + sum += d * d; + for (uint8_t j = 0; j < N; j++) { + float a = A.Get(i, j); + frob += a * a; + } + } + float scale = sqrtf(frob) + fabsf(lambda); + return sqrtf(sum) / scale; +} + +/** + * @brief Column of the eigenvector matrix; used for the residual check. + */ +template +static Matrix eigenColumn(const Matrix &V, uint8_t col) { + Matrix v{}; + for (uint8_t i = 0; i < N; i++) { + v[i][0] = V.Get(i, col); + } + return v; +} + +/** + * @brief Check V^T V ~ I (eigenvectors orthonormal). + */ +template +static bool isOrthogonal(const Matrix &V, float tol = 1e-4f) { + Matrix Vt = V.Transpose(); + Matrix VtV{}; + Vt.Mult(V, VtV); + for (uint8_t i = 0; i < N; i++) { + for (uint8_t j = 0; j < N; j++) { + float expected = (i == j) ? 1.0f : 0.0f; + if (fabsf(VtV.Get(i, j) - expected) > tol) { + return false; + } + } + } + return true; +} + +/** + * @brief Sign-invariant component check: |actual| within max(1e-4, 1e-3*|ref|) + * of ref (ref is the ABSOLUTE value from the scipy reference). + */ +static bool componentMatches(float actual, float refAbs) { + float a = fabsf(actual); + float tol = 1e-4f; + if (refAbs * 1e-3f > tol) { + tol = refAbs * 1e-3f; + } + return fabsf(a - refAbs) <= tol; +} + TEST_CASE("Eigenvalues and Vectors", "Matrix") { - SECTION("2x2 Eigen") { + SECTION("2x2 Eigen (nonsymmetric, closed form)") { Matrix<2, 2> A{1.0f, 2.0f, 3.0f, 4.0f}; Matrix<2, 2> vectors{}; Matrix<2, 1> values{}; @@ -614,27 +686,864 @@ TEST_CASE("Eigenvalues and Vectors", "Matrix") { REQUIRE_THAT(values[1][0], Catch::Matchers::WithinRel(-0.372281f, 1e-4f)); } - SECTION("3x3 Rank Defficient Eigen") { - SKIP("Skipping this because QR decomposition isn't ready for it"); - // this symmetrix tridiagonal matrix is well behaved for testing - Matrix<3, 3> A{1, 2, 3, 4, 5, 6, 7, 8, 9}; + // Reference values: numpy.linalg.eigh on float32 matrices. + // Eigenvector component references are ABSOLUTE values (signs arbitrary). + SECTION("3x3 Symmetric Eigen") { + Matrix<3, 3> A{1, 2, 3, 2, 5, 8, 3, 8, 9}; Matrix<3, 3> vectors{}; Matrix<3, 1> values{}; - A.EigenQR(vectors, values, 1000000, 1e-8f); + A.EigenQR(vectors, values, 10000, 1e-6f); - std::string strBuf1 = ""; - vectors.ToString(strBuf1); - std::cout << "Vectors:\n" << strBuf1 << std::endl; - strBuf1 = ""; - values.ToString(strBuf1); - std::cout << "Values:\n" << strBuf1 << std::endl; + // eigenvalues (descending) + REQUIRE_THAT(values[0][0], Catch::Matchers::WithinRel(16.102417f, 1e-4f)); + REQUIRE_THAT(values[1][0], Catch::Matchers::WithinRel(0.191920f, 1e-4f)); + REQUIRE_THAT(values[2][0], Catch::Matchers::WithinRel(-1.2943381f, 1e-4f)); - REQUIRE_THAT(vectors[0][0], Catch::Matchers::WithinRel(0.23197f, 1e-4f)); - REQUIRE_THAT(vectors[1][0], Catch::Matchers::WithinRel(0.525322f, 1e-4f)); - REQUIRE_THAT(vectors[2][0], Catch::Matchers::WithinRel(0.81867f, 1e-4f)); - REQUIRE_THAT(values[0][0], Catch::Matchers::WithinRel(-1.11684f, 1e-4f)); - REQUIRE_THAT(values[1][0], Catch::Matchers::WithinRel(0.0f, 1e-4f)); - REQUIRE_THAT(values[2][0], Catch::Matchers::WithinRel(16.1168f, 1e-4f)); + // eigenvector |components| (sign-invariant) + REQUIRE(componentMatches(vectors[0][0], 0.231657207f)); + REQUIRE(componentMatches(vectors[1][0], 0.59582746f)); + REQUIRE(componentMatches(vectors[2][0], 0.768976331f)); + REQUIRE(componentMatches(vectors[0][1], 0.956842422f)); + REQUIRE(componentMatches(vectors[1][1], 0.282139271f)); + REQUIRE(componentMatches(vectors[2][1], 0.0696421042f)); + REQUIRE(componentMatches(vectors[0][2], 0.175463736f)); + REQUIRE(componentMatches(vectors[1][2], 0.75192225f)); + REQUIRE(componentMatches(vectors[2][2], 0.635472536f)); + + // eigenvectors orthonormal; eigenpair residuals small + REQUIRE(isOrthogonal(vectors)); + for (uint8_t col = 0; col < 3; col++) { + REQUIRE(eigenResidual(A, values[col][0], eigenColumn(vectors, col)) < + 1e-4f); + } } -} \ No newline at end of file + + SECTION("3x3 Rank Deficient Eigen") { + // A = v v^T with v = [1, 2, 3]: eigenvalues {14, 0, 0} + Matrix<3, 3> A{1, 2, 3, 2, 4, 6, 3, 6, 9}; + Matrix<3, 3> vectors{}; + Matrix<3, 1> values{}; + A.EigenQR(vectors, values, 10000, 1e-6f); + + REQUIRE_THAT(values[0][0], Catch::Matchers::WithinRel(14.0f, 1e-4f)); + REQUIRE_THAT(values[1][0], Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + REQUIRE_THAT(values[2][0], Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + + // dominant eigenvector is v/|v| (sign-invariant); the two null-space + // eigenvectors may be ANY orthonormal basis of the null plane, so only + // orthogonality + residuals are checked for the full matrix. + REQUIRE(componentMatches(vectors[0][0], 0.267261237f)); + REQUIRE(componentMatches(vectors[1][0], 0.534522474f)); + REQUIRE(componentMatches(vectors[2][0], 0.801783741f)); + REQUIRE(isOrthogonal(vectors)); + for (uint8_t col = 0; col < 3; col++) { + REQUIRE(eigenResidual(A, values[col][0], eigenColumn(vectors, col)) < + 1e-4f); + } + } + + SECTION("4x4 Symmetric Eigen") { + Matrix<4, 4> A{2, 1, 0, 1, 1, 3, 1, 0, 0, 1, 4, 1, 1, 0, 1, 5}; + Matrix<4, 4> vectors{}; + Matrix<4, 1> values{}; + A.EigenQR(vectors, values, 10000, 1e-6f); + + // eigenvalues are exactly {6, 4, 3, 1} + REQUIRE_THAT(values[0][0], Catch::Matchers::WithinRel(6.0f, 1e-4f)); + REQUIRE_THAT(values[1][0], Catch::Matchers::WithinRel(4.0f, 1e-4f)); + REQUIRE_THAT(values[2][0], Catch::Matchers::WithinRel(3.0f, 1e-4f)); + REQUIRE_THAT(values[3][0], Catch::Matchers::WithinRel(1.0f, 1e-4f)); + + // eigenvector |components| (sign-invariant) + REQUIRE(componentMatches(vectors[0][0], 0.258198887f)); + REQUIRE(componentMatches(vectors[1][0], 0.258198887f)); + REQUIRE(componentMatches(vectors[2][0], 0.516397774f)); + REQUIRE(componentMatches(vectors[3][0], 0.774596691f)); + REQUIRE(componentMatches(vectors[0][1], 0.0f)); + REQUIRE(componentMatches(vectors[1][1], 0.577350259f)); + REQUIRE(componentMatches(vectors[2][1], 0.577350259f)); + REQUIRE(componentMatches(vectors[3][1], 0.577350259f)); + REQUIRE(componentMatches(vectors[0][2], 0.577350259f)); + REQUIRE(componentMatches(vectors[1][2], 0.577350259f)); + REQUIRE(componentMatches(vectors[2][2], 0.577350259f)); + REQUIRE(componentMatches(vectors[3][2], 0.0f)); + REQUIRE(componentMatches(vectors[0][3], 0.774596691f)); + REQUIRE(componentMatches(vectors[1][3], 0.516397774f)); + REQUIRE(componentMatches(vectors[2][3], 0.258198887f)); + REQUIRE(componentMatches(vectors[3][3], 0.258198887f)); + + REQUIRE(isOrthogonal(vectors)); + for (uint8_t col = 0; col < 4; col++) { + REQUIRE(eigenResidual(A, values[col][0], eigenColumn(vectors, col)) < + 1e-4f); + } + } + + SECTION("5x5 Symmetric Eigen") { + Matrix<5, 5> A{3, 1, 0, 0, 1, 1, 4, 1, 0, 0, 0, 1, 5, 1, 0, 0, 0, 1, 6, 1, + 1, 0, 0, 1, 7}; + Matrix<5, 5> vectors{}; + Matrix<5, 1> values{}; + A.EigenQR(vectors, values, 10000, 1e-6f); + + // eigenvalues (descending) + REQUIRE_THAT(values[0][0], Catch::Matchers::WithinRel(7.90154457f, 1e-4f)); + REQUIRE_THAT(values[1][0], Catch::Matchers::WithinRel(6.20044184f, 1e-4f)); + REQUIRE_THAT(values[2][0], Catch::Matchers::WithinRel(5.14503145f, 1e-4f)); + REQUIRE_THAT(values[3][0], Catch::Matchers::WithinRel(3.61823463f, 1e-4f)); + REQUIRE_THAT(values[4][0], Catch::Matchers::WithinRel(2.13474774f, 1e-4f)); + + // eigenvector |components| (sign-invariant) + REQUIRE(componentMatches(vectors[0][0], 0.182430908f)); + REQUIRE(componentMatches(vectors[1][0], 0.102749094f)); + REQUIRE(componentMatches(vectors[2][0], 0.21844925f)); + REQUIRE(componentMatches(vectors[3][0], 0.531091094f)); + REQUIRE(componentMatches(vectors[4][0], 0.791444063f)); + REQUIRE(componentMatches(vectors[0][1], 0.0877681747f)); + REQUIRE(componentMatches(vectors[1][1], 0.245861098f)); + REQUIRE(componentMatches(vectors[2][1], 0.628771126f)); + REQUIRE(componentMatches(vectors[3][1], 0.508941948f)); + REQUIRE(componentMatches(vectors[4][1], 0.526758015f)); + REQUIRE(componentMatches(vectors[0][2], 0.349721253f)); + REQUIRE(componentMatches(vectors[1][2], 0.628706098f)); + REQUIRE(componentMatches(vectors[2][2], 0.370167077f)); + REQUIRE(componentMatches(vectors[3][2], 0.575020194f)); + REQUIRE(componentMatches(vectors[4][2], 0.121457018f)); + REQUIRE(componentMatches(vectors[0][3], 0.429638386f)); + REQUIRE(componentMatches(vectors[1][3], 0.498553723f)); + REQUIRE(componentMatches(vectors[2][3], 0.619968951f)); + REQUIRE(componentMatches(vectors[3][3], 0.35809797f)); + REQUIRE(componentMatches(vectors[4][3], 0.232936427f)); + REQUIRE(componentMatches(vectors[0][4], 0.807540476f)); + REQUIRE(componentMatches(vectors[1][4], 0.534011006f)); + REQUIRE(componentMatches(vectors[2][4], 0.188524753f)); + REQUIRE(componentMatches(vectors[3][4], 0.00615991838f)); + REQUIRE(componentMatches(vectors[4][4], 0.164715111f)); + + REQUIRE(isOrthogonal(vectors)); + for (uint8_t col = 0; col < 5; col++) { + REQUIRE(eigenResidual(A, values[col][0], eigenColumn(vectors, col)) < + 1e-4f); + } + } + + SECTION("6x6 Symmetric Eigen") { + Matrix<6, 6> A{4, 1, 0, 0, 0, 1, 1, 5, 1, 0, 0, 0, 0, 1, 6, 1, 0, 0, 0, 0, + 1, 7, 1, 0, 0, 0, 0, 1, 8, 1, 1, 0, 0, 0, 1, 3}; + Matrix<6, 6> vectors{}; + Matrix<6, 1> values{}; + A.EigenQR(vectors, values, 10000, 1e-6f); + + // eigenvalues (descending) + REQUIRE_THAT(values[0][0], Catch::Matchers::WithinRel(8.86080551f, 1e-4f)); + REQUIRE_THAT(values[1][0], Catch::Matchers::WithinRel(7.25410175f, 1e-4f)); + REQUIRE_THAT(values[2][0], Catch::Matchers::WithinRel(6.11490774f, 1e-4f)); + REQUIRE_THAT(values[3][0], Catch::Matchers::WithinRel(4.88509226f, 1e-4f)); + REQUIRE_THAT(values[4][0], Catch::Matchers::WithinRel(3.74589825f, 1e-4f)); + REQUIRE_THAT(values[5][0], Catch::Matchers::WithinRel(2.13919425f, 1e-4f)); + + // eigenvector |components| (sign-invariant) + REQUIRE(componentMatches(vectors[0][0], 0.0430923924f)); + REQUIRE(componentMatches(vectors[1][0], 0.0662503168f)); + REQUIRE(componentMatches(vectors[2][0], 0.212687209f)); + REQUIRE(componentMatches(vectors[3][0], 0.542206466f)); + REQUIRE(componentMatches(vectors[4][0], 0.7962538f)); + REQUIRE(componentMatches(vectors[5][0], 0.143213451f)); + REQUIRE(componentMatches(vectors[0][1], 0.0623276457f)); + REQUIRE(componentMatches(vectors[1][1], 0.307613879f)); + REQUIRE(componentMatches(vectors[2][1], 0.631065309f)); + REQUIRE(componentMatches(vectors[3][1], 0.483806193f)); + REQUIRE(componentMatches(vectors[4][1], 0.508129358f)); + REQUIRE(componentMatches(vectors[5][1], 0.104793385f)); + REQUIRE(componentMatches(vectors[0][2], 0.374228716f)); + REQUIRE(componentMatches(vectors[1][2], 0.605694294f)); + REQUIRE(componentMatches(vectors[2][2], 0.301064402f)); + REQUIRE(componentMatches(vectors[3][2], 0.571099699f)); + REQUIRE(componentMatches(vectors[4][2], 0.204411641f)); + REQUIRE(componentMatches(vectors[5][2], 0.185764849f)); + REQUIRE(componentMatches(vectors[0][3], 0.571099699f)); + REQUIRE(componentMatches(vectors[1][3], 0.301064402f)); + REQUIRE(componentMatches(vectors[2][3], 0.605694294f)); + REQUIRE(componentMatches(vectors[3][3], 0.374228716f)); + REQUIRE(componentMatches(vectors[4][3], 0.185764849f)); + REQUIRE(componentMatches(vectors[5][3], 0.204411641f)); + REQUIRE(componentMatches(vectors[0][4], 0.483806193f)); + REQUIRE(componentMatches(vectors[1][4], 0.631065309f)); + REQUIRE(componentMatches(vectors[2][4], 0.307613879f)); + REQUIRE(componentMatches(vectors[3][4], 0.0623276457f)); + REQUIRE(componentMatches(vectors[4][4], 0.104793385f)); + REQUIRE(componentMatches(vectors[5][4], 0.508129358f)); + REQUIRE(componentMatches(vectors[0][5], 0.542206466f)); + REQUIRE(componentMatches(vectors[1][5], 0.212687209f)); + REQUIRE(componentMatches(vectors[2][5], 0.0662503168f)); + REQUIRE(componentMatches(vectors[3][5], 0.0430923924f)); + REQUIRE(componentMatches(vectors[4][5], 0.143213451f)); + REQUIRE(componentMatches(vectors[5][5], 0.7962538f)); + + REQUIRE(isOrthogonal(vectors)); + for (uint8_t col = 0; col < 6; col++) { + REQUIRE(eigenResidual(A, values[col][0], eigenColumn(vectors, col)) < + 1e-4f); + } + } +} + +// ============================================================================ +// SVD Tests — Reference values computed via scipy.linalg.svd (Python) +// ============================================================================ + +/** + * @brief Helper: compute Frobenius norm of a matrix. + */ +template +static float frobeniusNorm(const Matrix &M) { + float sum = 0; + for (uint8_t i = 0; i < rows; i++) { + for (uint8_t j = 0; j < columns; j++) { + float v = M.Get(i, j); + sum += v * v; + } + } + return sqrtf(sum); +} + +/** + * @brief Helper: compute reconstruction error ||A - UΣVᵀ||_F. + * + * Verifies the fundamental SVD identity A = U × diag(σ) × Vᵀ. + * For non-square matrices, only the first min(rows,cols) singular values + * contribute to the reconstruction. + */ +template +static float svdReconstructionError(const Matrix &A, + const Matrix &U, + const Matrix &sigma, + const Matrix &Vt) { + // Compute U × diag(σ): only first min(rows,cols) columns of U are used + constexpr uint8_t k = (rows < columns) ? rows : columns; + Matrix USigma{0}; + for (uint8_t i = 0; i < rows; i++) { + for (uint8_t j = 0; j < k; j++) { + USigma[i][j] = U.Get(i, j) * sigma.Get(j, 0); + } + } + + // Compute (UΣ) × Vᵀ: only first k rows of Vt are used + Matrix UVt{0}; + for (uint8_t i = 0; i < rows; i++) { + for (uint8_t j = 0; j < columns; j++) { + float sum = 0; + for (uint8_t p = 0; p < k; p++) { + sum += USigma[i][p] * Vt.Get(p, j); + } + UVt[i][j] = sum; + } + } + + // Compute ||A - UVᵀ||_F + Matrix diff{0}; + A.Sub(UVt, diff); + return frobeniusNorm(diff); +} + +/** + * @brief Helper: check orthogonality of the first k columns of M. + * Verifies M[:,0:k]ᵀ × M[:,0:k] ≈ I_k. + */ +template +static float orthogonalityError(const Matrix &M) { + constexpr uint8_t k = (rows < columns) ? rows : columns; + + // Compute Mᵀ × M (should be I_k in top-left) + Matrix Mt = M.Transpose(); + Matrix MtM{0}; + Mt.Mult(M, MtM); + + float err = 0; + for (uint8_t i = 0; i < k; i++) { + for (uint8_t j = 0; j < k; j++) { + float expected = (i == j) ? 1.0f : 0.0f; + err += (MtM.Get(i, j) - expected) * (MtM.Get(i, j) - expected); + } + } + return sqrtf(err); +} + +/** + * @brief Helper: check that singular values are sorted in descending order. + */ +template +static bool isSortedDescending(const Matrix &sigma, uint8_t count) { + for (uint8_t i = 0; i < count - 1; i++) { + if (sigma.Get(i + 1, 0) > sigma.Get(i, 0) + 1e-6f) { + return false; + } + } + return true; +} + +TEST_CASE("SVD: Simple 2x2 Matrix", "Matrix") { + // Reference: scipy.linalg.svd([[1,2],[3,4]]) + // σ = [5.4649857042, 0.3659661906] + Matrix<2, 2> A{1.0f, 2.0f, 3.0f, 4.0f}; + Matrix<2, 2> U{}, Vt{}; + Matrix<2, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + // Verify singular values (verified with Python scipy.linalg.svd) + REQUIRE_THAT(sigma.Get(0, 0), + Catch::Matchers::WithinRel(5.4649857042f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), + Catch::Matchers::WithinRel(0.3659661906f, 1e-4f)); + + // Verify descending order + REQUIRE(isSortedDescending(sigma, 2)); + + // Verify U is orthogonal: UᵀU ≈ I + REQUIRE_THAT(orthogonalityError(U), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + + // Verify Vt is orthogonal: VtVᵀ ≈ I + REQUIRE_THAT(orthogonalityError(Vt), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + + // Verify reconstruction: A ≈ U Σ Vᵀ + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-4f)); +} + +TEST_CASE("SVD: Symmetric Positive Definite 2x2", "Matrix") { + // Reference: scipy.linalg.svd([[5,3],[3,5]]) + // σ = [8.0, 2.0] (eigenvalues since symmetric PD) + Matrix<2, 2> A{5.0f, 3.0f, 3.0f, 5.0f}; + Matrix<2, 2> U{}, Vt{}; + Matrix<2, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(8.0f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(2.0f, 1e-4f)); + + // For symmetric PD matrices, U ≈ V (up to sign) + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-4f)); +} + +TEST_CASE("SVD: Full-Rank 3x3 Matrix", "Matrix") { + // Reference: scipy.linalg.svd([[1,2,3],[4,5,6],[7,8,10]]) + // σ = [17.4125051668, 0.8751613501, 0.1968665211] + Matrix<3, 3> A{1.0f, 2.0f, 3.0f, 4.0f, 5.0f, 6.0f, 7.0f, 8.0f, 10.0f}; + Matrix<3, 3> U{}, Vt{}; + Matrix<3, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), + Catch::Matchers::WithinRel(17.4125051668f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), + Catch::Matchers::WithinRel(0.8751613501f, 1e-4f)); + REQUIRE_THAT(sigma.Get(2, 0), + Catch::Matchers::WithinRel(0.1968665211f, 1e-4f)); + + REQUIRE(isSortedDescending(sigma, 3)); + REQUIRE_THAT(orthogonalityError(U), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + REQUIRE_THAT(orthogonalityError(Vt), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); +} + +TEST_CASE("SVD: Rank-Deficient 3x3 Matrix", "Matrix") { + // Reference: scipy.linalg.svd([[1,2,3],[4,5,6],[7,8,9]]) + // σ = [16.8481033526, 1.0683695146, ~0] (rank 2) + Matrix<3, 3> A{1.0f, 2.0f, 3.0f, 4.0f, 5.0f, 6.0f, 7.0f, 8.0f, 9.0f}; + Matrix<3, 3> U{}, Vt{}; + Matrix<3, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), + Catch::Matchers::WithinRel(16.8481033526f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), + Catch::Matchers::WithinRel(1.0683695146f, 1e-4f)); + // Third singular value should be ~0 (rank deficiency) + REQUIRE(sigma.Get(2, 0) < 1e-3f); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); +} + +TEST_CASE("SVD: Diagonal 3x3 Matrix", "Matrix") { + // For a diagonal matrix, σ = diagonal entries, U = V = I + // Row-major init: [10,0,0, 0,5,0, 0,0,2] = diag(10,5,2) + Matrix<3, 3> A{10.0f, 0.0f, 0.0f, 0.0f, 5.0f, 0.0f, 0.0f, 0.0f, 2.0f}; + Matrix<3, 3> U{}, Vt{}; + Matrix<3, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(10.0f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(5.0f, 1e-4f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(2.0f, 1e-4f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-4f)); +} + +TEST_CASE("SVD: Tall Matrix (4×3)", "Matrix") { + // Reference: scipy.linalg.svd with full_matrices=False + // σ = [25.4624074360, 1.2906616758, ~0] (rank 2) + Matrix<4, 3> A{1.0f, 2.0f, 3.0f, 4.0f, 5.0f, 6.0f, + 7.0f, 8.0f, 9.0f, 10.0f, 11.0f, 12.0f}; + Matrix<4, 3> U{}; + Matrix<3, 3> Vt{}; // Vt is always n×n + Matrix<3, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), + Catch::Matchers::WithinRel(25.4624074360f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), + Catch::Matchers::WithinRel(1.2906616758f, 1e-4f)); + REQUIRE(sigma.Get(2, 0) < 1e-3f); + + // U should be 4×3 with orthonormal columns + REQUIRE_THAT(orthogonalityError(U), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); +} + +TEST_CASE("SVD: Wide Matrix (3×5)", "Matrix") { + // Reference: scipy.linalg.svd with full_matrices=False + // σ = [35.1272233336, 2.4653966969, ~0] (rank 2) + Matrix<3, 5> A{1.0f, 2.0f, 3.0f, 4.0f, 5.0f, 6.0f, 7.0f, 8.0f, + 9.0f, 10.0f, 11.0f, 12.0f, 13.0f, 14.0f, 15.0f}; + Matrix<3, 5> U{}; + Matrix<5, 5> Vt{}; // Vt is always n×n + Matrix<5, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), + Catch::Matchers::WithinRel(35.1272233336f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), + Catch::Matchers::WithinRel(2.4653966969f, 1e-4f)); + REQUIRE(sigma.Get(2, 0) < 1e-3f); + + // Vt should be 5×5 with orthonormal rows (first k) + REQUIRE_THAT(orthogonalityError(Vt), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); +} + +TEST_CASE("SVD: 5×5 Symmetric Tridiagonal", "Matrix") { + // Reference: scipy.linalg.svd for discrete Laplacian-like matrix + // σ = [3.7320508076, 3.0, 2.0, 1.0, 0.2679491924] + Matrix<5, 5> A{2.0f, -1.0f, 0.0f, 0.0f, 0.0f, -1.0f, 2.0f, -1.0f, 0.0f, + 0.0f, 0.0f, -1.0f, 2.0f, -1.0f, 0.0f, 0.0f, 0.0f, -1.0f, + 2.0f, -1.0f, 0.0f, 0.0f, 0.0f, -1.0f, 2.0f}; + Matrix<5, 5> U{}, Vt{}; + Matrix<5, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), + Catch::Matchers::WithinRel(3.7320508076f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(3.0f, 1e-4f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(2.0f, 1e-4f)); + REQUIRE_THAT(sigma.Get(3, 0), Catch::Matchers::WithinRel(1.0f, 1e-4f)); + REQUIRE_THAT(sigma.Get(4, 0), + Catch::Matchers::WithinRel(0.2679491924f, 1e-4f)); + + REQUIRE(isSortedDescending(sigma, 5)); + REQUIRE_THAT(orthogonalityError(U), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + REQUIRE_THAT(orthogonalityError(Vt), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); +} + +TEST_CASE("SVD: 5×5 Full-Rank Random", "Matrix") { + // Reference: scipy.linalg.svd, np.random.default_rng(7).standard_normal((5,5)) + // cond ≈ 11.5 + // σ = [3.04651784, 2.22681732, 1.84290662, 1.02101969, 0.264826749] + Matrix<5, 5> A{0.00123015f, 0.298746f, -0.274138f, -0.890592f, -0.454671f, + -0.991647f, 0.0601436f, 1.34022f, -0.492207f, -0.620475f, + 0.489842f, 0.356887f, 0.105414f, -0.930468f, -0.0292518f, + 0.695303f, -1.34421f, -0.457616f, -1.90122f, -1.28954f, + -1.84174f, -0.235091f, -1.26745f, 0.271264f, 0.156751f}; + Matrix<5, 5> U{}, Vt{}; + Matrix<5, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(3.04651784f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(2.22681732f, 1e-4f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(1.84290662f, 1e-4f)); + REQUIRE_THAT(sigma.Get(3, 0), + Catch::Matchers::WithinRel(1.02101969f, 1e-4f)); + REQUIRE_THAT(sigma.Get(4, 0), Catch::Matchers::WithinRel(0.264826749f, 1e-4f)); + + REQUIRE(isSortedDescending(sigma, 5)); + REQUIRE_THAT(orthogonalityError(U), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + REQUIRE_THAT(orthogonalityError(Vt), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); +} + +TEST_CASE("SVD: 5×5 Rank-Deficient (rank 3)", "Matrix") { + // Reference: scipy.linalg.svd of rng.standard_normal((5,3)) @ + // rng.standard_normal((3,5)) — exactly rank 3 + // σ = [7.29829771, 2.76487864, 1.57392325, ~1e-16, ~1e-16] + Matrix<5, 5> A{3.46252f, -1.52873f, -0.111526f, 1.28954f, -5.18688f, + -1.34702f, 1.92936f, -0.0410797f, -0.958791f, 0.449623f, + 0.844592f, 0.0986352f, 0.408213f, 0.124867f, -2.45393f, + 1.34177f, -0.587312f, -1.39847f, 0.580032f, -0.167692f, + 0.915462f, -0.311165f, -1.16141f, 0.377283f, 0.055373f}; + Matrix<5, 5> U{}, Vt{}; + Matrix<5, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(7.29829771f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(2.76487864f, 1e-4f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(1.57392325f, 1e-4f)); + // The two rank-deficient singular values must be at noise level + REQUIRE(sigma.Get(3, 0) < 1e-3f); + REQUIRE(sigma.Get(4, 0) < 1e-3f); + + REQUIRE(isSortedDescending(sigma, 5)); + REQUIRE_THAT(orthogonalityError(U), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + REQUIRE_THAT(orthogonalityError(Vt), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + + // Rank-3 matrix: the top-3 SVD terms must reproduce A + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 5e-3f)); +} + +TEST_CASE("SVD: 5×5 Wide Dynamic Range (cond ≈ 9000)", "Matrix") { + // Symmetric banded, diagonal decays 50 → 1e-3, off-diagonals 3 → 0.01 + // Reference: scipy.linalg.svd + // σ = [50.1496395, 10.1719501, 1.02846061, 0.0207253626, 0.00557535757] + Matrix<5, 5> A{50.0f, 3.0f, 0.0f, 0.0f, 0.0f, + -3.0f, 10.0f, 0.5f, 0.0f, 0.0f, + 0.0f, -0.5f, 1.0f, 0.08f, 0.0f, + 0.0f, 0.0f, -0.08f, 0.01f, 0.01f, + 0.0f, 0.0f, 0.0f, -0.01f, 0.001f}; + Matrix<5, 5> U{}, Vt{}; + Matrix<5, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), + Catch::Matchers::WithinRel(50.1496395f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), + Catch::Matchers::WithinRel(10.1719501f, 1e-4f)); + REQUIRE_THAT(sigma.Get(2, 0), + Catch::Matchers::WithinRel(1.02846061f, 1e-4f)); + REQUIRE_THAT(sigma.Get(3, 0), + Catch::Matchers::WithinRel(0.0207253626f, 1e-3f)); + REQUIRE_THAT(sigma.Get(4, 0), + Catch::Matchers::WithinRel(0.00557535757f, 1e-3f)); + + REQUIRE(isSortedDescending(sigma, 5)); + REQUIRE_THAT(orthogonalityError(U), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + REQUIRE_THAT(orthogonalityError(Vt), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + // Relative to the largest entry (‖A‖F ≈ 50.1) + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 5e-2f)); +} + +TEST_CASE("SVD: 5×5 Symmetric Indefinite", "Matrix") { + // Symmetric with negative eigenvalues — σ must equal |eigenvalues| + // Reference: scipy.linalg.svd + // σ = [4.70141723, 3.76392521, 2.73426097, 1.15773083, 0.642665756] + Matrix<5, 5> A{2.0f, -1.0f, 0.0f, 0.0f, 0.5f, + -1.0f, 2.0f, -1.0f, 0.0f, 0.0f, + 0.0f, -1.0f, 3.0f, -1.0f, 0.0f, + 0.0f, 0.0f, -1.0f, 2.0f, -1.0f, + 0.5f, 0.0f, 0.0f, -1.0f, 4.0f}; + Matrix<5, 5> U{}, Vt{}; + Matrix<5, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), + Catch::Matchers::WithinRel(4.70141723f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), + Catch::Matchers::WithinRel(3.76392521f, 1e-4f)); + REQUIRE_THAT(sigma.Get(2, 0), + Catch::Matchers::WithinRel(2.73426097f, 1e-4f)); + REQUIRE_THAT(sigma.Get(3, 0), + Catch::Matchers::WithinRel(1.15773083f, 1e-4f)); + REQUIRE_THAT(sigma.Get(4, 0), + Catch::Matchers::WithinRel(0.642665756f, 1e-4f)); + + REQUIRE(isSortedDescending(sigma, 5)); + REQUIRE_THAT(orthogonalityError(U), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + REQUIRE_THAT(orthogonalityError(Vt), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); +} + +TEST_CASE("SVD: Non-Square with Negative Values (2×3)", "Matrix") { + // Reference: scipy.linalg.svd([[0.5,-0.3,0.8],[-0.2,0.7,0.1]]) + // σ = [1.0384009867, 0.6646227432] + Matrix<2, 3> A{0.5f, -0.3f, 0.8f, -0.2f, 0.7f, 0.1f}; + Matrix<2, 3> U{}; + Matrix<3, 3> Vt{}; // Vt is always n×n + Matrix<3, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), + Catch::Matchers::WithinRel(1.0384009867f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), + Catch::Matchers::WithinRel(0.6646227432f, 1e-4f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-4f)); +} + +TEST_CASE("SVD: Near-Singular 2×2 Matrix", "Matrix") { + // Condition number ≈ 1e6 — tests numerical stability + // Reference: scipy.linalg.svd([[1,0],[0,1e-6]]) + // σ = [1.0, 1e-6] + Matrix<2, 2> A{1.0f, 0.0f, 0.0f, 1e-6f}; + Matrix<2, 2> U{}, Vt{}; + Matrix<2, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(1.0f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(1e-6f, 1e-2f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); +} + +TEST_CASE("SVD: Orthogonal Matrix (3×3)", "Matrix") { + // For an orthogonal matrix, all singular values should be 1. + // Rotation matrix about z-axis by 45° + float c = sqrtf(0.5f); // cos(45°) + float s = sqrtf(0.5f); // sin(45°) + Matrix<3, 3> A{c, -s, 0.0f, s, c, 0.0f, 0.0f, 0.0f, 1.0f}; + Matrix<3, 3> U{}, Vt{}; + Matrix<3, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + // All singular values should be 1 for an orthogonal matrix + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(1.0f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(1.0f, 1e-4f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(1.0f, 1e-4f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-4f)); +} + +TEST_CASE("SVD: Identity Matrix", "Matrix") { + // For I, σ = [1, 1, 1], U = V = I + Matrix<3, 3> A{1.0f, 0.0f, 0.0f, 0.0f, 1.0f, 0.0f, 0.0f, 0.0f, 1.0f}; + Matrix<3, 3> U{}, Vt{}; + Matrix<3, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(1.0f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(1.0f, 1e-4f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(1.0f, 1e-4f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); +} + +TEST_CASE("SVD: Zero Matrix", "Matrix") { + // All singular values should be zero + Matrix<3, 3> A{0.0f}; + Matrix<3, 3> U{}, Vt{}; + Matrix<3, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE(sigma.Get(0, 0) < 1e-6f); + REQUIRE(sigma.Get(1, 0) < 1e-6f); + REQUIRE(sigma.Get(2, 0) < 1e-6f); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); +} + +TEST_CASE("SVD: 2×1 Column Vector", "Matrix") { + // For a column vector v, σ = ||v||, U = v/||v|| (with padding) + Matrix<2, 1> A{3.0f, 4.0f}; + Matrix<2, 1> U{}; + Matrix<1, 1> Vt{}; // Vt is always n×n + Matrix<1, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + // σ should be the Euclidean norm: ||[3,4]|| = 5 + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(5.0f, 1e-4f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-4f)); +} + +TEST_CASE("SVD: 1×2 Row Vector", "Matrix") { + // For a row vector vᵀ, σ = ||v||, Vt = v/||v|| (with padding) + Matrix<1, 2> A{3.0f, 4.0f}; + Matrix<1, 2> U{}; + Matrix<2, 2> Vt{}; // Vt is always n×n + Matrix<2, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + // σ should be the Euclidean norm: ||[3,4]|| = 5 + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(5.0f, 1e-4f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-4f)); +} +// ============================================================================ +// SVD Tests — Large-Size Instantiations (N > 5) +// +// The SVD is templated on N = max(rows, cols) with stack-only buffers, so +// these cases exercise instantiations beyond the old 5×5 hard limit: +// 7×5 (N=7, tall), 6×6 (N=6, square), 5×8 (N=8, wide/transpose path), +// 6×4 (N=6, tall, near rank-deficiency → deflation path). +// Reference singular values: scipy.linalg.svd. +// ============================================================================ + +TEST_CASE("SVD: Tall 7×5 Matrix (N=7)", "Matrix") { + // Reference: scipy.linalg.svd + // σ = [7.9180769443, 4.6593687008, 4.2921645616, 2.6009010840, 1.9842770351] + Matrix<7, 5> A{-0.7528f, 2.7043f, 1.392f, 0.592f, -2.0639f, + -2.064f, -2.6515f, 2.1971f, 0.6067f, 1.2484f, + -2.8765f, 2.8195f, 1.9947f, -1.726f, -1.9091f, + -1.8996f, -1.1745f, 0.1485f, -0.4083f, -1.2526f, + 0.6711f, -2.163f, -1.2471f, -0.8018f, -0.2636f, + 1.7111f, -1.802f, 0.0854f, 0.5545f, -2.7213f, + 0.6453f, -1.9769f, -2.6097f, 2.6933f, 2.7938f}; + Matrix<7, 5> U{}; + Matrix<5, 5> Vt{}; + Matrix<5, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(7.9180769443f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(4.6593687008f, 1e-4f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(4.2921645616f, 1e-4f)); + REQUIRE_THAT(sigma.Get(3, 0), Catch::Matchers::WithinRel(2.6009010840f, 1e-4f)); + REQUIRE_THAT(sigma.Get(4, 0), Catch::Matchers::WithinRel(1.9842770351f, 1e-4f)); + + REQUIRE(isSortedDescending(sigma, 5)); + REQUIRE_THAT(orthogonalityError(U), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + REQUIRE_THAT(orthogonalityError(Vt), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); +} + +TEST_CASE("SVD: Square 6×6 Matrix (N=6)", "Matrix") { + // Reference: scipy.linalg.svd (float32 inputs) + // σ = [5.018912792, 4.244967461, 2.505512476, + // 1.838801861, 0.9111995101, 0.4580149353] + Matrix<6, 6> A{1.2336f, -0.7815f, -1.6093f, 0.7369f, -0.2394f, -1.5118f, + -0.0193f, -1.8624f, 1.6373f, -0.9649f, 0.6501f, -0.7532f, + 0.0803f, 0.1868f, -1.2606f, 1.8783f, + 1.1005f, 1.758f, 1.5793f, 0.3916f, 1.6875f, -1.646f, + -1.2161f, -1.8191f, -0.6987f, -0.4453f, -0.9146f, 1.315f, + -0.573f, -0.8763f, 0.1708f, -1.4363f, 1.2088f, -1.7018f, + 1.089f, 1.9475f}; + Matrix<6, 6> U{}, Vt{}; + Matrix<6, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(5.018912792f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(4.244967461f, 1e-4f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(2.505512476f, 1e-4f)); + REQUIRE_THAT(sigma.Get(3, 0), Catch::Matchers::WithinRel(1.838801861f, 1e-4f)); + REQUIRE_THAT(sigma.Get(4, 0), Catch::Matchers::WithinRel(0.9111995101f, 1e-4f)); + REQUIRE_THAT(sigma.Get(5, 0), Catch::Matchers::WithinRel(0.4580149353f, 1e-4f)); + + REQUIRE(isSortedDescending(sigma, 6)); + REQUIRE_THAT(orthogonalityError(U), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + REQUIRE_THAT(orthogonalityError(Vt), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); +} + +TEST_CASE("SVD: Wide 5×8 Matrix (N=8, transpose path)", "Matrix") { + // Reference: scipy.linalg.svd + // σ = [5.8027782929, 4.1105282764, 3.7755966048, 3.3208483982, 2.0321410547] + // + // Wide matrices take the Aᵀ transpose path; Vt must be the FULL 8×8 + // orthogonal matrix (all 8 rows meaningful), not just the top 5. + Matrix<5, 8> A{-1.5064f, -2.4724f, 1.5773f, 1.0343f, 1.145f, 1.3564f, -2.1298f, -0.7077f, + -1.9207f, 1.8155f, 0.6165f, -0.8455f, -2.1822f, -0.9451f, -0.8741f, 1.148f, + 0.6878f, 1.9361f, -0.1389f, -1.902f, 1.0662f, 1.3039f, 0.3064f, 1.3548f, + -0.031f, 0.1137f, -0.3623f, -2.3729f, -1.9605f, -2.3429f, 0.6821f, -0.9282f, + 0.0429f, 2.0378f, -1.2535f, -0.4481f, 1.2778f, -1.356f, -2.1151f, -1.0512f}; + Matrix<5, 8> U{}; + Matrix<8, 8> Vt{}; + Matrix<8, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(5.8027782929f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(4.1105282764f, 1e-4f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(3.7755966048f, 1e-4f)); + REQUIRE_THAT(sigma.Get(3, 0), Catch::Matchers::WithinRel(3.3208483982f, 1e-4f)); + REQUIRE_THAT(sigma.Get(4, 0), Catch::Matchers::WithinRel(2.0321410547f, 1e-4f)); + // Remaining singular values must be at noise level + REQUIRE(sigma.Get(5, 0) < 1e-3f); + REQUIRE(sigma.Get(6, 0) < 1e-3f); + REQUIRE(sigma.Get(7, 0) < 1e-3f); + + REQUIRE(isSortedDescending(sigma, 8)); + REQUIRE_THAT(orthogonalityError(U), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + REQUIRE_THAT(orthogonalityError(Vt), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); +} + +TEST_CASE("SVD: Tall 6×4 Near Rank-Deficient (N=6, deflation path)", "Matrix") { + // Reference: scipy.linalg.svd + // σ = [5.9434060901, 3.2857910666, 0.3066158795, 6.48e-07] + // + // σ₄ ≈ 6.5e-7 forces the deflation logic to zero the last + // superdiagonal and isolate the trailing 1×1 block. + Matrix<6, 4> A{-0.086904f, 1.410225f, 1.308323f, 2.234762f, + 0.022123f, 0.896751f, 0.324176f, 0.773607f, + -0.473015f, 1.555111f, 0.290059f, 1.157726f, + -0.78371f, 1.398884f, -1.930606f, -1.548717f, + 0.201518f, -0.626835f, 0.976596f, 0.875294f, + -1.24206f, 1.60595f, -3.078089f, -2.73695f}; + Matrix<6, 4> U{}; + Matrix<4, 4> Vt{}; + Matrix<4, 1> sigma{}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(5.9434060901f, 1e-4f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(3.2857910666f, 1e-4f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(0.3066158795f, 1e-4f)); + // Fourth singular value is at noise level (matrix is ~rank 3) + REQUIRE(sigma.Get(3, 0) < 1e-4f); + + REQUIRE(isSortedDescending(sigma, 4)); + REQUIRE_THAT(orthogonalityError(U), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + REQUIRE_THAT(orthogonalityError(Vt), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + + float reconErr = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(reconErr, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); +} diff --git a/unit-tests/matrix-timing-tests.cpp b/unit-tests/matrix-timing-tests.cpp index 0722e0c..e736993 100644 --- a/unit-tests/matrix-timing-tests.cpp +++ b/unit-tests/matrix-timing-tests.cpp @@ -76,7 +76,8 @@ TEST_CASE("Timing Tests", "Matrix") { SECTION("Determinant") { for (uint32_t i{0}; i < 1000000; i++) { - float det1 = mat4.Det(); + float det = mat4.Det(); + (void)det; } } diff --git a/unit-tests/qr-build-blocks-tests.cpp b/unit-tests/qr-build-blocks-tests.cpp new file mode 100644 index 0000000..8b9d760 --- /dev/null +++ b/unit-tests/qr-build-blocks-tests.cpp @@ -0,0 +1,581 @@ +// include the unit test framework first +#include +#include + +// include the module you're going to test next +#include "Matrix.hpp" +#include "QR.hpp" + +// any other libraries +#include +#include +#include + +// ============================================================================ +// Helpers +// ============================================================================ + +/** + * @brief Frobenius norm of an N x N matrix. + */ +template +static float frob(const Matrix &M) { + float sum = 0.0f; + for (uint8_t i = 0; i < N; i++) + for (uint8_t j = 0; j < N; j++) { + float v = M.Get(i, j); + sum += v * v; + } + return sqrtf(sum); +} + +/** + * @brief Check M is orthogonal (M^T M ~ I). + */ +template +static bool isOrthogonal(const Matrix &M, float tol = 1e-5f) { + Matrix Mt = M.Transpose(); + Matrix MtM{}; + Mt.Mult(M, MtM); + for (uint8_t i = 0; i < N; i++) + for (uint8_t j = 0; j < N; j++) { + float expected = (i == j) ? 1.0f : 0.0f; + if (fabsf(MtM.Get(i, j) - expected) > tol) + return false; + } + return true; +} + +/** + * @brief 3x3 trace. + */ +static float trace3(const Matrix<3, 3> &A) { + return A.Get(0, 0) + A.Get(1, 1) + A.Get(2, 2); +} + +/** + * @brief 3x3 sum of principal 2x2 minors (2nd elementary invariant). + */ +static float e2_3x3(const Matrix<3, 3> &A) { + return A.Get(0, 0) * A.Get(1, 1) - A.Get(0, 1) * A.Get(0, 1) + + A.Get(0, 0) * A.Get(2, 2) - A.Get(0, 2) * A.Get(0, 2) + + A.Get(1, 1) * A.Get(2, 2) - A.Get(1, 2) * A.Get(1, 2); +} + +/** + * @brief 3x3 determinant. + */ +static float det3(const Matrix<3, 3> &A) { + return A.Get(0, 0) * + (A.Get(1, 1) * A.Get(2, 2) - A.Get(1, 2) * A.Get(2, 1)) - + A.Get(0, 1) * + (A.Get(1, 0) * A.Get(2, 2) - A.Get(1, 2) * A.Get(2, 0)) + + A.Get(0, 2) * + (A.Get(1, 0) * A.Get(2, 1) - A.Get(1, 1) * A.Get(2, 0)); +} + +/** + * @brief Sign-invariant comparison of |actual| against refAbs. + */ +static bool matchesAbs(float actual, float refAbs, float relTol = 1e-5f, + float absTol = 1e-6f) { + float a = fabsf(actual); + if (refAbs < 1e-3f) + return a < absTol + relTol; + return fabsf(a - refAbs) <= relTol * refAbs; +} + +// ============================================================================ +// TEST 1: GivensRotation +// ============================================================================ +TEST_CASE("QR Building Block: GivensRotation", "[Matrix][QR]") { + // R = [[c, s], [-s, c]] must satisfy R * (a, b)^T = (r, 0)^T. + + { + // Reference: hypot(2, 1) = sqrt(5) = 2.236067977 + float c = 0, s = 0; + QR::GivensRotation(2.0f, 1.0f, c, s); + REQUIRE_THAT(c, Catch::Matchers::WithinRel(0.894427191f, 1e-6f)); + REQUIRE_THAT(s, Catch::Matchers::WithinRel(0.447213595f, 1e-6f)); + REQUIRE_THAT(c * 2.0f + s * 1.0f, + Catch::Matchers::WithinRel(2.236067977f, 1e-6f)); + REQUIRE_THAT(-s * 2.0f + c * 1.0f, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } + + { + // Reference: hypot(3, 4) = 5 exactly + float c = 0, s = 0; + QR::GivensRotation(3.0f, 4.0f, c, s); + REQUIRE_THAT(c, Catch::Matchers::WithinRel(0.6f, 1e-6f)); + REQUIRE_THAT(s, Catch::Matchers::WithinRel(0.8f, 1e-6f)); + REQUIRE_THAT(c * 3.0f + s * 4.0f, Catch::Matchers::WithinRel(5.0f, 1e-6f)); + REQUIRE_THAT(-s * 3.0f + c * 4.0f, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } + + { + // Pure second component: c = 0, s = 1 + float c = 1, s = 1; + QR::GivensRotation(0.0f, 5.0f, c, s); + REQUIRE_THAT(c, Catch::Matchers::WithinAbs(0.0f, 1e-7f)); + REQUIRE_THAT(s, Catch::Matchers::WithinRel(1.0f, 1e-6f)); + } + + { + // Zero vector: identity rotation + float c = 0, s = 0; + QR::GivensRotation(0.0f, 0.0f, c, s); + REQUIRE_THAT(c, Catch::Matchers::WithinRel(1.0f, 1e-7f)); + REQUIRE_THAT(s, Catch::Matchers::WithinAbs(0.0f, 1e-7f)); + } + + { + // Negative first component preserves the sign of c + float c = 0, s = 0; + QR::GivensRotation(-2.0f, 1.0f, c, s); + REQUIRE_THAT(c, Catch::Matchers::WithinRel(-0.894427191f, 1e-6f)); + REQUIRE_THAT(s, Catch::Matchers::WithinRel(0.447213595f, 1e-6f)); + REQUIRE_THAT(-s * -2.0f + c * 1.0f, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } +} + +// ============================================================================ +// TEST 2: ApplyRotationBothSides (similarity A <- G A G^T) +// ============================================================================ +TEST_CASE("QR Building Block: ApplyRotationBothSides", "[Matrix][QR]") { + // Reference (numpy, float64): A = [[2,1,0],[1,3,1],[0,1,4]], i = 0, + // Givens(2,1) -> G A G^T = + // [[ 3.0, 1.0, 0.447213595], + // [ 1.0, 2.0, 0.894427191], + // [ 0.447213595, 0.894427191, 4.0]] + // (Note: G A G^T with G zeroing (2,1) sends the A[0][1] coupling into the + // (0,2) corner, NOT into the subdiagonal -- the subdiagonal-zeroing happens + // in the QR chase context where the bulge column has the right shape.) + { + Matrix<3, 3> A{2, 1, 0, 1, 3, 1, 0, 1, 4}; + float c = 0.894427191f, s = 0.447213595f; + + QR::ApplyRotationBothSides(A, 0, c, s); + + REQUIRE_THAT(A.Get(0, 0), Catch::Matchers::WithinRel(3.0f, 1e-5f)); + REQUIRE_THAT(A.Get(0, 1), Catch::Matchers::WithinRel(1.0f, 1e-5f)); + REQUIRE_THAT(A.Get(0, 2), + Catch::Matchers::WithinRel(0.447213595f, 1e-5f)); + REQUIRE_THAT(A.Get(1, 1), Catch::Matchers::WithinRel(2.0f, 1e-5f)); + REQUIRE_THAT(A.Get(1, 2), + Catch::Matchers::WithinRel(0.894427191f, 1e-5f)); + REQUIRE_THAT(A.Get(2, 2), Catch::Matchers::WithinRel(4.0f, 1e-5f)); + + // Symmetry must be preserved exactly in both triangles + for (uint8_t i = 0; i < 3; i++) + for (uint8_t j = 0; j < 3; j++) + REQUIRE(A.Get(i, j) == A.Get(j, i)); + } + + // Same check at i = 1. + // Reference (numpy, float64): B = [[5,0,1],[0,6,2],[1,2,7]], i = 1, + // Givens(6,2) -> G B G^T = + // [[ 5.0, 0.316227766, 0.948683298], + // [ 0.316227766, 7.3, 1.9], + // [ 0.948683298, 1.9, 5.7]] + { + Matrix<3, 3> B{5, 0, 1, 0, 6, 2, 1, 2, 7}; + float c = 0.948683298f, s = 0.316227766f; + + QR::ApplyRotationBothSides(B, 1, c, s); + + REQUIRE_THAT(B.Get(0, 0), Catch::Matchers::WithinRel(5.0f, 1e-5f)); + REQUIRE_THAT(B.Get(0, 1), + Catch::Matchers::WithinRel(0.316227766f, 1e-5f)); + REQUIRE_THAT(B.Get(0, 2), + Catch::Matchers::WithinRel(0.948683298f, 1e-5f)); + REQUIRE_THAT(B.Get(1, 1), Catch::Matchers::WithinRel(7.3f, 1e-5f)); + REQUIRE_THAT(B.Get(1, 2), Catch::Matchers::WithinRel(1.9f, 1e-5f)); + REQUIRE_THAT(B.Get(2, 2), Catch::Matchers::WithinRel(5.7f, 1e-5f)); + + for (uint8_t i = 0; i < 3; i++) + for (uint8_t j = 0; j < 3; j++) + REQUIRE(B.Get(i, j) == B.Get(j, i)); + } + + // Identity rotation leaves the matrix unchanged + { + Matrix<3, 3> C{1, 2, 3, 2, 4, 5, 3, 5, 6}; + QR::ApplyRotationBothSides(C, 1, 1.0f, 0.0f); + REQUIRE(C.Get(0, 0) == 1.0f); + REQUIRE(C.Get(0, 1) == 2.0f); + REQUIRE(C.Get(0, 2) == 3.0f); + REQUIRE(C.Get(1, 1) == 4.0f); + REQUIRE(C.Get(1, 2) == 5.0f); + REQUIRE(C.Get(2, 2) == 6.0f); + } + + // Spectrum invariants (trace, Frobenius norm) are preserved. (c, s) + // must be a unit vector for G A G^T to be a similarity transform. + { + Matrix<3, 3> D{1, 2, 3, 2, 5, 8, 3, 8, 9}; + float tr = trace3(D); + float fn = frob(D); + float c = 0.6f, s = 0.8f; + QR::ApplyRotationBothSides(D, 0, c, s); + REQUIRE_THAT(trace3(D), Catch::Matchers::WithinRel(tr, 1e-5f)); + REQUIRE_THAT(frob(D), Catch::Matchers::WithinRel(fn, 1e-5f)); + } +} +// ============================================================================ +// TEST 3: ApplyRotationToVectors (V <- V G^T) +// ============================================================================ +TEST_CASE("QR Building Block: ApplyRotationToVectors", "[Matrix][QR]") { + // V = I, i = 0, Givens(2,1): V <- I * G^T with G^T = [[c, -s], [s, c]] = + // [[ c, -s, 0], + // [ s, c, 0], + // [ 0, 0, 1]] + { + Matrix<3, 3> V{0}; + V[0][0] = 1; + V[1][1] = 1; + V[2][2] = 1; + float c = 0.894427191f, s = 0.447213595f; + + QR::ApplyRotationToVectors(V, 0, c, s); + + REQUIRE_THAT(V.Get(0, 0), Catch::Matchers::WithinRel(0.894427191f, 1e-6f)); + REQUIRE_THAT(V.Get(0, 1), Catch::Matchers::WithinRel(-0.447213595f, 1e-6f)); + REQUIRE_THAT(V.Get(0, 2), Catch::Matchers::WithinAbs(0.0f, 1e-7f)); + REQUIRE_THAT(V.Get(1, 0), Catch::Matchers::WithinRel(0.447213595f, 1e-6f)); + REQUIRE_THAT(V.Get(1, 1), Catch::Matchers::WithinRel(0.894427191f, 1e-6f)); + REQUIRE_THAT(V.Get(1, 2), Catch::Matchers::WithinAbs(0.0f, 1e-7f)); + REQUIRE_THAT(V.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-7f)); + REQUIRE_THAT(V.Get(2, 1), Catch::Matchers::WithinAbs(0.0f, 1e-7f)); + REQUIRE_THAT(V.Get(2, 2), Catch::Matchers::WithinRel(1.0f, 1e-7f)); + + // Product of rotations must stay orthogonal + REQUIRE(isOrthogonal(V)); + } + + // Two successive rotations accumulate (V <- V G1^T G2^T) + // Reference (numpy, float64): + // [[ 0.894427191, -0.424264069, 0.141421356], + // [ 0.447213595, 0.848528137, -0.282842712], + // [ 0.0, 0.316227766, 0.948683298]] + { + Matrix<3, 3> V{0}; + V[0][0] = 1; + V[1][1] = 1; + V[2][2] = 1; + QR::ApplyRotationToVectors(V, 0, 0.894427191f, 0.447213595f); + QR::ApplyRotationToVectors(V, 1, 0.948683298f, 0.316227766f); + REQUIRE(isOrthogonal(V)); + // Column 0 was only touched by the first rotation + REQUIRE_THAT(V.Get(0, 0), Catch::Matchers::WithinRel(0.894427191f, 1e-5f)); + REQUIRE_THAT(V.Get(1, 0), Catch::Matchers::WithinRel(0.447213595f, 1e-5f)); + REQUIRE_THAT(V.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-7f)); + REQUIRE_THAT(V.Get(0, 1), Catch::Matchers::WithinRel(-0.424264069f, 1e-5f)); + REQUIRE_THAT(V.Get(0, 2), Catch::Matchers::WithinRel(0.141421356f, 1e-5f)); + REQUIRE_THAT(V.Get(1, 2), Catch::Matchers::WithinRel(-0.282842712f, 1e-5f)); + REQUIRE_THAT(V.Get(2, 1), Catch::Matchers::WithinRel(0.316227766f, 1e-5f)); + REQUIRE_THAT(V.Get(2, 2), Catch::Matchers::WithinRel(0.948683298f, 1e-5f)); + } +} + +// ============================================================================ +// TEST 4: WilkinsonShift +// ============================================================================ +TEST_CASE("QR Building Block: WilkinsonShift", "[Matrix][QR]") { + // mu = (a+d)/2 - sign(a-d) * sqrt(((a-d)/2)^2 + b^2) + // Reference: eigenvalues of [[2,1],[1,4]] are 1.5858, 4.4142; closest + // to d = 4 is 4.414213562. + REQUIRE_THAT(QR::WilkinsonShift(2.0f, 1.0f, 4.0f), + Catch::Matchers::WithinRel(4.414213562f, 1e-6f)); + + // [[5,2],[2,1]]: eigenvalues 0.1716, 5.8284; closest to d = 1 is 0.171572875 + REQUIRE_THAT(QR::WilkinsonShift(5.0f, 2.0f, 1.0f), + Catch::Matchers::WithinRel(0.171572875f, 1e-5f)); + + // Zero off-diagonal: returns d itself (sign(0) = +1 picks d, not a) + REQUIRE_THAT(QR::WilkinsonShift(3.0f, 0.0f, 7.0f), + Catch::Matchers::WithinRel(7.0f, 1e-7f)); + REQUIRE_THAT(QR::WilkinsonShift(7.0f, 0.0f, 3.0f), + Catch::Matchers::WithinRel(3.0f, 1e-7f)); + + // a == d: shift is the larger-magnitude off-diagonal combination + // [[1,3],[3,1]]: eigenvalues -2, 4; closest to d = 1 is -2 + REQUIRE_THAT(QR::WilkinsonShift(1.0f, 3.0f, 1.0f), + Catch::Matchers::WithinRel(-2.0f, 1e-6f)); +} + +// ============================================================================ +// TEST 5: Solve2x2Eigen +// ============================================================================ +TEST_CASE("QR Building Block: Solve2x2Eigen", "[Matrix][QR]") { + // Symmetric block [[2,1],[1,3]]: + // eigenvalues 1.381966011, 3.618033989; + // eigenvector of 3.618033989 is +/- (0.525731112, 0.850650808) + { + Matrix<2, 2> A{2, 1, 1, 3}; + float lHi = 0, lLo = 0, c = 0, s = 0; + QR::Solve2x2Eigen(A, 0, lHi, lLo, c, s); + + REQUIRE_THAT(lHi, Catch::Matchers::WithinRel(3.618033989f, 1e-6f)); + REQUIRE_THAT(lLo, Catch::Matchers::WithinRel(1.381966011f, 1e-6f)); + REQUIRE(matchesAbs(c, 0.525731112f)); + REQUIRE(matchesAbs(s, 0.850650808f)); + + // Residual: A * vHi = lHi * vHi with vHi = (c, s) + REQUIRE_THAT(c * 2.0f + s * 1.0f, + Catch::Matchers::WithinRel(lHi * c, 1e-5f)); + REQUIRE_THAT(c * 1.0f + s * 3.0f, + Catch::Matchers::WithinRel(lHi * s, 1e-5f)); + // Second eigenvector vLo = (-s, c) + REQUIRE_THAT(-s * 2.0f + c * 1.0f, + Catch::Matchers::WithinRel(lLo * -s, 1e-5f)); + REQUIRE_THAT(-s * 1.0f + c * 3.0f, + Catch::Matchers::WithinRel(lLo * c, 1e-5f)); + } + + // Nonsymmetric block [[1,2],[3,4]] (used by the N == 2 entry point): + // eigenvalues 5.372281323, -0.372281323; + // eigenvector of 5.372281323 is +/- (0.415973558, 0.909376709) + { + Matrix<2, 2> A{1, 2, 3, 4}; + float lHi = 0, lLo = 0, c = 0, s = 0; + QR::Solve2x2Eigen(A, 0, lHi, lLo, c, s); + + REQUIRE_THAT(lHi, Catch::Matchers::WithinRel(5.372281323f, 1e-6f)); + REQUIRE_THAT(lLo, Catch::Matchers::WithinRel(-0.372281323f, 1e-6f)); + REQUIRE(matchesAbs(c, 0.415973558f)); + REQUIRE(matchesAbs(s, 0.909376709f)); + + // Both-row residual with vHi = (c, s): A v = l v + REQUIRE_THAT(c * 1.0f + s * 2.0f, + Catch::Matchers::WithinRel(lHi * c, 1e-5f)); + REQUIRE_THAT(c * 3.0f + s * 4.0f, + Catch::Matchers::WithinRel(lHi * s, 1e-5f)); + } + + // Diagonal blocks: eigenvectors are coordinate vectors + { + Matrix<2, 2> A{5, 0, 0, 2}; + float lHi = 0, lLo = 0, c = 0, s = 0; + QR::Solve2x2Eigen(A, 0, lHi, lLo, c, s); + REQUIRE_THAT(lHi, Catch::Matchers::WithinRel(5.0f, 1e-7f)); + REQUIRE_THAT(lLo, Catch::Matchers::WithinRel(2.0f, 1e-7f)); + REQUIRE_THAT(c, Catch::Matchers::WithinRel(1.0f, 1e-7f)); + REQUIRE_THAT(s, Catch::Matchers::WithinAbs(0.0f, 1e-7f)); + + A = Matrix<2, 2>{2, 0, 0, 5}; + QR::Solve2x2Eigen(A, 0, lHi, lLo, c, s); + REQUIRE_THAT(lHi, Catch::Matchers::WithinRel(5.0f, 1e-7f)); + REQUIRE_THAT(lLo, Catch::Matchers::WithinRel(2.0f, 1e-7f)); + REQUIRE_THAT(c, Catch::Matchers::WithinAbs(0.0f, 1e-7f)); + REQUIRE_THAT(s, Catch::Matchers::WithinRel(1.0f, 1e-7f)); + } +} +// ============================================================================ +// TEST 6: Deflate +// ============================================================================ +TEST_CASE("QR Building Block: Deflate", "[Matrix][QR]") { + // subdiag[0] = 1e-9 <= 1e-6 * (|2| + |3|) = 5e-6 -> deflated + // subdiag[1] = 0.5 > 1e-6 * (|3| + |4|) = 7e-6 -> kept + { + Matrix<3, 3> A{2, 1e-9f, 0, 1e-9f, 3, 0.5f, 0, 0.5f, 4}; + QR::Deflate(A, 0, 2, 1e-6f); + + REQUIRE(A.Get(1, 0) == 0.0f); + REQUIRE(A.Get(0, 1) == 0.0f); + REQUIRE_THAT(A.Get(2, 1), Catch::Matchers::WithinRel(0.5f, 1e-7f)); + REQUIRE_THAT(A.Get(1, 2), Catch::Matchers::WithinRel(0.5f, 1e-7f)); + // Diagonals untouched + REQUIRE_THAT(A.Get(0, 0), Catch::Matchers::WithinRel(2.0f, 1e-7f)); + REQUIRE_THAT(A.Get(1, 1), Catch::Matchers::WithinRel(3.0f, 1e-7f)); + REQUIRE_THAT(A.Get(2, 2), Catch::Matchers::WithinRel(4.0f, 1e-7f)); + } + + // Nothing deflated when all subdiagonals are well above tolerance + { + Matrix<3, 3> A{2, 0.1f, 0, 0.1f, 3, 0.2f, 0, 0.2f, 4}; + QR::Deflate(A, 0, 2, 1e-6f); + REQUIRE_THAT(A.Get(1, 0), Catch::Matchers::WithinRel(0.1f, 1e-7f)); + REQUIRE_THAT(A.Get(2, 1), Catch::Matchers::WithinRel(0.2f, 1e-7f)); + } +} + +// ============================================================================ +// TEST 7: Tridiagonalize +// ============================================================================ +TEST_CASE("QR Building Block: Tridiagonalize", "[Matrix][QR]") { + // 4x4 symmetric with a full (0,3) corner coupling + { + Matrix<4, 4> A{2, 1, 0, 1, 1, 3, 1, 0, 0, 1, 4, 1, 1, 0, 1, 5}; + Matrix<4, 4> Aorig = A; + Matrix<4, 4> U{0}; + + QR::Tridiagonalize(A, U); + + // Off-tridiagonal entries must be zero up to float32 roundoff (the + // Givens zeroing cancels only in exact arithmetic; residuals are + // ~1e-7 for O(1) entries). + REQUIRE_THAT(A.Get(0, 2), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + REQUIRE_THAT(A.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + REQUIRE_THAT(A.Get(0, 3), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + REQUIRE_THAT(A.Get(3, 0), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + REQUIRE_THAT(A.Get(1, 3), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + REQUIRE_THAT(A.Get(3, 1), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + + // Symmetry preserved exactly + for (uint8_t i = 0; i < 4; i++) + for (uint8_t j = 0; j < 4; j++) + REQUIRE(A.Get(i, j) == A.Get(j, i)); + + // U must be orthogonal + REQUIRE(isOrthogonal(U)); + + // Reconstruction: U * A_tri * U^T == Aorig (absolute check for + // originally-zero entries: WithinRel has no absolute fallback there) + Matrix<4, 4> UAt{}; + U.Mult(A, UAt); + Matrix<4, 4> UAtU{}; + UAt.Mult(U.Transpose(), UAtU); + for (uint8_t i = 0; i < 4; i++) + for (uint8_t j = 0; j < 4; j++) { + float actual = UAtU.Get(i, j); + float expected = Aorig.Get(i, j); + if (fabsf(expected) < 1e-3f) + REQUIRE_THAT(actual, Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + else + REQUIRE_THAT(actual, + Catch::Matchers::WithinRel(expected, 1e-5f)); + } + + // Spectrum invariants match the original + { + float tr0 = Aorig.Get(0, 0) + Aorig.Get(1, 1) + Aorig.Get(2, 2) + + Aorig.Get(3, 3); + float tr1 = A.Get(0, 0) + A.Get(1, 1) + A.Get(2, 2) + A.Get(3, 3); + REQUIRE_THAT(tr1, Catch::Matchers::WithinRel(tr0, 1e-6f)); + REQUIRE_THAT(frob(A), Catch::Matchers::WithinRel(frob(Aorig), 1e-6f)); + } + + // Eigenvalues of the tridiagonal match the original (scipy reference): + // 6.0, 4.0, 3.0, 1.0 + { + Matrix<4, 1> vals{}; + Matrix<4, 4> vecs{}; + QR::EigenQR(A, vecs, vals, 10000, 1e-6f); + REQUIRE_THAT(vals[0][0], Catch::Matchers::WithinRel(6.0f, 1e-4f)); + REQUIRE_THAT(vals[1][0], Catch::Matchers::WithinRel(4.0f, 1e-4f)); + REQUIRE_THAT(vals[2][0], Catch::Matchers::WithinRel(3.0f, 1e-4f)); + REQUIRE_THAT(vals[3][0], Catch::Matchers::WithinRel(1.0f, 1e-4f)); + } + } + + // 5x5 symmetric + { + Matrix<5, 5> A{3, 1, 0, 0, 1, 1, 4, 1, 0, 0, 0, 1, 5, 1, 0, 0, 0, 1, 6, 1, + 1, 0, 0, 1, 7}; + Matrix<5, 5> Aorig = A; + Matrix<5, 5> U{0}; + + QR::Tridiagonalize(A, U); + + // All |i - j| >= 2 entries zero up to float32 roundoff + for (uint8_t i = 0; i < 5; i++) + for (uint8_t j = 0; j < 5; j++) + if (i > j + 1 || j > i + 1) + REQUIRE_THAT(A.Get(i, j), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + + REQUIRE(isOrthogonal(U)); + + Matrix<5, 5> UAt{}; + U.Mult(A, UAt); + Matrix<5, 5> UAtU{}; + UAt.Mult(U.Transpose(), UAtU); + for (uint8_t i = 0; i < 5; i++) + for (uint8_t j = 0; j < 5; j++) { + float actual = UAtU.Get(i, j); + float expected = Aorig.Get(i, j); + if (fabsf(expected) < 1e-3f) + REQUIRE_THAT(actual, Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + else + REQUIRE_THAT(actual, + Catch::Matchers::WithinRel(expected, 1e-5f)); + } + } + + // Already tridiagonal: U must come out as the identity + { + Matrix<3, 3> A{1, 2, 0, 2, 5, 2, 0, 2, 9}; + Matrix<3, 3> U{0}; + QR::Tridiagonalize(A, U); + for (uint8_t i = 0; i < 3; i++) + for (uint8_t j = 0; j < 3; j++) { + float expected = (i == j) ? 1.0f : 0.0f; + REQUIRE_THAT(U.Get(i, j), Catch::Matchers::WithinAbs(expected, 1e-7f)); + } + } +} +// ============================================================================ +// TEST 8: One full shifted QR step (integration of the blocks) +// ============================================================================ +TEST_CASE("QR Building Block: Full Shifted QR Step", "[Matrix][QR]") { + // One Wilkinson-shifted QR step on the whole 3x3 block is a similarity + // transform, so all spectrum invariants (trace, sum of principal 2x2 + // minors, determinant) must be preserved. + // + // A = [[1,2,3],[2,5,8],[3,8,9]]: tr = 15, e2 = -18, det = -4 + { + Matrix<3, 3> A{1, 2, 3, 2, 5, 8, 3, 8, 9}; + float tr0 = trace3(A); // 15 + float e20 = e2_3x3(A); // -18 + float det0 = det3(A); // -4 + + // mu from the trailing 2x2 [[5,8],[8,9]]: eigenvalues + // -1.246211251, 15.246211251; closest to d = 9 is 15.246211251 (Wilkinson) + float mu = QR::WilkinsonShift(A.Get(1, 1), A.Get(2, 1), A.Get(2, 2)); + REQUIRE_THAT(mu, Catch::Matchers::WithinRel(15.246211251f, 1e-5f)); + + for (uint8_t i = 0; i < 3; i++) + A[i][i] -= mu; + + // Bulge chase: rotations on (0,1) then (1,2) + float c = 0, s = 0; + QR::GivensRotation(A.Get(0, 0), A.Get(1, 0), c, s); + QR::ApplyRotationBothSides(A, 0, c, s); + QR::GivensRotation(A.Get(1, 1), A.Get(2, 1), c, s); + QR::ApplyRotationBothSides(A, 1, c, s); + + for (uint8_t i = 0; i < 3; i++) + A[i][i] += mu; + + // Symmetry preserved + for (uint8_t i = 0; i < 3; i++) + for (uint8_t j = 0; j < 3; j++) + REQUIRE(A.Get(i, j) == A.Get(j, i)); + + // Spectrum invariants preserved + REQUIRE_THAT(trace3(A), Catch::Matchers::WithinRel(tr0, 1e-5f)); + REQUIRE_THAT(e2_3x3(A), Catch::Matchers::WithinRel(e20, 1e-5f)); + REQUIRE_THAT(det3(A), Catch::Matchers::WithinRel(det0, 1e-5f)); + } + + // For TRIDIAGONAL input a single step keeps the tridiagonal structure + { + Matrix<3, 3> T{1, 2, 0, 2, 5, 2, 0, 2, 9}; + float mu = QR::WilkinsonShift(T.Get(1, 1), T.Get(2, 1), T.Get(2, 2)); + for (uint8_t i = 0; i < 3; i++) + T[i][i] -= mu; + float c = 0, s = 0; + QR::GivensRotation(T.Get(0, 0), T.Get(1, 0), c, s); + QR::ApplyRotationBothSides(T, 0, c, s); + QR::GivensRotation(T.Get(1, 1), T.Get(2, 1), c, s); + QR::ApplyRotationBothSides(T, 1, c, s); + for (uint8_t i = 0; i < 3; i++) + T[i][i] += mu; + + // Corners must vanish up to float32 roundoff: tridiagonal form + // maintained. The cancellation is exact in exact arithmetic (the + // corner is s1*a - c1*b times a factor, and Givens gives s1*a = c1*b), + // so the residual is pure rounding, ~1e-6 for O(1) entries. + REQUIRE_THAT(T.Get(0, 2), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + REQUIRE_THAT(T.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + } +} diff --git a/unit-tests/qr-reference-values.py b/unit-tests/qr-reference-values.py new file mode 100644 index 0000000..8339763 --- /dev/null +++ b/unit-tests/qr-reference-values.py @@ -0,0 +1,246 @@ +#!/usr/bin/env python3 +""" +Reference values for the QR eigen-decomposition building block tests +(unit-tests/qr-build-blocks-tests.cpp). Run this to verify/implement the +C++ implementation in src/QR.hpp / src/QR.cpp against numpy/scipy. + +Conventions (match the C++ exactly): + * Givens zeroing rotation: G = [[c, s], [-s, c]], c = x/r, s = y/r, + r = hypot(x, y). G * (x, y)^T = (r, 0)^T. + * Similarity transform: A <- G A G^T (ApplyRotationBothSides). + * Eigenvector accumulation: V <- V G^T (ApplyRotationToVectors). + Vblock in the 2x2 closed form is [[c, -s], [s, c]] (same shape as G^T). + * Tridiagonalization: bottom-up Givens (i = N-2 down to k+1 per column k). + * Shifted QR loop: Wilkinson shift mu from the trailing 2x2, chase on the + trailing unreduced block [lo, hi], deflate by relative tolerance, peel + exact-zero subdiagonals, 2x2 closed-form termination. + * Pipeline: M0 = U * Mtri * U^T and Mtri = V * D * V^T => + eigenvectors of M0 = U * V (columns), eigenvalues = diag(D). + +Usage: python3 qr-reference-values.py +""" + +import numpy as np +import scipy.linalg as sla + +np.set_printoptions(precision=9, linewidth=120) + + +def givens(x, y): + """c = x/r, s = y/r with r = hypot(x, y).""" + r = np.hypot(x, y) + if r == 0.0: + return 1.0, 0.0 + return x / r, y / r + + +def rot(n, i, c, s): + """G = I with [[c, s], [-s, c]] embedded at (i, i+1).""" + G = np.eye(n) + G[i:i + 2, i:i + 2] = np.array([[c, s], [-s, c]]) + return G + + +def tridiagonalize(M0): + """Bottom-up Givens tridiagonalization. Returns (Mtri, U) with + M0 = U Mtri U^T.""" + n = len(M0) + M = M0.copy() + U = np.eye(n) + for k in range(n - 2): + for i in range(n - 2, k, -1): + c, s = givens(M[i, k], M[i + 1, k]) + G = rot(n, i, c, s) + M = G @ M @ G.T + U = U @ G.T + return M, U + + +def wilkinson(a, b, d): + """Eigenvalue of [[a, b], [b, d]] closest to d.""" + delta = 0.5 * (a - d) + spread = np.sqrt(delta * delta + b * b) + return 0.5 * (a + d) - (spread if delta >= 0 else -spread) + + +def solve2x2(A, lo): + """Closed form for the block at (lo, lo+1): (lHi, lLo, c, s) with + vHi = (c, s), vLo = (-s, c).""" + a = A[lo, lo] + b = A[lo, lo + 1] + e = A[lo + 1, lo] + d = A[lo + 1, lo + 1] + tr = a + d + det = a * d - b * e + disc = max(0.0, tr * tr - 4 * det) + lhi = 0.5 * (tr + np.sqrt(disc)) + llo = 0.5 * (tr - np.sqrt(disc)) + if b != 0.0: + v1 = lhi - a + nn = np.hypot(b, v1) + c, s = b / nn, v1 / nn + elif a >= d: + c, s = 1.0, 0.0 + else: + c, s = 0.0, 1.0 + return lhi, llo, c, s + + +def eigenqr(M0, tol=1e-12, max_iter=100000): + """Full pipeline mirroring QR::EigenQR. Returns (eigs, W) where W has + the eigenvectors of M0 as columns.""" + n = len(M0) + if n == 2: + l1, l2, c, s = solve2x2(M0, 0) + return np.array([l1, l2]), np.array([[c, -s], [s, c]]) + M, U = tridiagonalize(M0) + V = np.eye(n) + hi = n - 1 + for _ in range(max_iter): + # deflate: zero tiny subdiagonals (relative test) + for i in range(hi): + t = M[i + 1, i] + scale = abs(M[i, i]) + abs(M[i + 1, i + 1]) + if abs(t) <= tol * scale: + M[i + 1, i] = M[i, i + 1] = 0.0 + # peel exact-zero trailing subdiagonals + while hi > 0 and M[hi, hi - 1] == 0.0: + hi -= 1 + if hi == 0: + break + # find start of trailing unreduced block + lo = hi + for i in range(hi - 1, -1, -1): + if M[i + 1, i] == 0.0: + break + lo = i + if lo + 1 == hi: + # closed-form 2x2 termination: set diagonal, fold Vblock in + l1, l2, c, s = solve2x2(M, lo) + Vb = np.eye(n) + Vb[lo:lo + 2, lo:lo + 2] = np.array([[c, -s], [s, c]]) + V = V @ Vb + M[lo, lo] = l1 + M[lo + 1, lo + 1] = l2 + M[lo + 1, lo] = M[lo, lo + 1] = 0.0 + if lo == 0: + break + hi = lo - 1 + continue + # full shifted step on [lo, hi] (shift applies to the active block) + mu = wilkinson(M[hi - 1, hi - 1], M[hi, hi - 1], M[hi, hi]) + diag = M.diagonal().copy() + diag[lo:hi + 1] -= mu + np.fill_diagonal(M, diag) + c, s = givens(M[lo, lo], M[lo + 1, lo]) + G = rot(n, lo, c, s) + M = G @ M @ G.T + V = V @ G.T + for i in range(lo + 1, hi): + c, s = givens(M[i, i], M[i + 1, i]) + G = rot(n, i, c, s) + M = G @ M @ G.T + V = V @ G.T + diag = M.diagonal().copy() + diag[lo:hi + 1] += mu + np.fill_diagonal(M, diag) + + eigs = np.diag(M).astype(float) + order = np.argsort(eigs)[::-1] # descending, like the C++ test harness + eigs = eigs[order] + W = U @ V + W = W[:, order] + return eigs, W + + +def report(name, val, ref=None, tol=1e-6): + ok = "OK " if ref is None or np.allclose(val, ref, rtol=tol, atol=tol) else "FAIL" + print(f"[{ok}] {name} = {val}") + if ref is not None: + print(f" scipy/numpy ref = {ref}") + + +def main(): + print("=== TEST 1: GivensRotation ===") + c, s = givens(2.0, 1.0) + print(f" c = {c} s = {s}") + # G * (x, y)^T = (r, 0)^T: G = [[c, s], [-s, c]] + assert abs(c * 2 + s * 1 - np.sqrt(5)) < 1e-15 + assert abs(-s * 2 + c * 1) < 1e-15 + + print("\n=== TEST 2: ApplyRotationBothSides A <- G A G^T ===") + A = np.array([[3.0, 4.0, 5.0], [6.0, 7.0, 8.0], [9.0, 10.0, 11.0]]) + G = rot(3, 0, 0.6, 0.8) + B = G @ A @ G.T + print(B) + + A = np.array([[5.0, 0.0, 1.0], [0.0, 6.0, 2.0], [1.0, 2.0, 7.0]]) + c, s = givens(6.0, 2.0) + G = rot(3, 1, c, s) + B = G @ A @ G.T + print(B) + + print("\n=== TEST 3: V accumulation V <- V G^T ===") + V = np.eye(3) + G = rot(3, 0, 0.894427191, 0.447213595) + V = V @ G.T + print(V) + V2 = V @ rot(3, 1, 0.848874681, 0.528748047).T + print(V2) + + print("\n=== TEST 4: Solve2x2Eigen ===") + for A in (np.array([[5.0, 8.0], [8.0, 9.0]]), np.array([[1.0, 2.0], [3.0, 4.0]])): + l1, l2, c, s = solve2x2(A, 0) + ref = np.linalg.eigvalsh(A) if np.allclose(A, A.T) else np.linalg.eigvals(A) + print(f" A={A.ravel()} lHi={l1} lLo={l2} c={c} s={s} ref={np.sort(ref)[::-1]}") + + print("\n=== TEST 8: WilkinsonShift ===") + print(f" W(5, 8, 9) = {wilkinson(5, 8, 9)}") + print(f" W(4, 2, 7) = {wilkinson(4, 2, 7)}") + print(f" W(9, 2, 5) = {wilkinson(9, 2, 5)}") + + print("\n=== TEST 8b: one full shifted chase step on tridiagonal 3x3 ===") + T = np.array([[1.0, 2.0, 0.0], [2.0, 5.0, 2.0], [0.0, 2.0, 9.0]]) + mu = wilkinson(5, 2, 9) + M = T - mu * np.eye(3) + c, s = givens(M[0, 0], M[1, 0]) + M = rot(3, 0, c, s) @ M @ rot(3, 0, c, s).T + c, s = givens(M[1, 1], M[2, 1]) + M = rot(3, 1, c, s) @ M @ rot(3, 1, c, s).T + M = M + mu * np.eye(3) + print(f" mu = {mu}") + print(M) + print(f" corners: {M[0, 2]}, {M[2, 0]} (exact-arithmetic zeros)") + print(f" trace {M.trace():.15f} (was {T.trace()})") + + print("\n=== TEST 7: Tridiagonalize ===") + M4 = np.array([[2.0, 1, 0, 1], [1, 3, 1, 0], [0, 1, 4, 1], [1, 0, 1, 5]]) + M, U = tridiagonalize(M4) + print(" M4 tridiagonalized:\n", M) + print(f" reconstruction U M U^T == M4: {np.allclose(U @ M @ U.T, M4, atol=1e-9)}") + M5 = np.array([[3.0, 1, 0, 0, 1], [1, 4, 1, 0, 0], [0, 1, 5, 1, 0], + [0, 0, 1, 6, 1], [1, 0, 0, 1, 7]]) + M, U = tridiagonalize(M5) + print(" M5 tridiagonalized:\n", M) + print(f" reconstruction: {np.allclose(U @ M @ U.T, M5, atol=1e-9)}") + + print("\n=== End-to-end: random symmetric vs scipy.linalg.eigh ===") + rng = np.random.default_rng(12345) + worst = 0.0 + for n in range(3, 9): + M0 = rng.normal(size=(n, n)) + M0 = (M0 + M0.T) / 2 + eigs, W = eigenqr(M0.astype(float)) + ref = sla.eigh(M0) + e_err = np.max(np.abs(np.sort(eigs) - ref[0])) + resid = np.linalg.norm(W @ np.diag(eigs) @ W.T - M0) + ortho = np.linalg.norm(W.T @ W - np.eye(n)) + print(f" n={n}: eigs_err={e_err:.2e} resid={resid:.2e} ortho={ortho:.2e}") + worst = max(worst, e_err, resid, ortho) + print(f"\nworst over all n: {worst:.2e}") + assert worst < 1e-10, "end-to-end reference FAILED" + print("ALL REFERENCES OK") + + +if __name__ == "__main__": + main() diff --git a/unit-tests/svd-build-blocks-tests.cpp b/unit-tests/svd-build-blocks-tests.cpp new file mode 100644 index 0000000..fd03ef7 --- /dev/null +++ b/unit-tests/svd-build-blocks-tests.cpp @@ -0,0 +1,1647 @@ +// include the unit test framework first +#include +#include + +// include the module you're going to test next +#include "Matrix.hpp" +#include "SVD.hpp" + +// any other libraries +#include +#include +#include + +// ============================================================================ +// Helper: Frobenius norm of a 5×5 matrix +// ============================================================================ +static float frobeniusNorm5(const Matrix<5, 5> &M) { + float sum = 0.0f; + for (uint8_t i = 0; i < 5; i++) { + for (uint8_t j = 0; j < 5; j++) { + float v = M.Get(i, j); + sum += v * v; + } + } + return sqrtf(sum); +} + +// ============================================================================ +// Helper: Check if a matrix is orthogonal (Mᵀ·M ≈ I) +// ============================================================================ +static bool isOrthogonal5(const Matrix<5, 5> &M, float tol = 1e-6f) { + Matrix<5, 5> Mt = M.Transpose(); + Matrix<5, 5> MtM{0}; + Mt.Mult(M, MtM); + + for (uint8_t i = 0; i < 5; i++) { + for (uint8_t j = 0; j < 5; j++) { + float expected = (i == j) ? 1.0f : 0.0f; + if (fabsf(MtM.Get(i, j) - expected) > tol) { + return false; + } + } + } + return true; +} + +// ============================================================================ +// TEST 1: ComputeHouseholder +// ============================================================================ +TEST_CASE("SVD Building Block: ComputeHouseholder", "[Matrix][SVD]") { + // Test case: [3, 4] should give alpha = -5 (norm), v normalized + // Reference: scipy.linalg.householder([3, 4]) → v ≈ [0.894427191, + // 0.447213596], α = -5 + { + float x[] = {3.0f, 4.0f}; + float v[5] = {0}; + float alpha = 0; + + float norm = SVD::ComputeHouseholder(x, 2, v, alpha); + + // Norm should be 5.0 + REQUIRE_THAT(norm, Catch::Matchers::WithinRel(5.0f, 1e-6f)); + + // Alpha should be -5 (negative norm) + REQUIRE_THAT(alpha, Catch::Matchers::WithinRel(-5.0f, 1e-6f)); + + // v should be normalized: ||v|| ≈ 1 + float vNorm = sqrtf(v[0] * v[0] + v[1] * v[1]); + REQUIRE_THAT(vNorm, Catch::Matchers::WithinRel(1.0f, 1e-6f)); + + // Verify H·x = [alpha, 0]: (I - 2vvᵀ)·x should give [-5, 0] + float hx0 = x[0] - 2.0f * v[0] * (v[0] * x[0] + v[1] * x[1]); + float hx1 = x[1] - 2.0f * v[1] * (v[0] * x[0] + v[1] * x[1]); + REQUIRE_THAT(hx0, Catch::Matchers::WithinRel(alpha, 1e-6f)); + REQUIRE_THAT(hx1, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } + + // Test case: [1, 3] + // Reference: norm = √10 ≈ 3.16228, alpha = -√10 + { + float x[] = {1.0f, 3.0f}; + float v[5] = {0}; + float alpha = 0; + + float norm = SVD::ComputeHouseholder(x, 2, v, alpha); + + REQUIRE_THAT(norm, Catch::Matchers::WithinRel(sqrtf(10.0f), 1e-6f)); + REQUIRE_THAT(alpha, Catch::Matchers::WithinRel(-sqrtf(10.0f), 1e-6f)); + + // Verify H·x = [alpha, 0] + float dot = v[0] * x[0] + v[1] * x[1]; + float hx0 = x[0] - 2.0f * v[0] * dot; + float hx1 = x[1] - 2.0f * v[1] * dot; + REQUIRE_THAT(hx0, Catch::Matchers::WithinRel(alpha, 1e-6f)); + REQUIRE_THAT(hx1, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } + + // Test case: [1, 2, 3] (3D) + // Reference: norm = √14 ≈ 3.74166 + { + float x[] = {1.0f, 2.0f, 3.0f}; + float v[5] = {0}; + float alpha = 0; + + float norm = SVD::ComputeHouseholder(x, 3, v, alpha); + + REQUIRE_THAT(norm, Catch::Matchers::WithinRel(sqrtf(14.0f), 1e-6f)); + REQUIRE_THAT(alpha, Catch::Matchers::WithinRel(-sqrtf(14.0f), 1e-6f)); + + // Verify v is normalized + float vNorm = sqrtf(v[0] * v[0] + v[1] * v[1] + v[2] * v[2]); + REQUIRE_THAT(vNorm, Catch::Matchers::WithinRel(1.0f, 1e-6f)); + + // Verify H·x = [alpha, 0, 0] + float dot = v[0] * x[0] + v[1] * x[1] + v[2] * x[2]; + for (uint8_t i = 0; i < 3; i++) { + float hx_i = x[i] - 2.0f * v[i] * dot; + if (i == 0) { + REQUIRE_THAT(hx_i, Catch::Matchers::WithinRel(alpha, 1e-6f)); + } else { + REQUIRE_THAT(hx_i, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } + } + } + + // Test case: [0, 0, 1] (already has leading zeros) + { + float x[] = {0.0f, 0.0f, 1.0f}; + float v[5] = {0}; + float alpha = 0; + + float norm = SVD::ComputeHouseholder(x, 3, v, alpha); + + REQUIRE_THAT(norm, Catch::Matchers::WithinRel(1.0f, 1e-6f)); + REQUIRE_THAT(alpha, Catch::Matchers::WithinRel(-1.0f, 1e-6f)); + + // Verify H·x = [-1, 0, 0] + float dot = v[0] * x[0] + v[1] * x[1] + v[2] * x[2]; + float hx0 = x[0] - 2.0f * v[0] * dot; + float hx1 = x[1] - 2.0f * v[1] * dot; + float hx2 = x[2] - 2.0f * v[2] * dot; + REQUIRE_THAT(hx0, Catch::Matchers::WithinRel(alpha, 1e-6f)); + REQUIRE_THAT(hx1, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(hx2, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } + + // Test case: [5, -3, 2, 1] (4D) + { + float x[] = {5.0f, -3.0f, 2.0f, 1.0f}; + float v[5] = {0}; + float alpha = 0; + + float norm = SVD::ComputeHouseholder(x, 4, v, alpha); + + REQUIRE_THAT(norm, Catch::Matchers::WithinRel(sqrtf(39.0f), 1e-6f)); + REQUIRE_THAT(alpha, Catch::Matchers::WithinRel(-sqrtf(39.0f), 1e-6f)); + + // Verify H·x = [alpha, 0, 0, 0] + float dot = v[0] * x[0] + v[1] * x[1] + v[2] * x[2] + v[3] * x[3]; + for (uint8_t i = 0; i < 4; i++) { + float hx_i = x[i] - 2.0f * v[i] * dot; + if (i == 0) { + REQUIRE_THAT(hx_i, Catch::Matchers::WithinRel(alpha, 1e-6f)); + } else { + REQUIRE_THAT(hx_i, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } + } + } + + // Test case: zero vector + { + float x[] = {0.0f, 0.0f}; + float v[5] = {0}; + float alpha = 0; + + float norm = SVD::ComputeHouseholder(x, 2, v, alpha); + + REQUIRE_THAT(norm, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE(alpha == 0.0f); + } +} + +// ============================================================================ +// TEST 2: ApplyHouseholderLeft +// ============================================================================ +TEST_CASE("SVD Building Block: ApplyHouseholderLeft", "[Matrix][SVD]") { + // Test: Apply Householder to zero out column 0, rows 1:2 of a 3×3 matrix + // Input: [[1, 2, 3], [4, 5, 6], [7, 8, 9]] + // After applying HH on col 0 (rows 1:2): A[2,0] should be ~0 + { + Matrix<5, 5> W{1.0f, 2.0f, 3.0f, 0, 0, 4.0f, 5.0f, 6.0f, 0, + 0, 7.0f, 8.0f, 9.0f, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0}; + + // Compute Householder for column 0, rows 1:2 → vector [4, 7] + float x[] = {4.0f, 7.0f}; + float v[5] = {0}; + float alpha = 0; + SVD::ComputeHouseholder(x, 2, v, alpha); + + // Apply from left + SVD::ApplyHouseholderLeft(W, v, 1, 2); + + // A[2,0] should be ~0 + REQUIRE_THAT(W.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + + // Verify orthogonality of the transformation: W = H·W_original + Matrix<5, 5> W_orig{1.0f, 2.0f, 3.0f, 0, 0, 4.0f, 5.0f, 6.0f, 0, + 0, 7.0f, 8.0f, 9.0f, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0}; + + // Compute H_left explicitly: I - 2*v*vᵀ (on rows 1:2) + Matrix<5, 5> H_left{0}; + for (uint8_t i = 0; i < 5; i++) { + H_left[i][i] = 1.0f; + } + // Apply -2*v*vᵀ to the sub-block + float vv = v[0] * v[0] + v[1] * v[1]; + for (uint8_t i = 1; i <= 2; i++) { + for (uint8_t j = 1; j <= 2; j++) { + H_left[i][j] -= 2.0f * v[i - 1] * v[j - 1] / vv; + } + } + + // Verify: W ≈ H_left · W_orig + Matrix<5, 5> HLeftW{0}; + H_left.Mult(W_orig, HLeftW); + float err = frobeniusNorm5(W - HLeftW); + REQUIRE_THAT(err, Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + + // Verify H_left is orthogonal + REQUIRE(isOrthogonal5(H_left)); + } + + // Test: Apply to a larger block (4 rows) + { + Matrix<5, 5> W{1.0f, 2.0f, 0, 0, 0, 3.0f, 4.0f, 0, 0, + 0, 5.0f, 6.0f, 0, 0, 0, 7.0f, 8.0f, 0, + 0, 0, 0, 0, 0, 0, 0}; + + // Householder on [3, 5, 7] (rows 1:3) + float x[] = {3.0f, 5.0f, 7.0f}; + float v[5] = {0}; + float alpha = 0; + SVD::ComputeHouseholder(x, 3, v, alpha); + + SVD::ApplyHouseholderLeft(W, v, 1, 3); + + // A[2,0] and A[3,0] should be ~0 + REQUIRE_THAT(W.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(W.Get(3, 0), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } +} + +// ============================================================================ +// TEST 3: ApplyHouseholderRight +// ============================================================================ +TEST_CASE("SVD Building Block: ApplyHouseholderRight", "[Matrix][SVD]") { + // Test: Apply Householder to zero out row 0, cols 1:2 of a 3×3 matrix + // Input: [[1, 2, 3], [4, 5, 6], [7, 8, 9]] + // After applying HH on row 0 (cols 1:2): A[0,2] should be ~0 + { + Matrix<5, 5> W{1.0f, 2.0f, 3.0f, 0, 0, 4.0f, 5.0f, 6.0f, 0, + 0, 7.0f, 8.0f, 9.0f, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0}; + + // Householder for row 0, cols 1:2 → vector [2, 3] + float x[] = {2.0f, 3.0f}; + float v[5] = {0}; + float alpha = 0; + SVD::ComputeHouseholder(x, 2, v, alpha); + + // Apply from right + SVD::ApplyHouseholderRight(W, v, 1, 2); + + // A[0,2] should be ~0 + REQUIRE_THAT(W.Get(0, 2), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + + // Verify W ≈ W_orig · H_right + Matrix<5, 5> W_orig{1.0f, 2.0f, 3.0f, 0, 0, 4.0f, 5.0f, 6.0f, 0, + 0, 7.0f, 8.0f, 9.0f, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0}; + + // Compute H_right = I - 2*v*vᵀ (on cols 1:2) + Matrix<5, 5> H_right{0}; + for (uint8_t i = 0; i < 5; i++) { + H_right[i][i] = 1.0f; + } + float vv = v[0] * v[0] + v[1] * v[1]; + for (uint8_t i = 1; i <= 2; i++) { + for (uint8_t j = 1; j <= 2; j++) { + H_right[i][j] -= 2.0f * v[i - 1] * v[j - 1] / vv; + } + } + + Matrix<5, 5> WOrigH{0}; + W_orig.Mult(H_right, WOrigH); + float err = frobeniusNorm5(W - WOrigH); + REQUIRE_THAT(err, Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + + // Verify H_right is orthogonal + REQUIRE(isOrthogonal5(H_right)); + } + + // Test: Apply to wider block (4 cols) + { + Matrix<5, 5> W{1.0f, 2.0f, 3.0f, 4.0f, 0, 5.0f, 6.0f, 7.0f, 8.0f, + 0, 0, 0, 0, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0}; + + // Householder on [2, 3, 4] (cols 1:3) + float x[] = {2.0f, 3.0f, 4.0f}; + float v[5] = {0}; + float alpha = 0; + SVD::ComputeHouseholder(x, 3, v, alpha); + + SVD::ApplyHouseholderRight(W, v, 1, 3); + + // A[0,2] and A[0,3] should be ~0 + REQUIRE_THAT(W.Get(0, 2), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(W.Get(0, 3), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } +} + +// ============================================================================ +// TEST 4: ComputeGivens +// ============================================================================ +TEST_CASE("SVD Building Block: ComputeGivens", "[Matrix][SVD]") { + // Test case: [3, 4] → c = 0.6, s = 0.8 (3-4-5 triangle) + { + float c, s; + SVD::ComputeGivens(3.0f, 4.0f, c, s); + + REQUIRE_THAT(c, Catch::Matchers::WithinRel(0.6f, 1e-6f)); + REQUIRE_THAT(s, Catch::Matchers::WithinRel(0.8f, 1e-6f)); + + // Verify: [c s; -s c] · [3; 4] = [5; 0] + float r = c * 3.0f + s * 4.0f; + float z = -s * 3.0f + c * 4.0f; + REQUIRE_THAT(r, Catch::Matchers::WithinRel(5.0f, 1e-6f)); + REQUIRE_THAT(z, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + + // Verify c² + s² = 1 + REQUIRE_THAT(c * c + s * s, Catch::Matchers::WithinRel(1.0f, 1e-6f)); + } + + // Test case: [1, 0] → c = 1, s = 0 + { + float c, s; + SVD::ComputeGivens(1.0f, 0.0f, c, s); + REQUIRE_THAT(c, Catch::Matchers::WithinRel(1.0f, 1e-6f)); + REQUIRE_THAT(s, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } + + // Test case: [0, 5] → c = 0, s = 1 + { + float c, s; + SVD::ComputeGivens(0.0f, 5.0f, c, s); + REQUIRE_THAT(c, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(s, Catch::Matchers::WithinRel(1.0f, 1e-6f)); + + // Verify: [c s; -s c] · [0; 5] = [5; 0] + float r = c * 0.0f + s * 5.0f; + float z = -s * 0.0f + c * 5.0f; + REQUIRE_THAT(r, Catch::Matchers::WithinRel(5.0f, 1e-6f)); + REQUIRE_THAT(z, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } + + // Test case: [-3, -4] → c = -0.6, s = -0.8 + { + float c, s; + SVD::ComputeGivens(-3.0f, -4.0f, c, s); + REQUIRE_THAT(c, Catch::Matchers::WithinRel(-0.6f, 1e-6f)); + REQUIRE_THAT(s, Catch::Matchers::WithinRel(-0.8f, 1e-6f)); + + // Verify: [c s; -s c] · [-3; -4] = [5; 0] + float r = c * (-3.0f) + s * (-4.0f); + float z = -s * (-3.0f) + c * (-4.0f); + REQUIRE_THAT(r, Catch::Matchers::WithinRel(5.0f, 1e-6f)); + REQUIRE_THAT(z, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } + + // Test case: [1, -1] → c = 1/√2, s = -1/√2 (45°) + { + float c, s; + SVD::ComputeGivens(1.0f, -1.0f, c, s); + float invSqrt2 = 1.0f / sqrtf(2.0f); + REQUIRE_THAT(c, Catch::Matchers::WithinRel(invSqrt2, 1e-6f)); + REQUIRE_THAT(s, Catch::Matchers::WithinRel(-invSqrt2, 1e-6f)); + + // Verify: [c s; -s c] · [1; -1] = [√2; 0] + float r = c * 1.0f + s * (-1.0f); + float z = -s * 1.0f + c * (-1.0f); + REQUIRE_THAT(r, Catch::Matchers::WithinRel(sqrtf(2.0f), 1e-6f)); + REQUIRE_THAT(z, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } + + // Test case: [0, 0] → c = 1, s = 0 (identity) + { + float c, s; + SVD::ComputeGivens(0.0f, 0.0f, c, s); + REQUIRE_THAT(c, Catch::Matchers::WithinRel(1.0f, 1e-6f)); + REQUIRE_THAT(s, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } + + // Test case: [7, 24] → c = 7/25, s = 24/25 (7-24-25 triangle) + { + float c, s; + SVD::ComputeGivens(7.0f, 24.0f, c, s); + REQUIRE_THAT(c, Catch::Matchers::WithinRel(7.0f / 25.0f, 1e-6f)); + REQUIRE_THAT(s, Catch::Matchers::WithinRel(24.0f / 25.0f, 1e-6f)); + + float r = c * 7.0f + s * 24.0f; + float z = -s * 7.0f + c * 24.0f; + REQUIRE_THAT(r, Catch::Matchers::WithinRel(25.0f, 1e-6f)); + REQUIRE_THAT(z, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } +} + +// ============================================================================ +// TEST 5: ApplyGivensLeft +// ============================================================================ +TEST_CASE("SVD Building Block: ApplyGivensLeft", "[Matrix][SVD]") { + // Test: Apply Givens to zero out W[1,0] of a 2×2 matrix + // Input: [[3, 4], [1, 2]] + // Givens on rows 0,1 with x=W[0,0]=3, y=W[1,0]=1 + { + Matrix<5, 5> W{3.0f, 4.0f, 0, 0, 0, 1.0f, 2.0f, 0, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0}; + + float c, s; + SVD::ComputeGivens(3.0f, 1.0f, c, s); + + SVD::ApplyGivensLeft(W, 0, 1, c, s, 0, 4); + + // W[1,0] should be ~0 + REQUIRE_THAT(W.Get(1, 0), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + + // Verify W ≈ G · W_orig + Matrix<5, 5> W_orig{3.0f, 4.0f, 0, 0, 0, 1.0f, 2.0f, 0, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0}; + + // Givens rotation matrix (5×5) + Matrix<5, 5> G{0}; + for (uint8_t i = 0; i < 5; i++) { + G[i][i] = 1.0f; + } + G[0][0] = c; + G[0][1] = s; + G[1][0] = -s; + G[1][1] = c; + + Matrix<5, 5> GW{0}; + G.Mult(W_orig, GW); + float err = frobeniusNorm5(W - GW); + REQUIRE_THAT(err, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + + // Verify G is orthogonal + REQUIRE(isOrthogonal5(G)); + } + + // Test: Apply to larger range of columns + { + Matrix<5, 5> W{3.0f, 4.0f, 5.0f, 6.0f, 7.0f, 1.0f, 2.0f, 3.0f, 4.0f, + 5.0f, 0, 0, 0, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0}; + + float c, s; + SVD::ComputeGivens(3.0f, 1.0f, c, s); + + SVD::ApplyGivensLeft(W, 0, 1, c, s, 0, 4); + + REQUIRE_THAT(W.Get(1, 0), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } +} + +// ============================================================================ +// TEST 6: ApplyGivensRight +// ============================================================================ +TEST_CASE("SVD Building Block: ApplyGivensRight", "[Matrix][SVD]") { + // Test: Apply Givens to zero out W[0,1] of a 2×2 matrix + // Input: [[3, 4], [1, 2]] + // Givens on cols 0,1 with x=W[0,0]=3, y=W[0,1]=4 + { + Matrix<5, 5> W{3.0f, 4.0f, 0, 0, 0, 1.0f, 2.0f, 0, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0}; + + float c, s; + SVD::ComputeGivens(3.0f, 4.0f, c, s); + + SVD::ApplyGivensRight(W, 0, 1, c, s, 0, 4); + + // W[0,1] should be ~0 + REQUIRE_THAT(W.Get(0, 1), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + + // Verify W ≈ W_orig · G + Matrix<5, 5> W_orig{3.0f, 4.0f, 0, 0, 0, 1.0f, 2.0f, 0, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0}; + + // Givens rotation matrix (5×5) + Matrix<5, 5> G{0}; + for (uint8_t i = 0; i < 5; i++) { + G[i][i] = 1.0f; + } + G[0][0] = c; + G[0][1] = -s; + G[1][0] = s; + G[1][1] = c; + + Matrix<5, 5> WG{0}; + W_orig.Mult(G, WG); + float err = frobeniusNorm5(W - WG); + REQUIRE_THAT(err, Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + + // Verify G is orthogonal + REQUIRE(isOrthogonal5(G)); + } + + // Test: Apply to larger range of rows + { + Matrix<5, 5> W{3.0f, 4.0f, 0, 0, 0, 1.0f, 2.0f, 0, 0, 0, 5.0f, 6.0f, 0, + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0}; + + float c, s; + SVD::ComputeGivens(3.0f, 4.0f, c, s); + + SVD::ApplyGivensRight(W, 0, 1, c, s, 0, 2); + + REQUIRE_THAT(W.Get(0, 1), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } +} + +// ============================================================================ +// TEST 7: Full Bidiagonalization (composing Householder steps) +// ============================================================================ +TEST_CASE("SVD Building Block: Householder Bidiagonalization", + "[Matrix][SVD]") { + // Test: Bidiagonalize a 3×3 matrix and verify reconstruction + // Input: [[1, 2, 3], [4, 5, 6], [7, 8, 10]] + { + Matrix<5, 5> W{1.0f, 2.0f, 3.0f, 0, 0, 4.0f, 5.0f, 6.0f, 0, + 0, 7.0f, 8.0f, 10.0f, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0}; + + // Step 1: Left HH on column 0, rows 1:2 → zero out W[2,0] + { + float x[] = {4.0f, 7.0f}; + float v[5] = {0}; + float alpha = 0; + SVD::ComputeHouseholder(x, 2, v, alpha); + SVD::ApplyHouseholderLeft(W, v, 1, 2); + } + + // Step 2: Right HH on row 0, cols 1:2 → zero out W[0,2] + { + float x[] = {W.Get(0, 1), W.Get(0, 2)}; + float v[5] = {0}; + float alpha = 0; + SVD::ComputeHouseholder(x, 2, v, alpha); + SVD::ApplyHouseholderRight(W, v, 1, 2); + } + + // Step 3: Left HH on column 1, rows 2:2 → nothing to do (single element) + + // Verify bidiagonal structure: for 3x3, zero elements are A[2][0] (below + // subdiag in col 0) and A[0][2] (above superdiag in row 0) A[2][1] is the + // subdiagonal element of col 1 — valid in bidiagonal form + REQUIRE_THAT(W.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + REQUIRE_THAT(W.Get(0, 2), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + } + + // Test: Bidiagonalize a 4×3 matrix + { + Matrix<5, 5> W{1.0f, 2.0f, 3.0f, 0, 0, 4.0f, 5.0f, 6.0f, 0, + 0, 7.0f, 8.0f, 9.0f, 0, 0, 10.0f, 11.0f, 12.0f, + 0, 0, 0, 0, 0, 0, 0}; + + // Step 1: Left HH on col 0, rows 1:3 → zero out W[2,0], W[3,0] + { + float x[] = {4.0f, 7.0f, 10.0f}; + float v[5] = {0}; + float alpha = 0; + SVD::ComputeHouseholder(x, 3, v, alpha); + SVD::ApplyHouseholderLeft(W, v, 1, 3); + } + + // Step 2: Right HH on row 0, cols 1:2 → zero out W[0,2] + { + float x[] = {W.Get(0, 1), W.Get(0, 2)}; + float v[5] = {0}; + float alpha = 0; + SVD::ComputeHouseholder(x, 2, v, alpha); + SVD::ApplyHouseholderRight(W, v, 1, 2); + } + + // Step 3: Left HH on col 1, rows 2:3 → zero out W[3,1] + { + float x[] = {W.Get(2, 1), W.Get(3, 1)}; + float v[5] = {0}; + float alpha = 0; + SVD::ComputeHouseholder(x, 2, v, alpha); + SVD::ApplyHouseholderLeft(W, v, 2, 3); + } + + // Verify bidiagonal structure + REQUIRE_THAT(W.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + REQUIRE_THAT(W.Get(3, 0), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + REQUIRE_THAT(W.Get(3, 1), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + } + + // Test: Diagonal matrix (no transformations needed) + { + Matrix<5, 5> W{10.0f, 0, 0, 0, 0, 0, 5.0f, 0, 0, 0, 0, 0, 2.0f, + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0}; + + // Householder on zero vector should be identity + float x[] = {0.0f, 0.0f}; + float v[5] = {0}; + float alpha = 0; + SVD::ComputeHouseholder(x, 2, v, alpha); + + // Applying identity should not change anything + Matrix<5, 5> W_copy{10.0f, 0, 0, 0, 0, 0, 5.0f, 0, 0, 0, 0, 0, 2.0f, + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0}; + SVD::ApplyHouseholderLeft(W_copy, v, 1, 2); + + REQUIRE_THAT(frobeniusNorm5(W - W_copy), + Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } +} + +// ============================================================================ +// // +// ============================================================================ +// TEST 8: Givens QR step on bidiagonal matrix +// =========================================================================== +TEST_CASE("SVD Building Block: Givens QR Step on Bidiagonal", "[Matrix][SVD]") { + // Test: Apply left Givens to zero subdiagonal of a bidiagonal matrix, + // then apply right Givens with restricted row range to restore bidiagonal + // form. + // + // Input: 3x3 bidiagonal [[1, 2, 0], [3, -4, 5], [0, 6, -7]] + // Step 1: Left Givens on rows 0,1 with x=W[0][0]=1, y=W[1][0]=3 -> zero + // W[1][0] Step 2: Right Givens on cols 1,2 with x=W[0][1], y=W[0][2] -> zero + // W[0][2] + // Only applied to row 0 (to not reintroduce subdiagonal non-zeros) + { + Matrix<5, 5> W{1.0f, 2.0f, 0.0f, 0, 0, 3.0f, -4.0f, 5.0f, 0, + 0, 0.0f, 6.0f, -7.0f, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0}; + + float c, s; + SVD::ComputeGivens(W.Get(0, 0), W.Get(1, 0), c, s); + + // Apply from left to zero subdiagonal at W[1][0] + SVD::ApplyGivensLeft(W, 0, 1, c, s, 0, 4); + + REQUIRE_THAT(W.Get(1, 0), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + + // After left Givens, W[0][2] may have become non-zero (fill-in from row 0) + // Apply right Givens to cols 1,2 with x=W[0][1], y=W[0][2] -> zero W[0][2] + // Only apply to rows 0 (to preserve bidiagonal structure below row 0) + float c2, s2; + SVD::ComputeGivens(W.Get(0, 1), W.Get(0, 2), c2, s2); + SVD::ApplyGivensRight(W, 1, 2, c2, s2, 0, 0); + + // Should be bidiagonal: W[1][0] ~ 0 (from left Givens), W[0][2] ~ 0 (from + // right Givens) + REQUIRE_THAT(W.Get(1, 0), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + REQUIRE_THAT(W.Get(0, 2), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + } + + // Test: Verify that a full QR step (left + right Givens) preserves the + // bidiagonal structure when applied correctly with proper row ranges. + { + Matrix<5, 5> W{2.0f, 3.0f, 0, 0, 0, -1.0f, 4.0f, 5.0f, 0, 0, 0, 6.0f, -7.0f, + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0}; + + // Left Givens on col 0 (rows 0,1) + float c, s; + SVD::ComputeGivens(W.Get(0, 0), W.Get(1, 0), c, s); + SVD::ApplyGivensLeft(W, 0, 1, c, s, 0, 4); + + REQUIRE_THAT(W.Get(1, 0), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + + // Right Givens on row 0 (cols 1,2) - only affect row 0 + float c2, s2; + SVD::ComputeGivens(W.Get(0, 1), W.Get(0, 2), c2, s2); + SVD::ApplyGivensRight(W, 1, 2, c2, s2, 0, 0); + + // Bidiagonal structure preserved + REQUIRE_THAT(W.Get(1, 0), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + REQUIRE_THAT(W.Get(0, 2), Catch::Matchers::WithinAbs(0.0f, 1e-5f)); + } +} + +// ============================================================================TEST +// 9: Orthogonality preservation of Householder transformations +// ============================================================================ +TEST_CASE("SVD Building Block: Householder preserves orthogonality", + "[Matrix][SVD]") { + // Starting with an orthogonal matrix, applying Householder should preserve it + { + // Identity matrix is orthogonal + Matrix<5, 5> M{0}; + for (uint8_t i = 0; i < 5; i++) { + M[i][i] = 1.0f; + } + + // Householder on first 3 elements of column 0 + float x[] = {1.0f, 0.0f, 0.0f}; + float v[5] = {0}; + float alpha = 0; + SVD::ComputeHouseholder(x, 3, v, alpha); + + // Apply from left + Matrix<5, 5> M_left = M; + SVD::ApplyHouseholderLeft(M_left, v, 0, 2); + + // M_left should still be orthogonal + REQUIRE(isOrthogonal5(M_left)); + + // Apply from right + Matrix<5, 5> M_right = M; + SVD::ApplyHouseholderRight(M_right, v, 0, 2); + + REQUIRE(isOrthogonal5(M_right)); + } + + // Random orthogonal matrix (rotation) + { + float c = sqrtf(0.5f); + float s = sqrtf(0.5f); + Matrix<5, 5> M{0}; + M[0][0] = c; + M[0][1] = -s; + M[1][0] = s; + M[1][1] = c; + for (uint8_t i = 2; i < 5; i++) { + M[i][i] = 1.0f; + } + + REQUIRE(isOrthogonal5(M)); + + // Apply Householder on rows 0,1 + float x[] = {c, s}; + float v[5] = {0}; + float alpha = 0; + SVD::ComputeHouseholder(x, 2, v, alpha); + + Matrix<5, 5> M_test = M; + SVD::ApplyHouseholderLeft(M_test, v, 0, 1); + + REQUIRE(isOrthogonal5(M_test)); + } +} + +// ============================================================================ +// TEST 10: Orthogonality preservation of Givens transformations +// ============================================================================ +TEST_CASE("SVD Building Block: Givens preserves orthogonality", + "[Matrix][SVD]") { + // Starting with an orthogonal matrix, applying Givens should preserve it + { + Matrix<5, 5> M{0}; + for (uint8_t i = 0; i < 5; i++) { + M[i][i] = 1.0f; + } + + float c, s; + SVD::ComputeGivens(3.0f, 4.0f, c, s); + + // Apply from left + Matrix<5, 5> M_left = M; + SVD::ApplyGivensLeft(M_left, 0, 1, c, s, 0, 4); + + REQUIRE(isOrthogonal5(M_left)); + + // Apply from right + Matrix<5, 5> M_right = M; + SVD::ApplyGivensRight(M_right, 0, 1, c, s, 0, 4); + + REQUIRE(isOrthogonal5(M_right)); + } +} + +// ============================================================================ +// TEST 11: ExtractAndSortSingularValues (Phase 3) +// =========================================================================== +TEST_CASE("SVD Phase 3: ExtractAndSortSingularValues", "[Matrix][SVD]") { + // Test case 1: Diagonal matrix with unordered singular values + { + Matrix<5, 5> W{0}; + W[0][0] = 2.0f; + W[1][1] = 10.0f; + W[2][2] = 5.0f; + + Matrix<5, 1> sigma{0}; + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::ExtractAndSortSingularValues(W, sigma, 3, QL, QR); + + // Singular values should be sorted descending: [10, 5, 2] + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(10.0f, 1e-6f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(5.0f, 1e-6f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(2.0f, 1e-6f)); + + // QL and QR columns should have been swapped to match the sort order + // Original: col 0 → σ=2, col 1 → σ=10, col 2 → σ=5 + // After sort: col 0 has σ=10 (was orig col 1), col 1 has σ=5 (was orig col 2), + // col 2 has σ=2 (was orig col 0) + // Starting from identity: QL[:,0] = e₀, QL[:,1] = e₁, QL[:,2] = e₂ + // After swaps: QL[:,0] = e₁, QL[:,1] = e₂, QL[:,2] = e₀ + REQUIRE_THAT(QL.Get(0, 0), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(QL.Get(1, 0), Catch::Matchers::WithinRel(1.0f, 1e-6f)); + REQUIRE_THAT(QL.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(QL.Get(0, 1), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(QL.Get(1, 1), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(QL.Get(2, 1), Catch::Matchers::WithinRel(1.0f, 1e-6f)); + REQUIRE_THAT(QL.Get(0, 2), Catch::Matchers::WithinRel(1.0f, 1e-6f)); + REQUIRE_THAT(QL.Get(1, 2), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(QL.Get(2, 2), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + } + + // Test case 2: Negative diagonal elements (absolute value extraction) + { + Matrix<5, 5> W{0}; + W[0][0] = -3.0f; + W[1][1] = -7.0f; + W[2][2] = 5.0f; + + Matrix<5, 1> sigma{0}; + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::ExtractAndSortSingularValues(W, sigma, 3, QL, QR); + + // Should extract absolute values and sort: [7, 5, 3] + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(7.0f, 1e-6f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(5.0f, 1e-6f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(3.0f, 1e-6f)); + } + + // Test case 3: Already sorted (no swaps needed) + { + Matrix<5, 5> W{0}; + W[0][0] = 9.0f; + W[1][1] = 6.0f; + W[2][2] = 3.0f; + + Matrix<5, 1> sigma{0}; + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::ExtractAndSortSingularValues(W, sigma, 3, QL, QR); + + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(9.0f, 1e-6f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(6.0f, 1e-6f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(3.0f, 1e-6f)); + + // QL and QR should be unchanged (identity) + for (uint8_t i = 0; i < 5; i++) { + for (uint8_t j = 0; j < 5; j++) { + float expected = (i == j) ? 1.0f : 0.0f; + REQUIRE_THAT(QL.Get(i, j), Catch::Matchers::WithinRel(expected, 1e-6f)); + REQUIRE_THAT(QR.Get(i, j), Catch::Matchers::WithinRel(expected, 1e-6f)); + } + } + } + + // Test case 4: Single singular value + { + Matrix<5, 5> W{0}; + W[0][0] = 42.0f; + + Matrix<5, 1> sigma{0}; + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::ExtractAndSortSingularValues(W, sigma, 1, QL, QR); + + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(42.0f, 1e-6f)); + } +} + +// ============================================================================ +// TEST 12: AssembleUAndVt (Phase 4) +// =========================================================================== +TEST_CASE("SVD Phase 4: AssembleUAndVt", "[Matrix][SVD]") { + // Test case 1: Non-transpose case (m ≥ n) — U from QL, Vt from QRᵀ + { + uint8_t m = 3, n = 2, p = 2; + bool transposeNeeded = false; + + Matrix<5, 5> QL{0}; + // Make columns orthonormal + QL[0][0] = 3.0f / 5.0f; + QL[1][0] = 4.0f / 5.0f; + QL[2][0] = 0.0f; + QL[0][1] = 4.0f / 5.0f; + QL[1][1] = -3.0f / 5.0f; + QL[2][1] = 0.0f; + + Matrix<5, 5> QR{0}; + QR[0][0] = 1.0f; // Vt[:,0]ᵀ + QR[1][0] = 0.0f; + QR[0][1] = 0.0f; // Vt[:,1]ᵀ + QR[1][1] = 1.0f; + + Matrix<5, 5> U{0}; + Matrix<5, 5> Vt{0}; + + SVD::AssembleUAndVt(m, n, p, transposeNeeded, QL, QR, U, Vt); + + // U should be QL[:,0:2] + REQUIRE_THAT(U.Get(0, 0), Catch::Matchers::WithinRel(3.0f / 5.0f, 1e-6f)); + REQUIRE_THAT(U.Get(1, 0), Catch::Matchers::WithinRel(4.0f / 5.0f, 1e-6f)); + REQUIRE_THAT(U.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(U.Get(0, 1), Catch::Matchers::WithinRel(4.0f / 5.0f, 1e-6f)); + REQUIRE_THAT(U.Get(1, 1), Catch::Matchers::WithinRel(-3.0f / 5.0f, 1e-6f)); + REQUIRE_THAT(U.Get(2, 1), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + + // Vt should be QR[:,0:2]ᵀ + REQUIRE_THAT(Vt.Get(0, 0), Catch::Matchers::WithinRel(1.0f, 1e-6f)); + REQUIRE_THAT(Vt.Get(0, 1), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(Vt.Get(1, 0), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(Vt.Get(1, 1), Catch::Matchers::WithinRel(1.0f, 1e-6f)); + + // Verify U is orthogonal (first p columns) + Matrix<5, 5> Ut = U.Transpose(); + Matrix<5, 5> UtU{0}; + Ut.Mult(U, UtU); + REQUIRE_THAT(UtU.Get(0, 0), Catch::Matchers::WithinRel(1.0f, 1e-6f)); + REQUIRE_THAT(UtU.Get(0, 1), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(UtU.Get(1, 0), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(UtU.Get(1, 1), Catch::Matchers::WithinRel(1.0f, 1e-6f)); + } + + // Test case 2: Transpose case (m < n) — U from QRᵀ, Vt from QLᵀ + { + uint8_t m = 2, n = 3, p = 2; + bool transposeNeeded = true; + + Matrix<5, 5> QL{0}; + QL[0][0] = 1.0f; // Vt[:,0]ᵀ + QL[1][0] = 0.0f; + QL[0][1] = 0.0f; // Vt[:,1]ᵀ + QL[1][1] = 1.0f; + + Matrix<5, 5> QR{0}; + QR[0][0] = 3.0f / 5.0f; // U[:,0] + QR[1][0] = 4.0f / 5.0f; + QR[2][0] = 0.0f; + QR[0][1] = 4.0f / 5.0f; + QR[1][1] = -3.0f / 5.0f; + QR[2][1] = 0.0f; + + Matrix<5, 5> U{0}; + Matrix<5, 5> Vt{0}; + + SVD::AssembleUAndVt(m, n, p, transposeNeeded, QL, QR, U, Vt); + + // U should be QR[:,0:2]ᵀ → U[i][j] = QR[j][i] + REQUIRE_THAT(U.Get(0, 0), Catch::Matchers::WithinRel(3.0f / 5.0f, 1e-6f)); + REQUIRE_THAT(U.Get(0, 1), Catch::Matchers::WithinRel(4.0f / 5.0f, 1e-6f)); + REQUIRE_THAT(U.Get(1, 0), Catch::Matchers::WithinRel(4.0f / 5.0f, 1e-6f)); + REQUIRE_THAT(U.Get(1, 1), Catch::Matchers::WithinRel(-3.0f / 5.0f, 1e-6f)); + + // Vt should be QL[:,0:2]ᵀ → Vt[i][j] = QL[j][i] + REQUIRE_THAT(Vt.Get(0, 0), Catch::Matchers::WithinRel(1.0f, 1e-6f)); + REQUIRE_THAT(Vt.Get(0, 1), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(Vt.Get(1, 0), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + REQUIRE_THAT(Vt.Get(1, 1), Catch::Matchers::WithinRel(1.0f, 1e-6f)); + + // Verify U has correct dimensions (m×n = 2×3) + REQUIRE(U.Get(0, 2) == 0.0f); + REQUIRE(U.Get(1, 2) == 0.0f); + + // Verify Vt is orthogonal (first p rows) + Matrix<5, 5> VtVtT{0}; + Vt.Mult(Vt.Transpose(), VtVtT); + REQUIRE_THAT(VtVtT.Get(0, 0), Catch::Matchers::WithinRel(1.0f, 1e-6f)); + REQUIRE_THAT(VtVtT.Get(1, 1), Catch::Matchers::WithinRel(1.0f, 1e-6f)); + // Row 2 of Vt is all zeros (p=2 < n=3), so VtVtT[2][2] = 0 is expected + } + + // Test case 3: Square matrix (m = n) + { + uint8_t m = 2, n = 2, p = 2; + bool transposeNeeded = false; + + Matrix<5, 5> QL{0}; + QL[0][0] = 1.0f; QL[1][1] = 1.0f; + + Matrix<5, 5> QR{0}; + QR[0][0] = 0.6f; QR[0][1] = 0.8f; + QR[1][0] = 0.8f; QR[1][1] = -0.6f; + + Matrix<5, 5> U{0}; + Matrix<5, 5> Vt{0}; + + SVD::AssembleUAndVt(m, n, p, transposeNeeded, QL, QR, U, Vt); + + // U = QL[:,0:2] + REQUIRE_THAT(U.Get(0, 0), Catch::Matchers::WithinRel(1.0f, 1e-6f)); + REQUIRE_THAT(U.Get(1, 1), Catch::Matchers::WithinRel(1.0f, 1e-6f)); + + // Vt = QR[:,0:2]ᵀ + REQUIRE_THAT(Vt.Get(0, 0), Catch::Matchers::WithinRel(0.6f, 1e-6f)); + REQUIRE_THAT(Vt.Get(0, 1), Catch::Matchers::WithinRel(0.8f, 1e-6f)); + REQUIRE_THAT(Vt.Get(1, 0), Catch::Matchers::WithinRel(0.8f, 1e-6f)); + REQUIRE_THAT(Vt.Get(1, 1), Catch::Matchers::WithinRel(-0.6f, 1e-6f)); + } +} + +// ============================================================================ +// TEST 13: Bidiagonalize (Phase 1) - Square matrix +// =========================================================================== +TEST_CASE("SVD Phase 1: Bidiagonalize square matrix", "[Matrix][SVD]") { + // Test case 1: 3×3 matrix + // C++ verified reference: + // W[0] = [-4.123106, -5.335784, 6.548462] + // W[1] = [ 0.000000, 7.037714, -8.107580] + // W[2] = [ 0.000000, 0.000000, 0.620321] + // Note: W[0][2]=6.548462 is NOT zeroed because right HH at k=0 has only 1 element + { + Matrix<5, 5> W{1.0f, 2.0f, 3.0f, 0.0f, 0.0f, + 4.0f, 5.0f, 6.0f, 0.0f, 0.0f, + 0.0f, 7.0f, 8.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::Bidiagonalize(W, 3, 3, 3, QL, QR); + + // Subdiagonal elements should be zero: W[1][0], W[2][0], W[2][1] + REQUIRE_THAT(W.Get(1, 0), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + REQUIRE_THAT(W.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + REQUIRE_THAT(W.Get(2, 1), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + + // Verify orthogonality + REQUIRE(isOrthogonal5(QL)); + REQUIRE(isOrthogonal5(QR)); + } + + // Test case 2: 2×2 matrix (simplest non-trivial case) + // C++ verified reference: + // W[0] = [-3.162278, -4.427189] + // W[1] = [-0.000000, 0.632456] + { + Matrix<5, 5> W{3.0f, 4.0f, 0.0f, 0.0f, 0.0f, + 1.0f, 2.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::Bidiagonalize(W, 2, 2, 2, QL, QR); + + // For 2×2, bidiagonal form has no elements to zero out + REQUIRE(isOrthogonal5(QL)); + REQUIRE(isOrthogonal5(QR)); + } + + // Test case 3: Diagonal matrix (no transformations needed) + // C++ verified reference: unchanged + { + Matrix<5, 5> W{10.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 5.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 2.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + + Matrix<5, 5> W_orig{10.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 5.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 2.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::Bidiagonalize(W, 3, 3, 3, QL, QR); + + // Diagonal matrix may have sign flips but absolute values preserved + float err = 0.0f; + for (uint8_t i = 0; i < 3; i++) { + float diff = fabsf(W.Get(i, i)) - fabsf(W_orig.Get(i, i)); + err += diff * diff; + } + REQUIRE_THAT(sqrtf(err), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + + // QL and QR may have sign flips but should remain orthogonal + // Check that |QL[i][j]| and |QR[i][j]| match identity pattern + for (uint8_t i = 0; i < 5; i++) { + for (uint8_t j = 0; j < 5; j++) { + float expected = (i == j) ? 1.0f : 0.0f; + REQUIRE_THAT(fabsf(QL.Get(i, j)), Catch::Matchers::WithinAbs(expected, 1e-6f)); + REQUIRE_THAT(fabsf(QR.Get(i, j)), Catch::Matchers::WithinAbs(expected, 1e-6f)); + } + } + } +} + +// ============================================================================ +// TEST 14: Bidiagonalize (Phase 1) — Tall matrix (m > n) +// =========================================================================== +TEST_CASE("SVD Phase 1: Bidiagonalize tall matrix", "[Matrix][SVD]") { + // Test case 1: 4×3 matrix + // C++ verified reference (partial - subdiagonal zeros): + // W[0] = [-4.123106, -5.335784, 6.548462] + // W[1] = [-0.000000, 12.228222, 12.026182] + // W[2] = [ 0.000000, 0.000000, -1.577527] + // W[3] = [ 0.000000, 0.000000, 0.000000] + { + Matrix<5, 5> W{1.0f, 2.0f, 3.0f, 0.0f, 0.0f, + 4.0f, 5.0f, 6.0f, 0.0f, 0.0f, + 0.0f, 7.0f, 8.0f, 0.0f, 0.0f, + 0.0f, 10.0f, 9.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::Bidiagonalize(W, 4, 3, 3, QL, QR); + + // Zero below subdiagonal: W[2][0], W[3][0], W[3][1] + REQUIRE_THAT(W.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + REQUIRE_THAT(W.Get(3, 0), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + REQUIRE_THAT(W.Get(3, 1), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + + // Verify orthogonality + REQUIRE(isOrthogonal5(QL)); + REQUIRE(isOrthogonal5(QR)); + } + + // Test case 2: 3×2 matrix + // C++ verified reference: + // W[0] = [-5.916080, -7.437357] + // W[1] = [-0.000001, 0.828077] + // W[2] = [-0.000000, -0.000000] + { + Matrix<5, 5> W{1.0f, 2.0f, 0.0f, 0.0f, 0.0f, + 3.0f, 4.0f, 0.0f, 0.0f, 0.0f, + 5.0f, 6.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::Bidiagonalize(W, 3, 2, 2, QL, QR); + + // Zero below subdiagonal: W[2][0] ≈ 0 + REQUIRE_THAT(W.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + + // Verify orthogonality + REQUIRE(isOrthogonal5(QL)); + REQUIRE(isOrthogonal5(QR)); + } + + // Test case 3: 5×3 matrix (full 5-row tall) + { + Matrix<5, 5> W{1.0f, 2.0f, 3.0f, 0.0f, 0.0f, + 4.0f, 5.0f, 6.0f, 0.0f, 0.0f, + 0.0f, 7.0f, 8.0f, 0.0f, 0.0f, + 0.0f, 10.0f, 9.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::Bidiagonalize(W, 5, 3, 3, QL, QR); + + // Zero below subdiagonal: W[2][0], W[3][0], W[4][0], W[3][1], W[4][1] + REQUIRE_THAT(W.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + REQUIRE_THAT(W.Get(3, 0), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + REQUIRE_THAT(W.Get(4, 0), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + REQUIRE_THAT(W.Get(3, 1), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + REQUIRE_THAT(W.Get(4, 1), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + + // Verify orthogonality + REQUIRE(isOrthogonal5(QL)); + REQUIRE(isOrthogonal5(QR)); + } +} + +// ============================================================================ +// TEST 15: Bidiagonalize (Phase 1) — Wide matrix (m < n) +// =========================================================================== +TEST_CASE("SVD Phase 1: Bidiagonalize wide matrix", "[Matrix][SVD]") { + // Test case 1: 2×4 matrix + // C++ verified reference: + // W[0] = [-5.099020, -6.275717, 11.401754, 0.000000] + // W[1] = [ 0.000000, -0.784465, 2.806586, -0.350823] + // Note: W[1][2]=2.806586 and W[1][3]=-0.350823 are NOT zeroed because + // right HH at k=0 has only 2 elements (cols 2,3), so it zeros col 3 but preserves col 2 + { + Matrix<5, 5> W{1.0f, 2.0f, 3.0f, 4.0f, 0.0f, + 5.0f, 6.0f, 7.0f, 8.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::Bidiagonalize(W, 2, 4, 2, QL, QR); + + // For 2×4: right HH at k=0 has 2 elements (cols 2,3) + // It zeros col 3 but preserves col 2 as the superdiagonal element for row 1 + // W[0][3] should be zeroed (above superdiagonal in row 0) + REQUIRE_THAT(W.Get(0, 3), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + // W[1][2] and W[1][3] are part of the bidiagonal structure for row 1 + // (superdiagonal at col 2, and right HH preserves first element) + REQUIRE_THAT(W.Get(1, 2), !Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + + // Verify orthogonality + REQUIRE(isOrthogonal5(QL)); + REQUIRE(isOrthogonal5(QR)); + } + + // Test case 2: 3×5 matrix + { + Matrix<5, 5> W{1.0f, 2.0f, 3.0f, 4.0f, 5.0f, + 6.0f, 7.0f, 8.0f, 9.0f, 10.0f, + 0.0f, 11.0f, 12.0f, 13.0f, 14.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::Bidiagonalize(W, 3, 5, 3, QL, QR); + + // For 3×5: check that subdiagonal elements are zero + REQUIRE_THAT(W.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + REQUIRE_THAT(W.Get(2, 1), Catch::Matchers::WithinAbs(0.0f, 1e-4f)); + + // Verify orthogonality + REQUIRE(isOrthogonal5(QL)); + REQUIRE(isOrthogonal5(QR)); + } + + // Test case 3: 1×3 matrix (row vector) + // C++ verified reference: W[0] = [1.0, 2.0, 3.0] (no transformations needed) + { + Matrix<5, 5> W{1.0f, 2.0f, 3.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::Bidiagonalize(W, 1, 3, 1, QL, QR); + + // For 1×3, no transformations needed + REQUIRE_THAT(W.Get(0, 0), Catch::Matchers::WithinAbs(1.0f, 1e-4f)); + REQUIRE_THAT(W.Get(0, 1), Catch::Matchers::WithinAbs(2.0f, 1e-4f)); + REQUIRE_THAT(W.Get(0, 2), Catch::Matchers::WithinAbs(3.0f, 1e-4f)); + + // Verify orthogonality + REQUIRE(isOrthogonal5(QL)); + REQUIRE(isOrthogonal5(QR)); + } +} + +// ============================================================================ +// TEST 16: Bidiagonalize — Reconstruction property +// =========================================================================== +TEST_CASE("SVD Phase 1: Bidiagonalize reconstruction property", "[Matrix][SVD]") { + // Test: QLᵀ · W_original · QR = B (bidiagonal) + // This verifies that the accumulated transformations correctly represent + // the bidiagonalization. + { + Matrix<5, 5> W_orig{1.0f, 2.0f, 3.0f, 0.0f, 0.0f, + 4.0f, 5.0f, 6.0f, 0.0f, 0.0f, + 0.0f, 7.0f, 8.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + + Matrix<5, 5> W = W_orig; + + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::Bidiagonalize(W, 3, 3, 3, QL, QR); + + // Compute QLᵀ · W_orig · QR and verify it equals W (the bidiagonal result) + Matrix<5, 5> Qt = QL.Transpose(); + Matrix<5, 5> QtW_orig{0}; + Qt.Mult(W_orig, QtW_orig); + + Matrix<5, 5> QtW_origQR{0}; + QtW_orig.Mult(QR, QtW_origQR); + + // The reconstruction should match the bidiagonal result + float err = frobeniusNorm5(W - QtW_origQR); + REQUIRE_THAT(err, Catch::Matchers::WithinAbs(1e-3f, 1e-3f)); + } + + // Test: Tall matrix reconstruction (4×3) + { + Matrix<5, 5> W_orig{1.0f, 2.0f, 3.0f, 0.0f, 0.0f, + 4.0f, 5.0f, 6.0f, 0.0f, 0.0f, + 0.0f, 7.0f, 8.0f, 0.0f, 0.0f, + 0.0f, 10.0f, 9.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + + Matrix<5, 5> W = W_orig; + + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::Bidiagonalize(W, 4, 3, 3, QL, QR); + + Matrix<5, 5> Qt = QL.Transpose(); + Matrix<5, 5> QtW_orig{0}; + Qt.Mult(W_orig, QtW_orig); + + Matrix<5, 5> QtW_origQR{0}; + QtW_orig.Mult(QR, QtW_origQR); + + float err = frobeniusNorm5(W - QtW_origQR); + REQUIRE_THAT(err, Catch::Matchers::WithinAbs(1e-3f, 1e-3f)); + } + + // Test: Wide matrix reconstruction (2×4) + { + Matrix<5, 5> W_orig{1.0f, 2.0f, 3.0f, 4.0f, 0.0f, + 5.0f, 6.0f, 7.0f, 8.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + + Matrix<5, 5> W = W_orig; + + Matrix<5, 5> QL{0}, QR{0}; + for (uint8_t i = 0; i < 5; i++) { + QL[i][i] = 1.0f; + QR[i][i] = 1.0f; + } + + SVD::Bidiagonalize(W, 2, 4, 2, QL, QR); + + Matrix<5, 5> Qt = QL.Transpose(); + Matrix<5, 5> QtW_orig{0}; + Qt.Mult(W_orig, QtW_orig); + + Matrix<5, 5> QtW_origQR{0}; + QtW_orig.Mult(QR, QtW_origQR); + + float err = frobeniusNorm5(W - QtW_origQR); + REQUIRE_THAT(err, Catch::Matchers::WithinAbs(1e-3f, 1e-3f)); + } +} + +// ============================================================================ +// TEST: SolveBidiagonalBlock2x2 — 2×2 upper-bidiagonal block SVD +// ============================================================================ +// Reference singular values generated with scipy.linalg.svd for +// B = [[a, b], [0, d]]. +TEST_CASE("SVD Building Block: SolveBidiagonalBlock2x2", "[Matrix][SVD]") { + struct Case2x2 { + float a, b, d; + float refSigma[2]; + }; + const Case2x2 cases[] = { + {2.5f, -1.3f, 0.8f, {2.84346151f, 0.70336806f}}, + {3.0f, 0.0f, 1.0f, {3.0f, 1.0f}}, + {1.0f, 2.0f, 0.0f, {2.23606798f, 0.0f}}, + {-1.5f, 0.7f, -2.2f, {2.37779179f, 1.38784229f}}, + {1.0f, 1e-4f, 0.0f, {1.0f, 0.0f}}, + {-1.770486f, 0.281880f, 0.208573f, {1.79308863f, 0.20594385f}}, + {0.866025f, 1.0f, 0.5f, {1.37890797f, 0.31402567f}}, + }; + + for (const auto &tc : cases) { + float Ublock[2][2] = {{0}}, Vblock[2][2] = {{0}}, sigma[2] = {0}; + SVD::SolveBidiagonalBlock2x2(tc.a, tc.b, tc.d, Ublock, Vblock, sigma); + + // 1. Singular values match scipy + REQUIRE_THAT(sigma[0], + Catch::Matchers::WithinRel(tc.refSigma[0], 1e-3f)); + if (tc.refSigma[1] > 0.0f) { + REQUIRE_THAT(sigma[1], + Catch::Matchers::WithinRel(tc.refSigma[1], 1e-3f)); + } else { + REQUIRE(sigma[1] < 1e-3f); + } + REQUIRE(sigma[0] >= sigma[1]); + + // 2. Ublock and Vblock are orthogonal (MᵀM = I) + for (int i = 0; i < 2; i++) { + for (int j = i; j < 2; j++) { + float dotU = Ublock[0][i] * Ublock[0][j] + Ublock[1][i] * Ublock[1][j]; + float dotV = Vblock[0][i] * Vblock[0][j] + Vblock[1][i] * Vblock[1][j]; + float expected = (i == j) ? 1.0f : 0.0f; + REQUIRE_THAT(dotU, Catch::Matchers::WithinAbs(expected, 1e-3f)); + REQUIRE_THAT(dotV, Catch::Matchers::WithinAbs(expected, 1e-3f)); + } + } + + // 3. Ublock · diag(sigma) · Vblockᵀ reproduces B = [[a,b],[0,d]] + // (C[i][j] = sum_k U[i][k] * sigma[k] * V[j][k]) + float C[2][2] = {{0}, {0}}; + for (int i = 0; i < 2; i++) + for (int j = 0; j < 2; j++) + for (int k = 0; k < 2; k++) + C[i][j] += Ublock[i][k] * sigma[k] * Vblock[j][k]; + REQUIRE_THAT(C[0][0], Catch::Matchers::WithinAbs(tc.a, 1e-2f)); + REQUIRE_THAT(C[0][1], Catch::Matchers::WithinAbs(tc.b, 1e-2f)); + REQUIRE_THAT(C[1][0], Catch::Matchers::WithinAbs(0.0f, 1e-2f)); + REQUIRE_THAT(C[1][1], Catch::Matchers::WithinAbs(tc.d, 1e-2f)); + } +} + +// ============================================================================ +// TEST: JacobiEigenSymmetric — cyclic Jacobi eigenvalue decomposition +// ============================================================================ +// Reference eigenvalues generated with scipy.linalg.eigvalsh (desc). +TEST_CASE("SVD Building Block: JacobiEigenSymmetric", "[Matrix][SVD]") { + struct CaseJac { + float S[5][5]; + uint8_t n; + float refEig[5]; + }; + + // (i) T = BᵀB from a real bidiagonalization (3×3) + float T3[5][5] = { + {65.999993f, -124.470864f, 0.0f, 0.0f, 0.0f}, + {-124.470864f, 237.877008f, -0.499065f, 0.0f, 0.0f}, + {0.0f, -0.499065f, 0.122959f, 0.0f, 0.0f}, + {0.0f, 0.0f, 0.0f, 0.0f, 0.0f}, + {0.0f, 0.0f, 0.0f, 0.0f, 0.0f}, + }; + // (ii) random-looking 3×3 symmetric (seed 42) + float S3[5][5] = { + {0.304717f, -0.04971f, 0.439146f, 0.0f, 0.0f}, + {-0.04971f, -1.951035f, -0.809211f, 0.0f, 0.0f}, + {0.439146f, -0.809211f, -0.016801f, 0.0f, 0.0f}, + {0.0f, 0.0f, 0.0f, 0.0f, 0.0f}, + {0.0f, 0.0f, 0.0f, 0.0f, 0.0f}, + }; + // (iii) random-looking 4×4 symmetric (seed 42) + float S4[5][5] = { + {-0.853044f, 1.00332f, -0.090545f, -0.307449f, 0.0f}, + {1.00332f, 0.467509f, 0.009579f, 0.795646f, 0.0f}, + {-0.090545f, 0.009579f, -0.049926f, -0.169696f, 0.0f}, + {-0.307449f, 0.795646f, -0.169696f, -0.428328f, 0.0f}, + {0.0f, 0.0f, 0.0f, 0.0f, 0.0f}, + }; + + float refs[3][5] = { + {303.195295f, 0.765908223f, 0.0387564408f, 0, 0}, + {0.7227162f, -0.13661881f, -2.24921639f, 0, 0}, + {1.22127596f, -0.01555681f, -0.31307273f, -1.75643542f, 0}, + }; + uint8_t ns[3] = {3, 3, 4}; + float (*mats[3])[5] = {T3, S3, S4}; + float maxAbs[3] = {237.877008f, 1.951035f, 1.00332f}; + + for (int c = 0; c < 3; c++) { + float T[5][5]; + for (int i = 0; i < 5; i++) + for (int j = 0; j < 5; j++) + T[i][j] = mats[c][i][j]; + float S_orig[5][5]; + for (int i = 0; i < 5; i++) + for (int j = 0; j < 5; j++) + S_orig[i][j] = mats[c][i][j]; + + float evals[5] = {0}; + // JacobiEigenSymmetric operates on Matrix — copy the raw test + // data in, run the solver, copy the eigenvector matrix back out. + Matrix<5, 5> Tm{0}; + for (int i = 0; i < 5; i++) + for (int j = 0; j < 5; j++) + Tm[i][j] = T[i][j]; + Matrix<5, 5> Vm{0}; + SVD::JacobiEigenSymmetric(Tm, ns[c], evals, Vm); + float V[5][5] = {{0}}; + for (int i = 0; i < 5; i++) + for (int j = 0; j < 5; j++) + V[i][j] = Vm[i][j]; + + // 1. Sorted eigenvalues match scipy + float sorted[5] = {0}; + for (int i = 0; i < ns[c]; i++) sorted[i] = evals[i]; + // Sort descending to match the scipy reference order + for (int i = 0; i < ns[c] - 1; i++) { + int maxIdx = i; + for (int j = i + 1; j < ns[c]; j++) + if (sorted[j] > sorted[maxIdx]) + maxIdx = j; + if (maxIdx != i) { + float t = sorted[i]; + sorted[i] = sorted[maxIdx]; + sorted[maxIdx] = t; + } + } + for (int i = 0; i < ns[c]; i++) { + if (fabsf(refs[c][i]) > 0.01f) { + REQUIRE_THAT(sorted[i], + Catch::Matchers::WithinRel(refs[c][i], 1e-3f)); + } else { + REQUIRE_THAT(sorted[i], Catch::Matchers::WithinAbs(refs[c][i], 1e-3f)); + } + } + + // 2. V is orthogonal (VᵀV = I on the n×n part) + for (int i = 0; i < ns[c]; i++) { + for (int j = i; j < ns[c]; j++) { + float dot = 0.0f; + for (int k = 0; k < ns[c]; k++) dot += V[k][i] * V[k][j]; + float expected = (i == j) ? 1.0f : 0.0f; + REQUIRE_THAT(dot, Catch::Matchers::WithinAbs(expected, 1e-3f)); + } + } + + // 3. Residual ‖S_orig·V − V·diag(evals)‖ small + // (col i of S_orig·V must equal evals_i · col i of V) + float residual = 0.0f; + for (int i = 0; i < ns[c]; i++) { + for (int r = 0; r < ns[c]; r++) { + float Sv = 0.0f; + for (int k = 0; k < ns[c]; k++) Sv += S_orig[r][k] * V[k][i]; + float diff = Sv - evals[i] * V[r][i]; + residual += diff * diff; + } + } + residual = sqrtf(residual); + REQUIRE_THAT(residual, + Catch::Matchers::WithinAbs(0.0f, + 1e-2f * maxAbs[c])); + } +} + +// ============================================================================ +// TEST: DeflateBidiagonal / BidiagonalIsDiagonal +// ============================================================================ +TEST_CASE("SVD Building Block: DeflateBidiagonal and BidiagonalIsDiagonal", + "[Matrix][SVD]") { + float tol = 1e-8f; + + // IsDiagonal: true on a diagonal matrix + { + Matrix<5, 5> W{10.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 5.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 2.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + REQUIRE(SVD::BidiagonalIsDiagonal(W, 5, tol)); + } + + // IsDiagonal: false when a superdiagonal is significant + { + Matrix<5, 5> W{10.0f, 1e-3f, 0.0f, 0.0f, 0.0f, + 0.0f, 5.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 2.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; + REQUIRE_FALSE(SVD::BidiagonalIsDiagonal(W, 5, tol)); + } + + // Deflate: small superdiagonals zeroed, significant ones kept + { + Matrix<5, 5> W{1.0f, 0.5f, 0.0f, 0.0f, 0.0f, + 0.0f, 2.0f, 1e-9f, 0.0f, 0.0f, + 0.0f, 0.0f, 3.0f, 0.3f, 0.0f, + 0.0f, 0.0f, 0.0f, 4.0f, 1e-12f, + 0.0f, 0.0f, 0.0f, 0.0f, 5.0f}; + SVD::DeflateBidiagonal(W, 5, tol); + REQUIRE_THAT(W.Get(0, 1), Catch::Matchers::WithinAbs(0.5f, 1e-6f)); + REQUIRE(W.Get(1, 2) == 0.0f); + REQUIRE_THAT(W.Get(2, 3), Catch::Matchers::WithinAbs(0.3f, 1e-6f)); + REQUIRE(W.Get(3, 4) == 0.0f); + // Diagonal untouched + REQUIRE_THAT(W.Get(0, 0), Catch::Matchers::WithinAbs(1.0f, 1e-6f)); + REQUIRE_THAT(W.Get(4, 4), Catch::Matchers::WithinAbs(5.0f, 1e-6f)); + // NOT fully diagonal: significant superdiagonals (0.5, 0.3) remain + REQUIRE_FALSE(SVD::BidiagonalIsDiagonal(W, 5, tol)); + } + + // Deflate on an already-diagonal-ish matrix makes IsDiagonal true + { + Matrix<5, 5> W{1.0f, 1e-9f, 0.0f, 0.0f, 0.0f, + 0.0f, 2.0f, 1e-11f, 0.0f, 0.0f, + 0.0f, 0.0f, 3.0f, 0.0f, 0.0f, + 0.0f, 0.0f, 0.0f, 4.0f, 1e-10f, + 0.0f, 0.0f, 0.0f, 0.0f, 5.0f}; + SVD::DeflateBidiagonal(W, 5, tol); + REQUIRE(SVD::BidiagonalIsDiagonal(W, 5, tol)); + } +} diff --git a/unit-tests/svd-integration-test.cpp b/unit-tests/svd-integration-test.cpp new file mode 100644 index 0000000..48559e3 --- /dev/null +++ b/unit-tests/svd-integration-test.cpp @@ -0,0 +1,363 @@ +#include "Matrix.hpp" +#include "SVD.hpp" +#include +#include +#include + +// Generic helper functions for any matrix size +template +static float frobeniusNorm(const Matrix &M) { + float sum = 0.0f; + for (int i = 0; i < rows; i++) + for (int j = 0; j < columns; j++) { + float v = M.Get(i, j); + sum += v * v; + } + return sqrtf(sum); +} + +template +static bool isOrthogonal(const Matrix &M, float tol = 1e-4f) { + Matrix Mt = M.Transpose(); + Matrix MtM{0}; + Mt.Mult(M, MtM); + for (int i = 0; i < n; i++) + for (int j = 0; j < n; j++) { + float expected = (i == j) ? 1.0f : 0.0f; + if (fabsf(MtM.Get(i, j) - expected) > tol) + return false; + } + return true; +} + +TEST_CASE("SVD Integration: 2x2 [[1,2],[3,4]]", "[Matrix][SVD][Integration]") { + Matrix<2, 2> A{1, 2, 3, 4}; + Matrix<2, 2> U{0}; + Matrix<2, 1> sigma{0}; + Matrix<2, 2> Vt{0}; + + SVD::SVD(A, U, sigma, Vt); + + // Reference singular values from scipy: [5.464985704219, 0.365966190626] + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(5.4649857f, 1e-3f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(0.3659662f, 1e-3f)); + + // Check orthogonality of U and Vt (first 2x2 blocks) + REQUIRE(isOrthogonal<2>(U)); + REQUIRE(isOrthogonal<2>(Vt)); + + // Check reconstruction: A ≈ U · diag(sigma) · Vt + Matrix<2, 2> recon{0}; + Matrix<2, 2> Usig{0}; + for (int i = 0; i < 2; i++) + for (int j = 0; j < 2; j++) + Usig[i][j] = U.Get(i, j) * sigma.Get(j, 0); + + Usig.Mult(Vt, recon); + + float err = 0.0f; + for (int i = 0; i < 2; i++) + for (int j = 0; j < 2; j++) { + float diff = recon.Get(i, j) - A.Get(i, j); + err += diff * diff; + } + err = sqrtf(err); + REQUIRE_THAT(err, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + + std::cout << "SVD 2x2 [[1,2],[3,4]]:\n"; + std::cout << "Sigma: [" << sigma.Get(0, 0) << ", " << sigma.Get(1, 0) + << "]\n"; +} + +TEST_CASE("SVD Integration: 3x3 diagonal [10,5,2]", + "[Matrix][SVD][Integration]") { + Matrix<3, 3> A{10, 0, 0, 0, 5, 0, 0, 0, 2}; + Matrix<3, 3> U{0}; + Matrix<3, 1> sigma{0}; + Matrix<3, 3> Vt{0}; + + SVD::SVD(A, U, sigma, Vt); + + // Singular values should be [10, 5, 2] (already diagonal) + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(10.0f, 1e-3f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(5.0f, 1e-3f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(2.0f, 1e-3f)); + + // U and Vt should be identity (or close) for diagonal matrix + float uErr = frobeniusNorm(U - Matrix<3, 3>{1, 0, 0, 0, 1, 0, 0, 0, 1}); + float vtErr = frobeniusNorm(Vt - Matrix<3, 3>{1, 0, 0, 0, 1, 0, 0, 0, 1}); + REQUIRE_THAT(uErr, Catch::Matchers::WithinAbs(0.0f, 1e-2f)); + REQUIRE_THAT(vtErr, Catch::Matchers::WithinAbs(0.0f, 1e-2f)); +} + +TEST_CASE("SVD Integration: 3x3 rank-deficient [[1,2,3],[4,5,6],[7,8,9]]", + "[Matrix][SVD][Integration]") { + Matrix<3, 3> A{1, 2, 3, 4, 5, 6, 7, 8, 9}; + Matrix<3, 3> U{0}; + Matrix<3, 1> sigma{0}; + Matrix<3, 3> Vt{0}; + + SVD::SVD(A, U, sigma, Vt); + + // Reference: [16.848103352614, 1.068369514555, 0.0] + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(16.8481f, 1e-2f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(1.06837f, 1e-2f)); + // Third singular value should be ~0 (rank-deficient) + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-2f)); + + // Check reconstruction + Matrix<3, 3> recon{0}; + Matrix<3, 3> Usig{0}; + for (int i = 0; i < 3; i++) + for (int j = 0; j < 3; j++) + Usig[i][j] = U.Get(i, j) * sigma.Get(j, 0); + Usig.Mult(Vt, recon); + + float err = 0.0f; + for (int i = 0; i < 3; i++) + for (int j = 0; j < 3; j++) { + float diff = recon.Get(i, j) - A.Get(i, j); + err += diff * diff; + } + err = sqrtf(err); + REQUIRE_THAT(err, Catch::Matchers::WithinAbs(0.0f, 1e-2f)); + + std::cout << "SVD 3x3 rank-deficient:\n"; + std::cout << "Sigma: [" << sigma.Get(0, 0) << ", " << sigma.Get(1, 0) << ", " + << sigma.Get(2, 0) << "]\n"; +} + +TEST_CASE("SVD Integration: tall 4x3 matrix", "[Matrix][SVD][Integration]") { + Matrix<4, 3> A{1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12}; + Matrix<4, 3> U{0}; + Matrix<3, 1> sigma{0}; + Matrix<3, 3> Vt{0}; + + SVD::SVD(A, U, sigma, Vt); + + // Reference: [25.462407436036, 1.290661675761, 0.0] + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(25.4624f, 1e-2f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(1.29066f, 1e-2f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-2f)); + + // Check reconstruction + Matrix<4, 3> recon{0}; + Matrix<4, 3> Usig{0}; + for (int i = 0; i < 4; i++) + for (int j = 0; j < 3; j++) + Usig[i][j] = U.Get(i, j) * sigma.Get(j, 0); + Usig.Mult(Vt, recon); + + float err = 0.0f; + for (int i = 0; i < 4; i++) + for (int j = 0; j < 3; j++) { + float diff = recon.Get(i, j) - A.Get(i, j); + err += diff * diff; + } + err = sqrtf(err); + REQUIRE_THAT(err, Catch::Matchers::WithinAbs(0.0f, 1e-2f)); + + std::cout << "SVD tall 4x3:\n"; + std::cout << "Sigma: [" << sigma.Get(0, 0) << ", " << sigma.Get(1, 0) << ", " + << sigma.Get(2, 0) << "]\n"; +} + +TEST_CASE("SVD Integration: wide 3x5 matrix", "[Matrix][SVD][Integration]") { + Matrix<3, 5> A{1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15}; + Matrix<3, 5> U{0}; + Matrix<5, 1> sigma{0}; // sigma is columns x 1 = 5x1 for wide matrix + Matrix<5, 5> Vt{0}; // Vt is columns x columns = 5x5 + + SVD::SVD(A, U, sigma, Vt); + + // Reference: [35.127223333575, 2.465396696917, 0.0] + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(35.1272f, 1e-2f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(2.46540f, 1e-2f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-2f)); + + // Check reconstruction: A (3x5) = U * Sigma * Vt, where U (3x5) has + // its meaningful part in the first 3 columns, sigma (5x1) in the + // first 3 entries, and Vt (5x5) in its first 3 rows (right + // singular vectors as rows). So: + // A[i][j] = sum_k U[i][k] * sigma[k] * Vt[k][j] + + float err2 = 0.0f; + for (int i = 0; i < 3; i++) { + for (int j = 0; j < 5; j++) { + float recon_val = 0.0f; + for (int k = 0; k < 3; k++) { + recon_val += U.Get(i, k) * sigma.Get(k, 0) * Vt.Get(k, j); + } + float diff = recon_val - A.Get(i, j); + err2 += diff * diff; + } + } + err2 = sqrtf(err2); + REQUIRE_THAT(err2, Catch::Matchers::WithinAbs(0.0f, 1e-2f)); + + std::cout << "SVD wide 3x5:\n"; + std::cout << "Sigma: [" << sigma.Get(0, 0) << ", " << sigma.Get(1, 0) << ", " + << sigma.Get(2, 0) << "]\n"; +} + +TEST_CASE("SVD Integration: identity 3x3", "[Matrix][SVD][Integration]") { + Matrix<3, 3> A{1, 0, 0, 0, 1, 0, 0, 0, 1}; + Matrix<3, 3> U{0}; + Matrix<3, 1> sigma{0}; + Matrix<3, 3> Vt{0}; + + SVD::SVD(A, U, sigma, Vt); + + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(1.0f, 1e-3f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(1.0f, 1e-3f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinRel(1.0f, 1e-3f)); + + float err = frobeniusNorm(U - Matrix<3, 3>{1, 0, 0, 0, 1, 0, 0, 0, 1}); + REQUIRE_THAT(err, Catch::Matchers::WithinAbs(0.0f, 1e-2f)); +} + +TEST_CASE("SVD Integration: symmetric positive definite 2x2 [[5,3],[3,5]]", + "[Matrix][SVD][Integration]") { + Matrix<2, 2> A{5, 3, 3, 5}; + Matrix<2, 2> U{0}; + Matrix<2, 1> sigma{0}; + Matrix<2, 2> Vt{0}; + + SVD::SVD(A, U, sigma, Vt); + + // For SPD matrix, singular values = eigenvalues: [8, 2] + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(8.0f, 1e-3f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(2.0f, 1e-3f)); + + // Check reconstruction + Matrix<2, 2> recon{0}; + Matrix<2, 2> Usig{0}; + for (int i = 0; i < 2; i++) + for (int j = 0; j < 2; j++) + Usig[i][j] = U.Get(i, j) * sigma.Get(j, 0); + Usig.Mult(Vt, recon); + + float err = 0.0f; + for (int i = 0; i < 2; i++) + for (int j = 0; j < 2; j++) { + float diff = recon.Get(i, j) - A.Get(i, j); + err += diff * diff; + } + err = sqrtf(err); + REQUIRE_THAT(err, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); + + std::cout << "SVD SPD 2x2 [[5,3],[3,5]]:\n"; + std::cout << "Sigma: [" << sigma.Get(0, 0) << ", " << sigma.Get(1, 0) + << "]\n"; +} + +// ---------------------------------------------------------------------------- +// Matrix::SVD member wrapper (delegates to SVD::SVD) +// ---------------------------------------------------------------------------- + +/** + * Reconstruction error ‖U·diag(sigma)·Vᵀ − A‖_F. Zero-padded entries of + * U/sigma/Vt (wide/tall cases) are zero by the output conventions, so the + * full product equals U[:, :k]·diag(sigma[:k])·Vt[:k, :]. + */ +template +static float svdReconstructionError(const Matrix &A, + const Matrix &U, + const Matrix &sigma, + const Matrix &Vt) { + Matrix recon{0}; + Matrix Usig{0}; + for (int i = 0; i < rows; i++) + for (int j = 0; j < columns; j++) + Usig[i][j] = U.Get(i, j) * sigma.Get(j, 0); + Usig.Mult(Vt, recon); + float err = 0.0f; + for (int i = 0; i < rows; i++) + for (int j = 0; j < columns; j++) { + float diff = recon.Get(i, j) - A.Get(i, j); + err += diff * diff; + } + return sqrtf(err); +} + +/** + * Orthonormality of the first k columns of M: the k×k leading block of + * MᵀM must equal I_k. (For a tall SVD, U has k = min(rows, cols) + * meaningful columns and this is the full UᵀU.) + */ +template +static bool leadingColumnsOrthonormal(const Matrix &M, uint8_t k, + float tol = 1e-4f) { + Matrix Mt = M.Transpose(); + Matrix MtM{0}; + Mt.Mult(M, MtM); + for (int i = 0; i < k; i++) + for (int j = 0; j < k; j++) { + float expected = (i == j) ? 1.0f : 0.0f; + if (fabsf(MtM.Get(i, j) - expected) > tol) + return false; + } + return true; +} + +/** + * Orthonormality of the first k rows of M: the k×k leading block of + * M·Mᵀ must equal I_k. (Vᵀ may have zero-padded trailing rows in the + * wide case, so check only the meaningful leading block.) + */ +template +static bool leadingRowsOrthonormal(const Matrix &M, uint8_t k, + float tol = 1e-4f) { + Matrix Mt = M.Transpose(); + Matrix MMt{0}; + M.Mult(Mt, MMt); + for (int i = 0; i < k; i++) + for (int j = 0; j < k; j++) { + float expected = (i == j) ? 1.0f : 0.0f; + if (fabsf(MMt.Get(i, j) - expected) > tol) + return false; + } + return true; +} + +TEST_CASE("Matrix::SVD wrapper: 3x2 tall [[1,2],[3,4],[5,6]]", + "[Matrix][SVD][Wrapper]") { + Matrix<3, 2> A{1, 2, 3, 4, 5, 6}; + Matrix<3, 2> U{0}; + Matrix<2, 1> sigma{0}; + Matrix<2, 2> Vt{0}; + + A.SVD(U, sigma, Vt); + + // Reference singular values from numpy: [9.52552, 0.514301] + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(9.52552f, 1e-3f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(0.514301f, 1e-3f)); + + REQUIRE(leadingColumnsOrthonormal(U, 2)); + REQUIRE(leadingRowsOrthonormal(Vt, 2)); + + float err = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(err, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); +} + +TEST_CASE("Matrix::SVD wrapper: 2x3 wide [[1,2,3],[4,5,6]]", + "[Matrix][SVD][Wrapper]") { + Matrix<2, 3> A{1, 2, 3, 4, 5, 6}; + Matrix<2, 3> U{0}; + Matrix<3, 1> sigma{0}; + Matrix<3, 3> Vt{0}; + + A.SVD(U, sigma, Vt); + + // Reference singular values from numpy: [9.50803, 0.77287]; the third + // entry (wide-matrix padding) must be zero. + REQUIRE_THAT(sigma.Get(0, 0), Catch::Matchers::WithinRel(9.50803f, 1e-3f)); + REQUIRE_THAT(sigma.Get(1, 0), Catch::Matchers::WithinRel(0.77287f, 1e-3f)); + REQUIRE_THAT(sigma.Get(2, 0), Catch::Matchers::WithinAbs(0.0f, 1e-6f)); + + REQUIRE(leadingColumnsOrthonormal(U, 2)); + REQUIRE(leadingRowsOrthonormal(Vt, 2)); + + float err = svdReconstructionError(A, U, sigma, Vt); + REQUIRE_THAT(err, Catch::Matchers::WithinAbs(0.0f, 1e-3f)); +} diff --git a/unit-tests/svd-reference-values.py b/unit-tests/svd-reference-values.py new file mode 100644 index 0000000..d9e8504 --- /dev/null +++ b/unit-tests/svd-reference-values.py @@ -0,0 +1,513 @@ +#!/usr/bin/env python3 +""" +Generate reference values for SVD building block unit tests. +Run this to verify/implement the C++ SVD implementation against scipy/numpy. + +Usage: python3 svd-reference-values.py +""" + +import numpy as np +from scipy.linalg import svd, qr as scipy_qr +import json + +def compute_householder(x): + """Compute Householder reflector: H*x = [alpha, 0, 0, ...]^T. + + Returns (v_normalized, alpha) where v is the normalized Householder vector. + H = I - 2*v*v^T / (v^T*v) + """ + x = np.array(x, dtype=np.float64) + norm_x = np.linalg.norm(x) + + if norm_x < 1e-30: + return x.copy(), 0.0 + + alpha = -np.sign(x[0]) * norm_x if x[0] != 0 else -norm_x + v = x.copy() + v[0] -= alpha + v_norm = np.linalg.norm(v) + + if v_norm < 1e-30: + return np.zeros_like(x), alpha + + v /= v_norm + return v, alpha + +def apply_householder_left(A, v, start_row): + """Apply Householder reflection from the left: A = (I - 2vv^T) @ A. + + v is the normalized Householder vector operating on rows [start_row:]. + The length of v must match the number of rows affected. + """ + A = A.copy() + k = len(v) + + for col in range(A.shape[1]): + dot = np.dot(v, A[start_row:start_row+k, col]) + A[start_row:start_row+k, col] -= 2.0 * dot * v + + return A + +def apply_householder_right(A, v, start_col): + """Apply Householder reflection from the right: A = A @ (I - 2vv^T). + + v is the normalized Householder vector operating on columns [start_col:]. + The length of v must match the number of columns affected. + """ + A = A.copy() + k = len(v) + + for row in range(A.shape[0]): + dot = np.dot(A[row, start_col:start_col+k], v) + A[row, start_col:start_col+k] -= 2.0 * dot * v + + return A + +def compute_givens(x, y): + """Compute Givens rotation that zeros out y. + + Returns (c, s) such that [c s; -s c] @ [x; y] = [r; 0]. + """ + r = np.sqrt(x*x + y*y) + if r < 1e-30: + return 1.0, 0.0 + c = x / r + s = y / r + return c, s + +def apply_givens_left(A, i, j, c, s): + """Apply Givens rotation from the left to rows i and j of A. + + [c s] [row_i] + [-s c] @ [row_j] = [new_row_i] + [new_row_j] + """ + A = A.copy() + new_i = c * A[i] + s * A[j] + new_j = -s * A[i] + c * A[j] + A[i] = new_i + A[j] = new_j + return A + +def apply_givens_right(A, i, j, c, s): + """Apply Givens rotation from the right to columns i and j of A. + + [col_i col_j] @ [c -s] = [new_col_i new_col_j] + [s c] + """ + A = A.copy() + new_i = c * A[:, i] + s * A[:, j] + new_j = -s * A[:, i] + c * A[:, j] + A[:, i] = new_i + A[:, j] = new_j + return A + +def householder_bidiagonalization(A): + """Full Householder bidiagonalization: A = Q_L @ B @ Q_R^T. + + Returns (B, Q_L, Q_R) where B is upper bidiagonal. + """ + m, n = A.shape + p = min(m, n) + + QL = np.eye(m, dtype=np.float64) + QR = np.eye(n, dtype=np.float64) + W = A.copy() + + for k in range(p): + # Left HH: zero out W[k+1:, k] + if k < m - 1: + x = W[k+1:, k].copy() + v, alpha = compute_householder(x) + if np.linalg.norm(v) > 1e-30: + W = apply_householder_left(W, v, k + 1) + QL = apply_householder_right(QL, v, k + 1) + + # Right HH: zero out W[k, k+2:] (superdiagonal) + if k < p - 1 and k + 2 <= n: + x = W[k, k+2:].copy() + v, alpha = compute_householder(x) + if np.linalg.norm(v) > 1e-30: + W = apply_householder_right(W, v, k + 2) + QR = apply_householder_right(QR, v, k + 2) + + return W, QL, QR + +def implicit_qr_iteration(B, QR_acc): + """Implicit QR iteration on a bidiagonal matrix. + + Returns (Sigma, QR_acc) where Sigma is diagonal with singular values + and QR_acc contains the accumulated right transformations. + """ + m, n = B.shape + p = min(m, n) + W = B.copy() + + max_iter = 1000 + tol = 1e-10 + + for iteration in range(max_iter): + # Deflate negligible subdiagonal elements + for i in range(p - 1, 0, -1): + if abs(W[i, i-1]) < tol * (abs(W[i-1, i-1]) + abs(W[i, i])): + W[i, i-1] = 0.0 + + # Find smallest unreduced block [start, end] + start = 0 + for i in range(p - 1): + if abs(W[i+1, i]) >= tol * (abs(W[i, i]) + abs(W[i+1, i+1])): + start = i + 1 + + end = p - 1 + for i in range(p - 2, -1, -1): + if abs(W[i+1, i]) >= tol * (abs(W[i, i]) + abs(W[i+1, i+1])): + end = i + break + + if start >= end: + continue + + # Wilkinson shift from bottom 2x2 corner + a, b = W[end-1, end-1], W[end-1, end] + c_val, d = W[end, end-1], W[end, end] + trace = a + d + det = a * d - b * c_val + disc = trace**2 - 4 * det + + if disc >= 0: + sqrt_disc = np.sqrt(disc) + e1, e2 = (trace + sqrt_disc) / 2, (trace - sqrt_disc) / 2 + shift = e1 if abs(e1 - d) < abs(e2 - d) else e2 + else: + shift = d + + # Implicit QR step using Givens rotations + # Process from top to bottom within the block + x = W[start, start] - shift + y = W[start + 1, start] + + for i in range(start, end): + r = np.sqrt(x*x + y*y) + if r < 1e-30: + x = W[i + 1, i] + y = W[i + 1, i + 1] if i + 2 <= end else 0.0 + continue + + c_rot = x / r + s_rot = y / r + + # Apply from left to rows i, i+1 (columns i..n-1) + for j in range(i, n): + t1, t2 = W[i, j], W[i + 1, j] + W[i, j] = c_rot * t1 + s_rot * t2 + W[i + 1, j] = -s_rot * t1 + c_rot * t2 + + # Apply from right to columns i, i+1 (rows 0..i) + if i > start: + for j in range(i + 1): + t1, t2 = W[j, i], W[j, i + 1] + W[j, i] = c_rot * t1 + s_rot * t2 + W[j, i + 1] = -s_rot * t1 + c_rot * t2 + + # Accumulate into QR_acc + for j in range(QR_acc.shape[0]): + t1, t2 = QR_acc[j, i], QR_acc[j, i + 1] + QR_acc[j, i] = c_rot * t1 + s_rot * t2 + QR_acc[j, i + 1] = -s_rot * t1 + c_rot * t2 + + # Prepare for next rotation + x = W[i + 1, i] + y = W[i + 1, i + 1] if i + 2 <= end else 0.0 + + return W, QR_acc + + +def main(): + print("=" * 70) + print("SVB BUILDING BLOCK REFERENCE VALUES") + print("Generated with scipy/numpy for C++ unit test verification") + print("=" * 70) + + # ------------------------------------------------------------------ + # Test 1: Householder Vector Computation + # ------------------------------------------------------------------ + print("\n" + "=" * 70) + print("TEST 1: computeHouseholderVector") + print("=" * 70) + + test_vectors = [ + ("2D [1,3]", [1.0, 3.0]), + ("2D [3,4] (norm=5)", [3.0, 4.0]), + ("3D [1,2,3]", [1.0, 2.0, 3.0]), + ("3D [0,0,1]", [0.0, 0.0, 1.0]), + ("4D [5,-3,2,1]", [5.0, -3.0, 2.0, 1.0]), + ] + + for name, vec in test_vectors: + v, alpha = compute_householder(vec) + x = np.array(vec) + Hx = x - 2 * np.dot(v, x) * v + + print(f"\n{name}:") + print(f" Input: {list(x)}") + print(f" ||x||: {np.linalg.norm(x):.15f}") + print(f" alpha: {alpha:.15f}") + print(f" v (normalized): {[round(float(vi), 12) for vi in v]}") + print(f" H*x = [alpha,0..]: {[round(float(xi), 12) for xi in Hx]}") + print(f" Off-diagonal ~0: {np.allclose(Hx[1:], 0, atol=1e-12)}") + + # ------------------------------------------------------------------ + # Test 2: Householder Apply Left + # ------------------------------------------------------------------ + print("\n" + "=" * 70) + print("TEST 2: applyHouseholderLeft") + print("=" * 70) + + A_test = np.array([[1.0, 2.0, 3.0], [4.0, 5.0, 6.0], [7.0, 8.0, 9.0]], dtype=np.float64) + x_col = A_test[1:, 0].copy() + v_left, _ = compute_householder(x_col) + + print(f"\nInput matrix:\n{A_test}") + print(f"Householder vector (rows 1:3): {[round(float(vi), 12) for vi in v_left]}") + + A_result = apply_householder_left(A_test, v_left, 1) + print(f"\nAfter applyHouseholderLeft:\n{A_result}") + print(f" A[1,0] = {A_result[1,0]:.2e}, A[2,0] = {A_result[2,0]:.2e} (should be ~0)") + + # ------------------------------------------------------------------ + # Test 3: Householder Apply Right + # ------------------------------------------------------------------ + print("\n" + "=" * 70) + print("TEST 3: applyHouseholderRight") + print("=" * 70) + + A_test = np.array([[1.0, 2.0, 3.0], [4.0, 5.0, 6.0], [7.0, 8.0, 9.0]], dtype=np.float64) + x_row = A_test[0, 1:].copy() + v_right, _ = compute_householder(x_row) + + print(f"\nInput matrix:\n{A_test}") + print(f"Householder vector (cols 1:3): {[round(float(vi), 12) for vi in v_right]}") + + A_result = apply_householder_right(A_test, v_right, 1) + print(f"\nAfter applyHouseholderRight:\n{A_result}") + print(f" A[0,1] = {A_result[0,1]:.2e}, A[0,2] = {A_result[0,2]:.2e} (should be ~0)") + + # ------------------------------------------------------------------ + # Test 4: Givens Rotation Computation + # ------------------------------------------------------------------ + print("\n" + "=" * 70) + print("TEST 4: computeGivens") + print("=" * 70) + + givens_tests = [ + ("3-4-5 triangle", 3.0, 4.0), + ("y already zero", 1.0, 0.0), + ("x is zero", 0.0, 5.0), + ("Both negative", -3.0, -4.0), + ("45 degree case", 1.0, -1.0), + ] + + for name, x, y in givens_tests: + c, s = compute_givens(x, y) + result_x = c * x + s * y + result_y = -s * x + c * y + + print(f"\n{name}: x={x}, y={y}") + print(f" r = {np.sqrt(x*x+y*y):.12f}") + print(f" c = {c:.12f}, s = {s:.12f}") + print(f" [c s; -s c] @ [x;y] = [{result_x:.2e}, {result_y:.2e}]") + + # ------------------------------------------------------------------ + # Test 5: Apply Givens Left/Right + # ------------------------------------------------------------------ + print("\n" + "=" * 70) + print("TEST 5: applyGivensLeft / applyGivensRight") + print("=" * 70) + + A_test = np.array([[3.0, 4.0], [1.0, 2.0]], dtype=np.float64) + c, s = compute_givens(3.0, 1.0) + + print(f"\nInput matrix:\n{A_test}") + print(f"Givens rotation (rows 0,1): c={c:.12f}, s={s:.12f}") + + A_left = apply_givens_left(A_test, 0, 1, c, s) + print(f"\nAfter applyGivensLeft:\n{A_left}") + print(f" A[1,0] = {A_left[1,0]:.2e} (should be ~0)") + + A_test = np.array([[3.0, 1.0], [4.0, 2.0]], dtype=np.float64) + c, s = compute_givens(3.0, 4.0) + + print(f"\nInput matrix:\n{A_test}") + print(f"Givens rotation (cols 0,1): c={c:.12f}, s={s:.12f}") + + A_right = apply_givens_right(A_test, 0, 1, c, s) + print(f"\nAfter applyGivensRight:\n{A_right}") + print(f" A[0,1] = {A_right[0,1]:.2e} (should be ~0)") + + # ------------------------------------------------------------------ + # Test 6: Full Bidiagonalization + # ------------------------------------------------------------------ + print("\n" + "=" * 70) + print("TEST 6: householderBidiagonalization") + print("=" * 70) + + bidiag_tests = [ + ("2x2 [[1,2],[3,4]]", np.array([[1.0, 2.0], [3.0, 4.0]])), + ("3x3 SPD [[5,3],[3,5]]", np.array([[5.0, 3.0], [3.0, 5.0]])), + ("3x3 diag [[10,0,0],[0,5,0],[0,0,2]]", + np.array([[10.0, 0, 0], [0, 5.0, 0], [0, 0, 2.0]])), + ("3x3 full [[1,2,3],[4,5,6],[7,8,10]]", + np.array([[1.0, 2.0, 3.0], [4.0, 5.0, 6.0], [7.0, 8.0, 10.0]])), + ("Tall 4x3", np.array([[1,2,3],[4,5,6],[7,8,9],[10,11,12]], dtype=np.float64)), + ] + + for name, A in bidiag_tests: + B, QL, QR = householder_bidiagonalization(A) + m, n = A.shape + p = min(m, n) + + print(f"\n{name}:") + print(f" Original:\n{A}") + print(f"\n Bidiagonal B:\n{B}") + print(f" Diagonal: {[round(float(B[i,i]), 10) for i in range(p)]}") + print(f" Superdiag: {[round(float(B[i,i+1]), 10) for i in range(min(p-1, n-1))]}") + + recon = QL @ B @ QR.T + err = np.linalg.norm(recon - A, 'fro') + print(f" ||QL @ B @ QR^T - A||_F = {err:.2e}") + + # ------------------------------------------------------------------ + # Test 7: Full SVD Reference Values + # ------------------------------------------------------------------ + print("\n" + "=" * 70) + print("TEST 7: Full SVD Reference Values (scipy.linalg.svd)") + print("=" * 70) + + test_matrices = [ + ("Simple 2x2", np.array([[1,2],[3,4]], dtype=np.float64)), + ("SPD 2x2", np.array([[5,3],[3,5]], dtype=np.float64)), + ("Full-rank 3x3", np.array([[1,2,3],[4,5,6],[7,8,10]], dtype=np.float64)), + ("Rank-deficient 3x3", np.array([[1,2,3],[4,5,6],[7,8,9]], dtype=np.float64)), + ("Diagonal 3x3", np.array([[10,0,0],[0,5,0],[0,0,2]], dtype=np.float64)), + ("Tall 4x3", np.array([[1,2,3],[4,5,6],[7,8,9],[10,11,12]], dtype=np.float64)), + ("Wide 3x5", np.array([[1,2,3,4,5],[6,7,8,9,10],[11,12,13,14,15]], dtype=np.float64)), + ("Symmetric tri 5x5", np.array([[2,-1,0,0,0],[-1,2,-1,0,0],[0,-1,2,-1,0],[0,0,-1,2,-1],[0,0,0,-1,2]], dtype=np.float64)), + ("Neg values 2x3", np.array([[0.5,-0.3,0.8],[-0.2,0.7,0.1]], dtype=np.float64)), + ("Near-singular 2x2", np.array([[1,0],[0,1e-6]], dtype=np.float64)), + ("Orthogonal 3x3", np.array([[np.cos(np.pi/4), -np.sin(np.pi/4), 0], + [np.sin(np.pi/4), np.cos(np.pi/4), 0], + [0, 0, 1]], dtype=np.float64)), + ("Identity 3x3", np.eye(3)), + ("Zero 3x3", np.zeros((3,3))), + ("Col vector 2x1", np.array([[3],[4]], dtype=np.float64)), + ("Row vector 1x2", np.array([[3,4]], dtype=np.float64)), + # Large-size instantiation cases (N > 5). Literals MUST match the + # C++ test matrices in unit-tests/matrix-tests.cpp exactly, and the + # C++ references use float32 inputs: cast to float32 before svd(). + ("Tall 7x5", np.array([ + [-0.7528, 2.7043, 1.392, 0.592, -2.0639], + [-2.064, -2.6515, 2.1971, 0.6067, 1.2484], + [-2.8765, 2.8195, 1.9947, -1.726, -1.9091], + [-1.8996, -1.1745, 0.1485, -0.4083, -1.2526], + [0.6711, -2.163, -1.2471, -0.8018, -0.2636], + [1.7111, -1.802, 0.0854, 0.5545, -2.7213], + [0.6453, -1.9769, -2.6097, 2.6933, 2.7938]], dtype=np.float32)), + ("Square 6x6", np.array([ + [1.2336, -0.7815, -1.6093, 0.7369, -0.2394, -1.5118], + [-0.0193, -1.8624, 1.6373, -0.9649, 0.6501, -0.7532], + [0.0803, 0.1868, -1.2606, 1.8783, 1.1005, 1.758], + [1.5793, 0.3916, 1.6875, -1.646, -1.2161, -1.8191], + [-0.6987, -0.4453, -0.9146, 1.315, -0.573, -0.8763], + [0.1708, -1.4363, 1.2088, -1.7018, 1.089, 1.9475]], dtype=np.float32)), + ("Wide 5x8", np.array([ + [-1.5064, -2.4724, 1.5773, 1.0343, 1.145, 1.3564, -2.1298, -0.7077], + [-1.9207, 1.8155, 0.6165, -0.8455, -2.1822, -0.9451, -0.8741, 1.148], + [0.6878, 1.9361, -0.1389, -1.902, 1.0662, 1.3039, 0.3064, 1.3548], + [-0.031, 0.1137, -0.3623, -2.3729, -1.9605, -2.3429, 0.6821, -0.9282], + [0.0429, 2.0378, -1.2535, -0.4481, 1.2778, -1.356, -2.1151, -1.0512]], dtype=np.float32)), + ("Tall 6x4 rank-def", np.array([ + [-0.086904, 1.410225, 1.308323, 2.234762], + [0.022123, 0.896751, 0.324176, 0.773607], + [-0.473015, 1.555111, 0.290059, 1.157726], + [-0.78371, 1.398884, -1.930606, -1.548717], + [0.201518, -0.626835, 0.976596, 0.875294], + [-1.24206, 1.60595, -3.078089, -2.73695]], dtype=np.float32)), + ] + + for name, A in test_matrices: + U, s, Vt = svd(A, full_matrices=False) + + print(f"\n{name}: shape={A.shape}") + print(f" Singular values: {[round(float(x), 12) for x in s]}") + print(f" U:\n{np.array2string(U, precision=6, floatmode='maxprec_equal')}") + print(f" Vt:\n{np.array2string(Vt, precision=6, floatmode='maxprec_equal')}") + recon_err = np.linalg.norm(A - U @ np.diag(s) @ Vt, 'fro') + print(f" Reconstruction error: {recon_err:.2e}") + + # ------------------------------------------------------------------ + # Test 8: Implicit QR Iteration on Bidiagonal + # ------------------------------------------------------------------ + print("\n" + "=" * 70) + print("TEST 8: implicitQRIteration") + print("=" * 70) + + qr_tests = [ + ("2x2 [[1,2],[3,4]]", np.array([[1.0, 2.0], [3.0, 4.0]])), + ("3x3 diag", np.array([[10.0, 0, 0], [0, 5.0, 0], [0, 0, 2.0]])), + ] + + for name, A in qr_tests: + B, QL, QR = householder_bidiagonalization(A) + Sigma, QR_final = implicit_qr_iteration(B.copy(), QR.copy()) + + print(f"\n{name}:") + print(f" Bidiagonal B:\n{B}") + print(f" After QR iteration (Sigma):\n{Sigma}") + print(f" Diagonal entries: {[round(float(Sigma[i,i]), 10) for i in range(min(Sigma.shape))]}") + + # Verify: QL @ Sigma @ QR_final^T ≈ A + recon = QL @ Sigma @ QR_final.T + err = np.linalg.norm(recon - A, 'fro') + print(f" ||QL @ Sigma @ QR^T - A||_F = {err:.2e}") + + # ------------------------------------------------------------------ + # JSON output for easy import into C++ tests + # ------------------------------------------------------------------ + print("\n" + "=" * 70) + print("JSON OUTPUT (for easy C++ integration)") + print("=" * 70) + + json_data = {} + + # Householder test vectors + hh_tests = {} + for name, vec in test_vectors: + v, alpha = compute_householder(vec) + x = np.array(vec) + Hx = x - 2 * np.dot(v, x) * v + hh_tests[name] = { + "input": [float(xi) for xi in x], + "norm": float(np.linalg.norm(x)), + "alpha": float(alpha), + "v_normalized": [round(float(vi), 12) for vi in v], + "Hx": [round(float(xi), 12) for xi in Hx], + } + json_data["householder_vectors"] = hh_tests + + # Full SVD reference values + svd_tests = {} + for name, A in test_matrices: + U, s, Vt = svd(A, full_matrices=False) + svd_tests[name] = { + "shape": list(A.shape), + "singular_values": [round(float(x), 12) for x in s], + "U": [[round(float(U[i,j]), 8) for j in range(U.shape[1])] for i in range(U.shape[0])], + "Vt": [[round(float(Vt[i,j]), 8) for j in range(Vt.shape[1])] for i in range(Vt.shape[0])], + } + json_data["svd_reference"] = svd_tests + + print(json.dumps(json_data, indent=2)) + + +if __name__ == "__main__": + main()