diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index f4fc7e7..542089b 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -19,7 +19,7 @@ jobs: version: 12 platform: x64 - name: make test - run: make CXX=g++-12 test && ./test_test + run: make CXX=g++-12 test && ./build/test_test - name: make example - run: make CXX=g++-12 example && ./test_example + run: make CXX=g++-12 example && ./build/test_example diff --git a/.gitignore b/.gitignore index 207aa78..ceac4d2 100644 --- a/.gitignore +++ b/.gitignore @@ -9,3 +9,4 @@ images/mandelbrot_4/* images/mandelbrot_5/* images/julia_set/* tmp/* +tools/__pycache__/ diff --git a/Makefile b/Makefile index 3469911..debe25d 100644 --- a/Makefile +++ b/Makefile @@ -1,41 +1,52 @@ OPENCV := 0 ifeq ($(OPENCV), 1) - OPENCVOP = -DOPENCV `pkg-config --cflags opencv4` -Wno-deprecated-enum-enum-conversion + OPENCVOP = -DFENG_MATRIX_OPENCV `pkg-config --cflags opencv4` -Wno-deprecated-enum-enum-conversion OPENCVLOP = `pkg-config --libs opencv4` else OPENCVOP = OPENCVLOP = endif -OP = -DPARALLEL +# Portable -O2 by default; FAST=1 restores the old -Ofast -march=native set (D-006). +FAST ?= 0 +OP = -DFENG_MATRIX_PARALLEL CXX = g++ -CXXFLAGS = -std=c++20 -Wall -Wextra -fmax-errors=1 -Ofast -flto=auto -funroll-all-loops -pipe -march=native $(OP) $(OPENCVOP) -LFLAGS = -Ofast $(OPENCVLOP) -pthread -lstdc++fs -Wl,--gc-sections -flto +ifeq ($(FAST), 1) +CXXFLAGS = -std=c++20 -Wall -Wextra -fmax-errors=1 -Ofast -flto=auto -funroll-all-loops -pipe -march=native $(OP) -isystem tests -pthread $(OPENCVOP) +LFLAGS = -Ofast $(OPENCVLOP) -pthread -lstdc++fs -Wl,--gc-sections -flto=auto +else +CXXFLAGS = -std=c++20 -O2 -Wall -Wextra $(OP) -isystem tests -pthread $(OPENCVOP) +LFLAGS = -O2 $(OPENCVLOP) -pthread +endif LINK = $(CXX) ####### Output directory -OBJECTS_DIR = . -BIN_DIR = . -LIB_DIR = . -LOG_DIR = . +BUILD_DIR ?= build +OBJECTS_DIR = $(BUILD_DIR) +BIN_DIR = $(BUILD_DIR) all: test example -test: tests/test.cc ./matrix.hpp +test: $(BIN_DIR)/test_test +example: $(BIN_DIR)/test_example + +$(BUILD_DIR): + mkdir -p $(BUILD_DIR) + +$(BIN_DIR)/test_test: tests/test.cc ./matrix.hpp $(wildcard tests/cases/*.hpp) | $(BUILD_DIR) $(CXX) -c $(CXXFLAGS) -o $(OBJECTS_DIR)/test.o tests/test.cc $(LINK) -o $(BIN_DIR)/test_test $(OBJECTS_DIR)/test.o $(LFLAGS) -example: examples/example.cc ./matrix.hpp +$(BIN_DIR)/test_example: examples/example.cc ./matrix.hpp $(wildcard examples/cases/*.hpp) | $(BUILD_DIR) $(CXX) -c $(CXXFLAGS) -o $(OBJECTS_DIR)/example.o examples/example.cc $(LINK) -o $(BIN_DIR)/test_example $(OBJECTS_DIR)/example.o $(LFLAGS) -.PHONY: clean clean_obj clean_test clean_example +.PHONY: all test example clean clean_obj clean_test clean_example clean: clean_obj clean_test clean_example clean_obj: - rm ./*.o + rm -f $(OBJECTS_DIR)/test.o $(OBJECTS_DIR)/example.o clean_test: - rm ./test_test + rm -f $(BIN_DIR)/test_test clean_example: - rm ./test_example - + rm -f $(BIN_DIR)/test_example diff --git a/ReadMe.md b/ReadMe.md index 809e40c..8a85549 100644 --- a/ReadMe.md +++ b/ReadMe.md @@ -22,6 +22,7 @@ A modern, C++20-native, single-file header-only dense 2D matrix library. - [det -- matrix determinant](#det----matrix-determinant) - [operator `/=`](#operator-divide-equal) - [inverse](#matrix-inverse) + - [linear algebra: lu_factor, svd_factor, linalg_status](#linear-algebra) - [save matrix to images with colormap](#save-matrix-to-images-with-colormap) - [save/load bmp](#save-load-bmp) - [save png](#save-png) @@ -42,6 +43,7 @@ A modern, C++20-native, single-file header-only dense 2D matrix library. - [linspace](#linspace) - [magic](#magic-function) - [matrix convolution](#matrix-convolution) + - [fft](#fft) - [make_view](#make-view-function) - [lu_decomposition](#lu-decomposition) - [guass_jordan_elimination](#gauss-jordan-elimination) @@ -98,11 +100,24 @@ A modern, C++20-native, single-file header-only dense 2D matrix library. g++ -o your_exe_file your_source_code.cpp -std=c++20 -O2 -pthread -lstdc++fs ``` -Please note [`std::thread`](https://en.cppreference.com/w/cpp/header/thread) is not enabled by default. If you prefer multi-thread mode, pass `-DPARALLEL` option to compiler, and add necessary link options. +This is the portable flag set the Makefile uses by default (`make test example`, outputs in `build/`); `-Ofast` and `-march=native` are not needed, and `make FAST=1 ...` opts back into them. See [Building tests and examples](#building-tests-and-examples). + +Please note [`std::thread`](https://en.cppreference.com/w/cpp/header/thread) is not enabled by default. If you prefer multi-thread mode, pass `-DFENG_MATRIX_PARALLEL` to the compiler (the old `-DPARALLEL` is deprecated but still accepted), and add necessary link options. [`std::filesystem`](https://en.cppreference.com/w/cpp/filesystem/path) is used, make sure corresponding library option is passed during link time (`-lstdc++fs` for g++). Variadic macro `__VA_OPT__` is used. It is officially supported since c++20([link1](http://www.open-std.org/jtc1/sc22/wg21/docs/papers/2018/p1042r1.html), [link2](http://www.open-std.org/jtc1/sc22/wg21/docs/papers/2017/p0306r4.html)), so the compiler must be compatible with c++20. +#### Configuration macros + +- `FENG_MATRIX_PARALLEL` enables the `std::jthread` helpers (the old name `PARALLEL` is deprecated but still accepted and maps to it); each call starts its own threads and joins them before returning, on min( hardware_concurrency, work / grain ) workers, so small calls run on the calling thread (see [operator multiply equal](#operator-multiply-equal) below); +- `FENG_MATRIX_OPENCV` enables the `cv::Mat` interface (the old name `OPENCV` is deprecated but still accepted); +- `FENG_MATRIX_CHECKED_ITERATORS` adds bounds checks to the stride and view iterators (see [Views and iterator invalidation](#views-and-iterator-invalidation)); +- `FENG_MATRIX_NO_CONFIG_CHECK` turns off the mixed-configuration link check below. + +All translation units of a program must agree on the first three. On ELF targets every translation unit that includes `matrix.hpp` defines the symbol `feng_matrix_configuration_mismatch_between_translation_units` for its configuration, so linking two translation units with different settings fails with a multiple-definition error naming that symbol, while matching ones link. The check does not exist on non-ELF targets and does not span separate shared libraries. + +Results that should not be ignored are `[[nodiscard]]`: the `bool` of the loaders (`load_txt`, `load_binary`, `load_npy`) and writers (`save_as_txt`, `save_as_binary`, `save_as_npy`, `save_as_png`, `save_as_bmp`, `save_as_pgm`), the `linalg_status` overloads such as `inverse( A, out )`, and the side-effect-free functions that return a new matrix. Discarding one warns (`-Wunused-result`, an error under `-Werror`); write `(void)m.save_as_bmp( ... );` to discard on purpose. Value-returning functions no longer return top-level `const`, so `m = f()` moves. See the `## S9` section of [docs/migration.md](docs/migration.md). + ### basic #### creating matrices @@ -114,7 +129,7 @@ Variadic macro `__VA_OPT__` is used. It is officially supported since c++20([lin ```cpp feng::matrix m{ 12, 34 }; ``` - - creating a random matrix of size `12 X 34`, in range `[0.0, 1.0]`: + - creating a random matrix of size `12 X 34`, in the open interval `(0, 1)` (see [Random numbers](#random-numbers-statistics-and-arithmetic-types)): ```cpp auto rand = feng::rand(12, 34); @@ -185,12 +200,12 @@ Variadic macro `__VA_OPT__` is used. It is officially supported since c++20([lin ``` However, to compile and link the code above, make sure to - 1. define the opencv guard by passing `-DOPENCV` option to the compiler (g++), + 1. define the opencv guard by passing `-DFENG_MATRIX_OPENCV` to the compiler (g++; the old `-DOPENCV` is deprecated but still accepted), 2. tell the compile where to find the opencv header files, for example, passing `pkg-config --cflags opencv4` to the compiler (g++), and 3. tell the linker which libraries to link against, for example, passing `pkg-config --libs opencv4` to the linker (g++). Or 4. compile and link in a single command, such as ```bash - g++ -o ./test_test -std=c++20 -Wall -Wextra -fmax-errors=1 -Ofast -flto=auto -funroll-all-loops -pipe -march=native -DPARALLEL -DOPENCV `pkg-config --cflags opencv4` -Wno-deprecated-enum-enum-conversion `pkg-config --libs opencv4` -pthread -lstdc++fs -Wl,--gc-sections -flto tests/test.cc + g++ -o ./build/test_test -std=c++20 -O2 -Wall -Wextra -DFENG_MATRIX_PARALLEL -isystem tests -DFENG_MATRIX_OPENCV `pkg-config --cflags opencv4` -Wno-deprecated-enum-enum-conversion tests/test.cc `pkg-config --libs opencv4` -pthread ``` @@ -250,6 +265,61 @@ m.save_as_bmp( "./images/0019_create.bmp" ); ![create](./images/0019_create.bmp) +#### Contract checks + +The library is exception-free. A violated precondition writes one line `feng::matrix: contract violation: (:) ` to stderr and calls `std::abort()`. The checks are always on: debug and `-DNDEBUG` builds behave the same, and there is no handler to install. Checked preconditions include: + +- sizes: a negative signed dimension, rows×cols or its byte count overflowing, a byte count above `PTRDIFF_MAX`, or rows×cols above the allocator's `max_size`, checked before anything is allocated; +- `reshape(r, c)` when r×c overflows or differs from `size()`, before the matrix changes; +- `m.at(r, c)` (const and non-const on a matrix, returning a reference; const only on a view from `make_view`) and `m(r, c)` when `r >= m.row()` or `c >= m.col()`, and `m[r]` when `r >= m.row()`; on a 0×0, 0×N or N×0 matrix every index aborts; +- `clone(other, r0, r1, c0, c1)`, its brace-list forms and the slicing constructors when `r1 > other.row()`, `c1 > other.col()`, `r0 >= r1` or `c0 >= c1` (an empty range aborts), or a brace list does not hold two values; +- `shrink_to_size(r, c)` with a zero extent, and `flipdim(m, dim)` with a `dim` other than 1 or 2; +- operand shapes: a `std::valarray` or `std::vector` × matrix product whose length differs from the matrix rows, a matrix × `std::valarray` or `std::vector` product whose length differs from the matrix columns, `+`, `-`, `+=`, `-=` and the matrix–matrix element-wise functions (`fma`, `ldexp`, `scalbn`, `scalbln`, `pow`, `hypot`, `fmod`, `remainder`, `copysign`, `nextafter`, `fdim`, `fmax`, `fmin`, `atan2`, and `feng::matrix_details::map` with two or three matrices) on different shapes, `*` and `*=` when the left columns differ from the right rows, and `/` and `/=` (matrix division) unless the divisor is square with as many rows as the dividend has columns, all checked before any element is read; +- `pooling(m, …, action)` with an action other than `mean`, `average`, `max` or `min`. + +The row pointer returned by `m[r]` is not checked further, so `m[r][c]` stays an unchecked escape hatch; use `at(r, c)` for a checked access. File open, read and write failures also abort for now, until a later stage replaces them with status returns. + +Self-assignment (`a = a`, `a.copy(a)`) keeps the contents, `a *= a` gives the product of the old `a`, and `m.copy(src, {r0, r1}, {c0, c1})` whose source overlaps the destination block (for example a `make_view` of `m` itself) gives the same result as copying a snapshot of the source; `m.copy(v)` and `feng::matrix n{ v }` accept either view type, and `m.copy(v)` with `v` a view of `m` itself copies a snapshot of `v`. The curried helpers `feng::matrix_details::map(func)` and `feng::matrix_details::reduce(func, init)` return lambdas that hold copies of `func` and `init`, so they may be stored (`auto f = feng::matrix_details::reduce(func, init); f(m);`) and called after the arguments are gone; `reduce(func, init)(m)` folds `init` in exactly once and equals `std::accumulate(m.begin(), m.end(), init, func)` for an associative `func`. See the `## S2` and `## S4` sections of [docs/migration.md](docs/migration.md) for the old and new behaviour. + +`shrink_to_size(r, c)` keeps the overlapping top-left block exactly and fills the rest with zeros. The flips follow the MATLAB/NumPy convention: `fliplr(m)` flips columns (same as `flipdim(m, 2)`, column j becomes column cols−1−j) and `flipud(m)` flips rows (same as `flipdim(m, 1)`, row i becomes row rows−1−i); before S2 the two names had their axes swapped. + +#### Element types, storage and callbacks + +`feng::matrix` and `feng::matrix_view` require `T` to satisfy the concept `feng::matrix_element`: a non-const, non-volatile object type other than `bool` whose default, copy and move constructors, copy and move assignments and destructor are all `noexcept` (`std::complex` is accepted even though libstdc++ does not mark its default constructor `noexcept`). `matrix`, `matrix>` and `matrix` are fine; `matrix` and `matrix` do not compile. Use `matrix` for a 0/1 mask: `is_inf`, `isinf`, `is_nan` and `isnan` return one. + +The elements live in a private `std::vector` and the extents are private too; use `m.row()`, `m.col()`, `m.size()`, `m.data()` and `m.get_allocator()`. Copy assignment, move assignment and `swap` follow the allocator's propagation traits; `m.copy(rhs)` keeps `m`'s allocator. The allocator's `value_type` must be `T` (`matrix>` fails a `static_assert`; rebind the allocator instead), and matrix products are built with the left operand's allocator; `real`, `imag`, `abs`, `arg` and `norm` of a complex matrix rebind the operand's allocator (`m.get_allocator()`), so a stateful allocator carries over. Every allocation checks rows×cols against overflow and the allocator's `max_size()` first and aborts with a `matrix size: …` message if it does not fit. A moved-from matrix is 0×0, and `a = std::move(a)` keeps the contents. `matrix(view)` copies exactly the rectangle the view shows. + +Callbacks passed to the library (`apply`, `for_each`, the element-wise maps and the parallel helpers) must not throw. The library is exception-free and gives no exception guarantee; every public operation of `matrix`, `matrix_view` and the free functions in `feng` is `noexcept`, so an allocation failure or an exception escaping a callback calls `std::terminate`. See the `## S3` section of [docs/migration.md](docs/migration.md). + +#### Random numbers, statistics and arithmetic types + +Random generation takes a caller-owned engine: any `g` satisfying `std::uniform_random_bit_generator` (for example `std::mt19937_64 g{ 42 };`) can be passed to `feng::random( r, c, g )`, `random( n, g )`, `rand( r, c, g )`, `rand( n, g )`, `rand_like( m, g )` and `random_like( m, g )`. Elements are drawn in row-major order from `g` alone, so equally seeded engines give identical matrices and each thread can own its engine. `float`, `double` and `long double` elements lie in the open interval (0, 1) (draws of exactly 0 or 1 are redrawn); a `std::complex` element has its real and imaginary parts each in (0, 1); integer elements come from `std::uniform_int_distribution` over the full range of the type, `numeric_limits::min()` to `max()`. The engine-less overloads (`rand( r, c, seed = 0 )`, `rand( n )`, `random( r, c )`, `random( n )`, `rand_like( m )`, `random_like( m )`, `randn_like( m )`) build a local `std::mt19937_64` from the nonzero seed, or from the clock for seed 0, so `rand( r, c, 7 )` equals `random( r, c, g )` with `std::mt19937_64 g{ 7 }`; there is no global random state. + +`feng::mean( m )` returns `double` for integer `T`, `T` for floating-point `T` and `std::complex` for complex `T` (the mean of `int` {1, 2} is 1.5). `feng::variance( m, ddof = 0 )` returns the sum of |x − mean|² divided by n − `ddof`, as the mean's real type, and `feng::standard_deviation( m, ddof = 0 )` is its square root; both share the `ddof` (delta degrees of freedom) parameter, so pass `ddof = 1` for the sample estimate as in NumPy. `max`, `min`, `minmax`, `mean`, `variance` and `standard_deviation` of an empty matrix, and `variance` or `standard_deviation` with n ≤ `ddof`, abort with a message naming the function. + +Mixed-type arithmetic follows one promotion policy, with the kinds ordered integral < floating < complex: + +- `m + n`, `m - n`, `m * n` and `m / n` for `matrix` and `matrix` give `matrix< feng::matrix_details::common_element_t >`: `std::common_type_t` for two real types, `std::complex< std::common_type_t >` when either is complex (X and Y the real parts), so `matrix + matrix` is `matrix` and `matrix + matrix>` is `matrix>`; +- `m ⊕ s` and `s ⊕ m` (⊕ one of `+ - * /`) with a scalar `s` (an arithmetic type or `std::complex`) keep `T` when the scalar's kind is not higher than `T`'s, converting the scalar to `T` (or to `T`'s value type for complex `T`) first, and otherwise give `common_element_t` (the trait `scalar_result_t`): `matrix * 2.0` stays `matrix`, `image + 1` on a `matrix` stays `std::uint8_t`, `matrix * 0.5` is `matrix` and `matrix * 2` compiles; `s / m` is `s` times `m.inverse()`; +- compound assignment (`m += s`, `m *= n`, ...) keeps `T`, converting the operand to `T` as by `static_cast` first (`m *= 0.5` on a `matrix` multiplies by 0); +- the result's allocator is the left matrix's allocator rebound to the result element type. + +Signed integer overflow in element arithmetic is undefined behaviour and not checked: avoiding overflow is a precondition of the caller, as is avoiding integer division by zero; unsigned arithmetic wraps. See the `## S6` section of [docs/migration.md](docs/migration.md). + +#### Linear algebra + +The linear-algebra routines take floating-point or `std::complex` elements; `inverse`, `inv`, `try_inverse`, `det`, `inverse( A, out )`, `lu_factor`, `svd_factor`, `cholesky_factor`, `row_echelon`, `solve`, `pinverse`, `expm`, their factorization classes and the legacy wrappers (`lu_decomposition`, `lu_solver`, `singular_value_decomposition`, `svd`, `pinv`, `svd_inverse`, `cholesky_decomposition`, `rref`, `gauss_jordan_elimination`) are constrained by the concept `feng::linalg_element< T >` (`std::floating_point< T >` or `std::complex`), so an integer matrix fails that constraint at compile time and the diagnostic names `linalg_element`; convert first (`m.astype().det()`). `ctranspose` and `conj` are constrained by `ComplexMatrix`. Numeric failures are reported, never hidden in NaN: `feng::linalg_status` is one of `ok`, `singular`, `not_positive_definite`, `not_converged` and `nonfinite`, and `feng::linalg_result` holds a `value` and a `status` (`r.ok()`, or test `r` in a boolean context). Non-square inputs and mismatched shapes are contract violations and abort with a message. + +- `feng::lu_factor( A )` returns an `lu_factorization` from one partial-pivoting LU: `status()`, `rank()` (|u_kk| > n·ε·max|U|), `pivots()` (row i of P·A is row `pivots()[i]` of A), `l()`, `u()`, `p()`, `det()` and `solve( B )` / `inverse()` (each a `linalg_result`). `feng::solve( A, B )` and `feng::try_inverse( A )` forward to it. `det()` is exactly 0 for a rank-deficient A and 1 for a 0×0 A; `inverse()`, `inverse( A )` and `inv( A )` return an empty 0×0 matrix for a singular or nonfinite A, while `inverse( A, out )` returns the `linalg_status` and leaves `out` unchanged unless it is `ok`. +- `feng::svd_factor( A, max_sweeps = 64 )` returns an `svd_factorization` (one-sided Jacobi, real or complex, any shape) with thin `u()` (m×k), `s()` (k singular values, descending), `v()` (n×k), k = min(m, n), `status()` (`ok` or `not_converged`) and `sweeps()`. `pinverse( A, rtol )`, `pinv( A, rtol )` and `svd_inverse( A )` are one pseudoinverse that drops s_i ≤ rtol·s_1 (default rtol = max(m, n)·ε, as in NumPy 2) and return 0×0 on failure; `pinverse( A, out, rtol )` returns the status. +- `feng::cholesky_factor( A )` returns a `cholesky_factorization` with lower-triangular `l()` (A = L·Lᴴ) or status `not_positive_definite`; the legacy `cholesky_decomposition( m, a )` returns 0 on success and 1 (with `a` unchanged) otherwise. +- `feng::row_echelon( A )` returns an `rref_result{ r, pivot_columns, rank, status }` for any m×n A (pivots below max(m, n)·ε·‖A‖∞ count as zero, as in MATLAB's `rref`); `rref( A )` and `gauss_jordan_elimination( A )` return the same `r`, and `std::nullopt` only for a NaN or inf input. +- `expm( A )` gives an all-NaN matrix for a nonfinite input and `expm( A, out )` returns `nonfinite` instead; `cgs` and `bicgstab` treat `eps` as relative to ‖b‖ and return 1 when they run out of iterations. + +Experimental: `eigen_jacobi`, `cyclic_eigen_jacobi`, `eigen_real_symmetric`, `eigen_hermitian`, `eigen_power_iteration` and `householder` are experimental: S7 did not qualify them, so their results and loop bounds are not tested (D-029). See the `## S7` section of [docs/migration.md](docs/migration.md). + +Under `__cpp_lib_expected` (C++23 and later), `feng::to_expected( r )` turns a `linalg_result`, a factorization or an `rref_result` into a `std::expected< …, linalg_status >`, and `feng::load_expected< T >( path, feng::io_format::npy )` (or `txt`, `binary`) returns a `std::expected< matrix< T >, feng::io_status >`. In C++20 they are absent; `linalg_result`, the factorization objects and the `bool` loaders stay as they are. See the `## S9` section of [docs/migration.md](docs/migration.md). + ---------- @@ -341,6 +411,26 @@ p.save_as_bmp( "./images/0002_slicing.bmp" ); ![matrix slicing](./images/0002_slicing.bmp) +#### Views and iterator invalidation + +`feng::make_view(m, {r0, r1}, {c0, c1})` returns a `matrix_view` (read-only) and `feng::make_mutable_view(m, {r0, r1}, {c0, c1})` returns a `mutable_matrix_view` that writes the owner's elements; a mutable view converts to a const view. Both borrow from an lvalue owner only: a view of a temporary owner, such as `make_view(feng::matrix{3, 3}, {0, 1}, {0, 1})`, does not compile, and `make_mutable_view` of a const owner does not compile either. The ranges must satisfy `r0 <= r1 <= m.row()` and `c0 <= c1 <= m.col()`, with exactly two values per brace list, or the call aborts (in every build); an empty range gives an empty view, and nothing is clamped or normalized. A view carries the parent's row stride (`v.row_stride() == m.col()`), so `v[r]`, `v.row_begin(r)`, `v.col_begin(c)` and `v.begin()` walk the parent's memory directly; `feng::matrix n{ v };` copies the viewed rectangle. + +```cpp +feng::matrix m{ 4, 5 }; +auto w = feng::make_mutable_view( m, {1, 3}, {2, 5} ); // 2x3 block of m +std::fill( w.begin(), w.end(), 1.0 ); // writes m[1..2][2..4] +feng::matrix_view> const v = w; // read-only view of the same block +``` + +A view is not a lifetime guarantee: it holds a pointer to the owner and to its elements. Views and the column, diagonal and anti-diagonal iterators (and the raw row pointers) are invalidated by + +- the owner's destruction; +- reallocation of the owner: `resize`, `reshape`, `clear`, `clone`, a copy into it, a move from it, and `swap`; +- assignment to the owner (copy or move); +- any change of the owner's shape. + +Using an invalidated view or iterator is undefined behaviour. Column and diagonal iterators compare by their logical position, so `col_end(c)` never forms an address past the owner's storage, and an out-of-range column or diagonal index aborts. Building with `-DFENG_MATRIX_CHECKED_ITERATORS` adds the owner's origin and extent to the stride and view iterators and aborts on any dereference or position outside the range (the sanitizer and tagged lanes of `tools/check.sh` build this way); it changes the iterator layout, so all translation units of a program must agree on it. See the `## S4` section of [docs/migration.md](docs/migration.md). + #### meshgrid meshgrid returns 2-D grid coordinates based on the coordinates contained in interger x and y. @@ -759,6 +849,8 @@ after modification ![data](./images/0001_data.bmp) +Span and mdspan adapters: `feng::as_span( m )` is a `std::span` over the contiguous row-major storage and `feng::row_span( m, r )` one over row `r` (an out-of-range `r` aborts), in every language mode. Under `__cpp_lib_mdspan` (C++23 and later) `feng::to_mdspan( m )` is a layout_right `std::mdspan` of shape `m.row()`×`m.col()`, and under `__cpp_lib_submdspan` (C++26) `feng::submdspan( m, {r0, r1}, {c0, c1} )` is a validated block of it; in C++20 use `row_span` or the views of [Views and iterator invalidation](#views-and-iterator-invalidation). The adapters borrow the matrix, so a temporary owner is rejected, and they are invalidated like views. See the `## S9` section of [docs/migration.md](docs/migration.md). + -------------------- @@ -776,6 +868,8 @@ generated output is 1069.00941294551 : 1069.0094129455 ``` +`det()` (and `feng::det( m )`) is sign(P)·∏u_kk from the partial-pivoting LU of [`lu_factor`](#linear-algebra): exactly 0 when the matrix is rank deficient, 1 for a 0×0 matrix, and NaN for a NaN or inf input. An integer matrix must be converted first, e.g. `m.astype().det()`. + ------------- #### operator divide-equal @@ -803,6 +897,8 @@ identity.save_as_bmp( "./images/0000_inverse.bmp" ); ![matrix inverse](./images/0000_inverse.bmp) +`inverse()`, `feng::inverse( m )` and `feng::inv( m )` return an empty 0×0 matrix when `m` is singular or holds a NaN or inf; to see the reason use `feng::try_inverse( m )` (a `linalg_result`) or `feng::linalg_status feng::inverse( m, out )`, which leaves `out` unchanged unless the status is `ok`. + ---------------------------------------- @@ -1061,6 +1157,8 @@ feng::matrix mat; mat.load_npy( "./images/64.npy"); ``` +The loaders validate their input and are transactional: `load_npy`, `load_txt` and `load_binary` return `false` on a missing, malformed, truncated or oversized file and leave the matrix unchanged, `feng::load_bmp` returns an empty optional, and `operator>>` sets `failbit` and keeps the matrix; none of them aborts, and each failing file loader prints one line to `std::cerr`. `load_npy` checks the magic, version, header dict, shape and payload size; the dtype must match the element type exactly (D-021: ``, ``, with `i1`–`i8`, `u1`–`u8`, `c8` and `c16` likewise, no conversion), a foreign byte order is swapped, `fortran_order: True` loads the logical matrix, and a 1-D array loads as 1×n; bool, `f2`, object, structured, rank 0 and rank 3+ arrays are rejected. A dtype size or shape dimension with a leading zero (`>` parse every token completely with `std::from_chars`: ragged rows, trailing characters, an out-of-range value (including a floating-point underflow such as `1e-400` into `double`) and `-1` into an unsigned type are rejected, and `inf` and `nan` load. Every writer (`save_as_txt`, `save_as_binary`, `save_as_npy`, `save_as_bmp`, `save_as_png`, `save_as_pgm`) returns `false` on a directory, open, write or close failure. `m.save_as_npy( "a.npy" )` writes a v1.0 C-order file with the element type's dtype that `numpy.load` and `load_npy` read back. See the `## S5` section of [docs/migration.md](docs/migration.md). + #### save load bmp @@ -1248,6 +1346,8 @@ m.save_as_bmp("images/0001_multiply_equal.bmp"); ![image multiply equal](images/0001_multiply_equal.bmp) +Matrix products (`operator*`, `operator*=` and `direct_multiply`) run one cache-blocked row-major kernel in serial and parallel builds; each entry is the sum over k in ascending order, so serial and parallel builds give bit-identical products. Serial builds no longer switch to Strassen recursion for dimensions of 17 and up (`strassen_multiply` stays callable). In `FENG_MATRIX_PARALLEL` builds the default worker count is work-based: an elementwise call, `for_each`, map or reduction over n elements uses min( hardware_concurrency, n / 65536 ) workers and a product M×K by K×N min( hardware_concurrency, M·K·N / 262144, M ), at least one; starting a thread costs about 19 µs, so a 4×4 add no longer starts 24 threads. Elementwise results do not depend on the worker count; parallel-build reductions of more than 131071 elements fold in that many chunks, so their floating-point rounding depends on the size and the core count (an explicit worker count fixes it). `tools/check.sh bench compare [--smoke]` builds `bench/bench.cc` against the stage start header and the working tree with `-std=c++20 -O2 -pthread`, interleaves pinned runs and judges each optimization in `bench/kept.txt` (at least 10% faster, and no stable workload more than 5% slower); the numbers are in [bench/results.md](bench/results.md). See the `## S10` section of [docs/migration.md](docs/migration.md). + #### operator plus equal @@ -1367,6 +1467,11 @@ matrix linspace( T start, T stop, const std::uint_least64_t num = 50ULL, bool #### magic function +```cpp +template < typename T = std::uint_least64_t, typename A = std::allocator< T > > matrix< T, A > magic( n ) +``` + +`magic( n )` is unchanged and `magic< double >( n )` builds the same square with `double` elements. Since `magic` is a template, `&feng::magic` and passing `feng::magic` as a callable no longer compile; use `&feng::magic<>` or a lambda. See the `## S9` section of [docs/migration.md](docs/migration.md). Calling `magic` method is quite straightforward: @@ -1449,7 +1554,7 @@ m.save_as_bmp( "./images/0000_conv.bmp", "gray" ); ![convolution 1](./images/0000_conv.bmp) -Full 2D matrix convolution with zero-paddings is given by `conv` or `conv2` : +`conv( A, B )` and `conv( A, B, mode )` (`conv2` is the same function) follow `scipy.signal.convolve2d` (D-004, D-030): the kernel B is reversed, out[i][j] = Σ A[p][q]·B[i−p][j−q], so `conv` of [1, 2] with [3, 4] is [3, 10, 8]. The value type is `Mat::value_type`. The default mode is `full`, the (ra+rb−1)×(ca+cb−1) result with zero-paddings: ```cpp @@ -1463,9 +1568,9 @@ edge.save_as_bmp( "./images/0001_conv.bmp", "gray" ); ![convolution 2](./images/0001_conv.bmp) -The convolution has three modes: `valid`, `same`, and `full` +The convolution has three modes, `"full"`, `"same"` and `"valid"`; any other mode string aborts with a message. -The valid mode gives out convolution result without zero-paddings: +The `valid` mode keeps only the entries computed without zero-paddings, (ra−rb+1)×(ca−cb+1) when A contains B in both dimensions; when B contains A the operands are swapped (as scipy does), and when neither contains the other the call aborts: ```cpp auto const& edge_valid = feng::conv( m, filter, "valid" ); @@ -1474,7 +1579,7 @@ edge_valid.save_as_bmp( "./images/0001_conv_valid.bmp", "gray" ); ![convolution valid](./images/0001_conv_valid.bmp) -The `same` mode returns the central part of the convolution result with zero-paddings, of the same size as the larger matrix passed to function `conv` +The `same` mode returns A's shape, cropped from the full result at row (rb−1)/2 and column (cb−1)/2 (integer division), for every kernel shape, including even and 1×1 kernels: ```cpp auto const& edge_same = feng::conv( m, filter, "same" ); @@ -1494,6 +1599,35 @@ edge_full.save_as_bmp( "./images/0001_conv_full.bmp", "gray" ); ![convolution full](./images/0001_conv_full.bmp) +Because the kernel is reversed, an asymmetric kernel gives the true convolution, not a correlation; with this Sobel kernel the result is the horizontal derivative (left minus right neighbours): + +```cpp +feng::matrix sobel{3, 3, {-1.0, 0.0, 1.0, + -2.0, 0.0, 2.0, + -1.0, 0.0, 1.0}}; +auto const& edge_sobel = feng::conv( m, sobel, "same" ); +edge_sobel.save_as_bmp( "./images/0001_conv_sobel.bmp", "gray" ); +``` + +![convolution sobel](./images/0001_conv_sobel.bmp) + +An operand with a zero dimension gives an empty 0×0 matrix in `full` and `valid` mode and zeros of A's shape in `same` mode. See the `## S8` section of [docs/migration.md](docs/migration.md). + + +#### fft + +`feng::fft( x )` and `feng::ifft( x )` follow `numpy.fft.fft2` and `numpy.fft.ifft2`: X[k][l] = Σ x[m][n]·exp(−2πi(km/R + ln/C)), and `ifft` uses exp(+2πi(km/R + ln/C)) divided by R·C, so `ifft( fft( x ) )` ≈ x. Both run in O(N log N) for every size (radix-2 for power-of-two lengths, Bluestein otherwise, no external dependency, D-007). The result is a `feng::matrix` of complex elements: `float`, `double` and `long double` give `std::complex` of the same real type, `std::complex` stays, and integer types give `std::complex` (D-031); an empty input gives an empty result of the same shape. + +`feng::fftshift( x )` and `feng::ifftshift( x )` are pure permutations, no transform: they roll x by (⌊R/2⌋, ⌊C/2⌋) and by (−⌊R/2⌋, −⌊C/2⌋) like `numpy.fft.fftshift` and `ifftshift`, return x's own matrix type, and `ifftshift( fftshift( x ) )` equals x for odd and even sizes. To centre a spectrum write `feng::fftshift( feng::fft( x ) )`. + +```cpp +feng::matrix x( 4, 6 ); +// ... fill x ... +auto const X = feng::fft( x ); // feng::matrix> +auto const centred = feng::fftshift( X ); // zero frequency at (2, 3) +auto const back = feng::ifft( feng::ifftshift( centred ) ); // ≈ x +``` + @@ -1577,7 +1711,9 @@ The noisy image now looks like ![svd_2](./images/0001_singular_value_decomposition.bmp) -We execute Singular Value Decomposition by calling function `std::optional> singular_value_decomposition( matrix const& )`, or `svd` +We execute Singular Value Decomposition by calling function `std::optional> singular_value_decomposition( matrix const& )`, or `svd`. +For an m×n matrix the tuple is `(u, w, v)` with thin shapes: `u` is m×k, `w` is the k×k diagonal matrix of singular values (descending) and `v` is n×k, k = min(m, n), so that `m == u * w * vᴴ` (use `feng::ctranspose( v )` for complex elements). +It is `std::nullopt` when the one-sided Jacobi iteration does not converge within 64 sweeps; `feng::svd_factor( m )` returns the factors with a status instead (see [Linear algebra](#linear-algebra)). ```cpp @@ -1592,10 +1728,10 @@ If the svd is successfully, we can verify the accuricy by reconstructing the noi // check svd result if (svd) // case successful { - // extracted svd result matrices, u, v w - auto const& [u, v, w] = (*svd); - // try to reconstruct matrix using u * v * w' - auto const& m_ = u * v * (w.transpose()); + // extracted svd result matrices: u (r x k), w (k x k, diagonal), v (c x k), k = min(r, c) + auto const& [u, w, v] = (*svd); + // try to reconstruct matrix using u * w * v' + auto const& m_ = u * w * (v.transpose()); // record reconstructed matrix m_.save_as_bmp( "./images/0002_singular_value_decomposition.bmp", "gray" ); ``` @@ -1617,10 +1753,10 @@ The code below demonstrates how: auto new_dm = dm / factor; feng::matrix const new_u{ u, std::make_pair(0UL, r), std::make_pair(0UL, new_dm) }; - feng::matrix const new_v{ v, std::make_pair(0UL, new_dm), std::make_pair(0UL, new_dm) }; - feng::matrix const new_w{ w, std::make_pair(0UL, c), std::make_pair(0UL, new_dm) }; + feng::matrix const new_w{ w, std::make_pair(0UL, new_dm), std::make_pair(0UL, new_dm) }; + feng::matrix const new_v{ v, std::make_pair(0UL, c), std::make_pair(0UL, new_dm) }; - auto const& new_m = new_u * new_v * new_w.transpose(); + auto const& new_m = new_u * new_w * new_v.transpose(); new_m.save_as_bmp( "./images/0003_singular_value_decomposition_"+std::to_string(new_dm)+".bmp", "gray" ); @@ -1763,6 +1899,8 @@ else std::cout << "Failed to execute Gauss-Jordan Elimination for matrix m.\n"; ``` +`gauss_jordan_elimination` and `rref` accept a matrix of any shape (square, tall, wide, rank deficient) and return its reduced row echelon form; a column without a pivot above max(m, n)·ε·‖m‖∞ is skipped, as in MATLAB's `rref`. They return `std::nullopt` only when the matrix holds a NaN or inf; `feng::row_echelon( m )` also gives the pivot columns and the rank. + after applying Gauss Jordan elimination, the matrix is reduced to a form of @@ -1817,9 +1955,11 @@ we can do LU decomposition simply with `std::optional ``` the result of the LU decompositon is a Maybe monad of a tuple of two matrices, i.e., `std::optional, feng::matrix>>`, -therefor, we need to check its value before using it. +therefor, we need to check its value before using it. It is `std::nullopt` when the matrix is singular (rank below n) or holds a NaN or inf. +The factorization uses partial pivoting, and, like MATLAB's two-output `[L, U] = lu(A)`, `L` is the row-permuted lower triangular matrix Pᵀ·L₀, so `A == L * U`; +`feng::lu_factor( m )` gives the unit lower triangular factor, the pivots and the status separately (see [Linear algebra](#linear-algebra)). -Then we can draw the lower matrix `L` +Then we can draw the (permuted) lower matrix `L` ![lu_2](./images/0002_lu_decomposition.bmp) @@ -1856,7 +1996,7 @@ We can multiply `L` and `U` back to see if the decomposition is correct or not. A typical use of LU Decomposition is to solve an equation in the form of `Ax=b`, this is done by calling `auto const& x = feng::lu_solver(A,b)`, and, again, the returned value is a Maybe monad, `std::optional lu_solver( matrix const&, matrix const& )`, -therefore we need to check its value before using it. +therefore we need to check its value before using it (it is `std::nullopt` for a singular `A`). `feng::solve( A, B )` solves for several right-hand sides at once and returns a `linalg_result`. ```cpp @@ -1914,7 +2054,9 @@ This is a single-file header-only library. Put `matrix.hpp` directly into the pr ## Building tests and examples -Simple execute `make` or `make test` or `make example` at the root folder. +Run `make test example` (or `make`) at the root folder. It builds with portable `-std=c++20 -O2 -Wall -Wextra -DFENG_MATRIX_PARALLEL -isystem tests -pthread` into `build/test_test` and `build/test_example` (`BUILD_DIR=...` picks another directory). `make FAST=1 test example` restores the old `-Ofast -flto=auto -funroll-all-loops -march=native` flags. Run both binaries from the root folder: the tests read `./images/*.npy` and the example rewrites `./images/*`. + +The checks run through `tools/check.sh [args] [--smoke]`, which builds only under `build/` and ends with `LANE PASS` or `LANE FAIL: `. Lanes: `gcc` and `clang` (C++20/23/26, debug and NDEBUG, serial and FENG_MATRIX_PARALLEL), `sanitize` (ASan+UBSan), `warnings` (`-Wall -Wextra -Werror`), `api` (one translation unit per API family, with and without `-fno-exceptions`), `examples` (runs the example in a scratch copy of `images/`), `make`, `ci`, `docs `, `tagged ''` and `all` (every regression lane; `tools/check.sh all --smoke` for a quick run). ## Notes and references @@ -2110,10 +2252,10 @@ namespace xxx void plot( string const& file_name_, string const& builtin_color_scheme_name_ ) const; //unary operators - matrix const operator+() const; - matrix const operator-() const; - matrix const operator~() const; - matrix const operator~() const; //bool only + matrix operator+() const; + matrix operator-() const; + matrix operator~() const; + matrix operator~() const; //bool only //computed assignment //TODO: return optional? @@ -2141,123 +2283,124 @@ namespace xxx }; //building functions - template< typename T, typename A > matrix const make_eye( size_type row_, size_type col_, A const& alloc_ ); - template< typename T > matrix> const make_eye( size_type row_, size_type col_ ); - template< typename T, typename A > matrix const make_zeros( size_type row_, size_type col_, A const& alloc_ ); - template< typename T > matrix> const make_zeros( size_type row_, size_type col_ ); - template< typename T, typename A > matrix const make_ones( size_type row_, size_type col_, A const& alloc_ ); - template< typename T > matrix> const make_ones( size_type row_, size_type col_ ); - template< typename T, typename A > matrix const make_diag( matrix const ); - template< typename T, typename A > matrix const make_triu( matrix const ); - template< typename T, typename A > matrix const make_tril( matrix const ); - template< typename T, typename A > matrix const make_rand( size_type row_, size_type col_, A const& alloc_ ); - template< typename T > matrix> const make_rand( size_type row_, size_type col_ ); - template< typename T, typename A > matrix const make_hilb( size_type n_, A const& alloc_ ); - template< typename T > matrix> const make_hilb( size_type n_ ); - template< typename T, typename A > matrix const make_magic( size_type n_, A const& alloc_ ); - template< typename T > matrix> const make_magic( size_type n_ ); - template< typename T, typename A, typename Input_Itor_1, typename Input_Iterator_2 > matrix const make_toeplitz( Input_Iterator_1 begin_, Input_Iterator_1 end_, Input_Iterator_2 begin_2_, A const alloc_ ); - template< typename T, typename A, typename Input_Itor > matrix const make_toeplitz( Input_Iterator begin_, Input_Iterator end_, A const alloc_ ); - template< typename T, typename Input_Itor_1, typename Input_Iterator_2 > const matrix > make_toeplitz( Input_Iterator_1 begin_, Input_Iterator_1 end_, Input_Iterator_2 begin_2_ ): - template< typename T, typename Input_Itor > matrix> const make_toeplitz( Input_Iterator begin_, Input_Iterator end_ ); - template< typename T, typename A > matrix const make_horizontal_cons( matrix const&, matrix const& ); - template< typename T, typename A > matrix const make_vertical_cons( matrix const&, matrix const& ); + template< typename T, typename A > matrix make_eye( size_type row_, size_type col_, A const& alloc_ ); + template< typename T > matrix> make_eye( size_type row_, size_type col_ ); + template< typename T, typename A > matrix make_zeros( size_type row_, size_type col_, A const& alloc_ ); + template< typename T > matrix> make_zeros( size_type row_, size_type col_ ); + template< typename T, typename A > matrix make_ones( size_type row_, size_type col_, A const& alloc_ ); + template< typename T > matrix> make_ones( size_type row_, size_type col_ ); + template< typename T, typename A > matrix make_diag( matrix const ); + template< typename T, typename A > matrix make_triu( matrix const ); + template< typename T, typename A > matrix make_tril( matrix const ); + template< typename T, typename A > matrix make_rand( size_type row_, size_type col_, A const& alloc_ ); + template< typename T > matrix> make_rand( size_type row_, size_type col_ ); + template< typename T, typename A > matrix make_hilb( size_type n_, A const& alloc_ ); + template< typename T > matrix> make_hilb( size_type n_ ); + template< typename T, typename A > matrix make_magic( size_type n_, A const& alloc_ ); + template< typename T > matrix> make_magic( size_type n_ ); + template< typename T, typename A, typename Input_Itor_1, typename Input_Iterator_2 > matrix make_toeplitz( Input_Iterator_1 begin_, Input_Iterator_1 end_, Input_Iterator_2 begin_2_, A const alloc_ ); + template< typename T, typename A, typename Input_Itor > matrix make_toeplitz( Input_Iterator begin_, Input_Iterator end_, A const alloc_ ); + template< typename T, typename Input_Itor_1, typename Input_Iterator_2 > matrix > make_toeplitz( Input_Iterator_1 begin_, Input_Iterator_1 end_, Input_Iterator_2 begin_2_ ): + template< typename T, typename Input_Itor > matrix> make_toeplitz( Input_Iterator begin_, Input_Iterator end_ ); + template< typename T, typename A > matrix make_horizontal_cons( matrix const&, matrix const& ); + template< typename T, typename A > matrix make_vertical_cons( matrix const&, matrix const& ); //binary operation //TODO: return optional? - template< typename T, typename A > matrix const operator * ( matrix const&, matrix const& ); - template< typename T, typename A > matrix const operator * ( matrix const&, T const& ); - template< typename T, typename A > matrix const operator * ( T const&, matrix const& ); + template< typename T, typename A > matrix operator * ( matrix const&, matrix const& ); + template< typename T, typename A > matrix operator * ( matrix const&, T const& ); + template< typename T, typename A > matrix operator * ( T const&, matrix const& ); - template< typename T, typename A > matrix const operator / ( matrix const&, matrix const& ); - template< typename T, typename A > matrix const operator / ( matrix const&, T const& ); - template< typename T, typename A > matrix const operator / ( T const&, matrix const& ); + template< typename T, typename A > matrix operator / ( matrix const&, matrix const& ); + template< typename T, typename A > matrix operator / ( matrix const&, T const& ); + template< typename T, typename A > matrix operator / ( T const&, matrix const& ); - template< typename T, typename A > matrix const operator + ( matrix const&, matrix const& ); - template< typename T, typename A > matrix const operator + ( matrix const&, T const& ); - template< typename T, typename A > matrix const operator + ( T const&, matrix const& ); + template< typename T, typename A > matrix operator + ( matrix const&, matrix const& ); + template< typename T, typename A > matrix operator + ( matrix const&, T const& ); + template< typename T, typename A > matrix operator + ( T const&, matrix const& ); - template< typename T, typename A > matrix const operator - ( matrix const&, matrix const& ); - template< typename T, typename A > matrix const operator - ( matrix const&, T const& ); - template< typename T, typename A > matrix const operator - ( T const&, matrix const& ); + template< typename T, typename A > matrix operator - ( matrix const&, matrix const& ); + template< typename T, typename A > matrix operator - ( matrix const&, T const& ); + template< typename T, typename A > matrix operator - ( T const&, matrix const& ); - template< typename T, typename A > matrix const operator % ( matrix const&, T const& ); - template< typename T, typename A > matrix const operator ^ ( matrix const&, T const& ); - template< typename T, typename A > matrix const operator & ( matrix const&, T const& ); - template< typename T, typename A > matrix const operator | ( matrix const&, T const& ); - template< typename T, typename A > matrix const operator << ( matrix const&, T const& ); - template< typename T, typename A > matrix const operator >> ( matrix const&, T const& ); + template< typename T, typename A > matrix operator % ( matrix const&, T const& ); + template< typename T, typename A > matrix operator ^ ( matrix const&, T const& ); + template< typename T, typename A > matrix operator & ( matrix const&, T const& ); + template< typename T, typename A > matrix operator | ( matrix const&, T const& ); + template< typename T, typename A > matrix operator << ( matrix const&, T const& ); + template< typename T, typename A > matrix operator >> ( matrix const&, T const& ); template< typename T, typename A > std::ostream& operator << ( std::ostream&, matrix const& ); template< typename T, typename A > std::istream& operator << ( std::istream&, matrix const& ); - //logical - template< typename T, typename A > matrix::rebind_alloc > const operator == ( matrix const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator == ( matrix const&, T const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator == ( T const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator != ( matrix const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator != ( matrix const&, T const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator != ( T const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator >= ( matrix const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator >= ( matrix const&, T const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator >= ( T const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator <= ( matrix const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator <= ( matrix const&, T const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator <= ( T const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator > ( matrix const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator > ( matrix const&, T const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator > ( T const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator < ( matrix const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator < ( matrix const&, T const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator < ( T const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator && ( matrix const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator && ( matrix const&, T const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator && ( T const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator || ( matrix const&, matrix const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator || ( matrix const&, T const& ); - template< typename T, typename A > matrix::rebind_alloc > const operator || ( T const&, matrix const& ); + //logical (design only, not implemented: since S3 `matrix` is ill-formed, see `matrix_element`; the + //implemented matrix-matrix ==, <, >, <= and >= return a single bool) + template< typename T, typename A > matrix::rebind_alloc > operator == ( matrix const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator == ( matrix const&, T const& ); + template< typename T, typename A > matrix::rebind_alloc > operator == ( T const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator != ( matrix const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator != ( matrix const&, T const& ); + template< typename T, typename A > matrix::rebind_alloc > operator != ( T const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator >= ( matrix const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator >= ( matrix const&, T const& ); + template< typename T, typename A > matrix::rebind_alloc > operator >= ( T const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator <= ( matrix const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator <= ( matrix const&, T const& ); + template< typename T, typename A > matrix::rebind_alloc > operator <= ( T const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator > ( matrix const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator > ( matrix const&, T const& ); + template< typename T, typename A > matrix::rebind_alloc > operator > ( T const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator < ( matrix const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator < ( matrix const&, T const& ); + template< typename T, typename A > matrix::rebind_alloc > operator < ( T const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator && ( matrix const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator && ( matrix const&, T const& ); + template< typename T, typename A > matrix::rebind_alloc > operator && ( T const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator || ( matrix const&, matrix const& ); + template< typename T, typename A > matrix::rebind_alloc > operator || ( matrix const&, T const& ); + template< typename T, typename A > matrix::rebind_alloc > operator || ( T const&, matrix const& ); //numeric functions - template< typename T, typename A, typename F > matrix const element_wise_apply ( matrix const&, F const& f_ ); + template< typename T, typename A, typename F > matrix element_wise_apply ( matrix const&, F const& f_ ); //Linear Equations //mldivide Solve systems of linear equations Ax = B for x - template< typename T, typename A, typename F > matrix const mldivide ( matrix const&, matrix const& ); + template< typename T, typename A, typename F > matrix mldivide ( matrix const&, matrix const& ); //mrdivide Solve systems of linear equations xA = B for x - template< typename T, typename A, typename F > matrix const mrdivide ( matrix const&, matrix const& ); + template< typename T, typename A, typename F > matrix mrdivide ( matrix const&, matrix const& ); //linsolve Solve linear system of equations - template< typename T, typename A, typename F > matrix const mrdivide ( matrix const&, matrix const& ); + template< typename T, typename A, typename F > matrix mrdivide ( matrix const&, matrix const& ); //inv Matrix inverse - template< typename T, typename A, typename F > matrix const inv ( matrix const& ); + template< typename T, typename A, typename F > matrix inv ( matrix const& ); //pinv Moore-Penrose pseudoinverse of matrix - template< typename T, typename A, typename F > matrix const pinv ( matrix const& ); + template< typename T, typename A, typename F > matrix pinv ( matrix const& ); //lscov Least-squares solution in presence of known covariance - template< typename T, typename A, typename F > matrix const lscov ( matrix const&, matrix const& ); - template< typename T, typename A, typename F > matrix const lscov ( matrix const&, matrix const&, matrix const& ); + template< typename T, typename A, typename F > matrix lscov ( matrix const&, matrix const& ); + template< typename T, typename A, typename F > matrix lscov ( matrix const&, matrix const&, matrix const& ); //lsqnonneg Solve nonnegative linear least-squares problem - template< typename T, typename A, typename F > matrix const lsqnonneg ( matrix const&, matrix const& ); - template< typename T, typename A, typename F > matrix const lsqnonneg ( matrix const&, matrix const&, matrix const& ); + template< typename T, typename A, typename F > matrix lsqnonneg ( matrix const&, matrix const& ); + template< typename T, typename A, typename F > matrix lsqnonneg ( matrix const&, matrix const&, matrix const& ); //sylvester Solve Sylvester equation AX + XB = C for X - template< typename T, typename A, typename F > matrix const sylvester ( matrix const&, matrix const&, matrix const& ); + template< typename T, typename A, typename F > matrix sylvester ( matrix const&, matrix const&, matrix const& ); //Eigenvalues ans Singular Values //eig Eigenvalues and eigenvectors - template< typename T, typename A, typename F > std::tuple,matrix> const eig( matrix const& ); + template< typename T, typename A, typename F > std::tuple,matrix> eig( matrix const& ); //eigs Subset of eigenvalues and eigenvectors -- TODO //balance Diagonal scaling to improve eigenvalue accuracy //svd Singular value decomposition - template< typename T, typename A, typename F > std::tuple,matrix,matrix> const svd( matrix const& ); + template< typename T, typename A, typename F > std::tuple,matrix,matrix> svd( matrix const& ); //svds Subset of singular values and vectors -- TODO //gsvd Generalized singular value decomposition - template< typename T, typename A, typename F > std::tuple,matrix,matrix,matrix> const gsvd( matrix const&, matrix const& ); + template< typename T, typename A, typename F > std::tuple,matrix,matrix,matrix> gsvd( matrix const&, matrix const& ); //ordeig Eigenvalues of quasitriangular matrices - template< typename T, typename A, typename F > matrix const ordeig( matrix const& ); - template< typename T, typename A, typename F > matrix const ordeig( matrix const&, matrix const& ); + template< typename T, typename A, typename F > matrix ordeig( matrix const& ); + template< typename T, typename A, typename F > matrix ordeig( matrix const&, matrix const& ); diff --git a/bench/bench.cc b/bench/bench.cc new file mode 100644 index 0000000..43489b6 --- /dev/null +++ b/bench/bench.cc @@ -0,0 +1,294 @@ +// bench/bench.cc: the S10-R1/S10-R2 (D-008) benchmark harness, built twice by `tools/check.sh bench compare` (once on +// the stage start commit's matrix.hpp, once on the working tree) with the same flags; it uses only API present at the +// baseline. One translation unit: matrix.hpp plus the standard library. +// +// Each workload builds its inputs from std::mt19937_64 seeded with a constant plus a hash of its name (without a par/ +// prefix), then: +// warmup: runs a batch of `reps` operations, doubling reps (from 1, capped) until one batch lasts >= 5 ms (1 ms with +// --smoke); an operation slower than that keeps reps = 1; +// timing: n samples of one batch each on steady_clock, printed as `SAMPLES reps ... ` in seconds +// per operation. +// Every result feeds a checksum sink, printed as `CHECKSUM `, so nothing is optimized away. The run starts with +// `INFO cpu `, `INFO cores `, `INFO compiler <__VERSION__>` and `INFO flags `. +// CLI: --list (names, one per line; with --smoke the smoke subset, one small size per family), --only (repeat +// to run several), --samples (default 10; 3 with --smoke), --smoke. +// Parallel variant (S10-T6): built with -DFENG_MATRIX_PARALLEL the harness offers only the par/ set (in_par_set: +// elementwise and statistics at every size, gemm 8/16/17/18/32/64/256/1024/tall/wide, gemv 8/32/64/1024), each +// name prefixed `par/` (e.g. par/add/4x4); --list, --smoke and --only work on those prefixed names. +#include "matrix.hpp" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#ifndef BENCH_FLAGS +#define BENCH_FLAGS "unknown" +#endif + +namespace +{ + using dmat = feng::matrix< double >; + using cmat = feng::matrix< std::complex< double > >; + + double sink = 0.0; + + void touch( double x ) { sink += x; } + void touch( std::complex< double > x ) { sink += x.real() + x.imag(); } + template < typename M > + void touch_matrix( M const& m ) + { + if ( m.size() != 0 ) touch( m[m.row() / 2][m.col() / 2] ); + } + + std::uint64_t name_hash( std::string const& s ) + { + std::uint64_t h = 1469598103934665603ULL; + for ( unsigned char c : s ) { h ^= c; h *= 1099511628211ULL; } + return h; + } + + dmat random_dmat( std::mt19937_64& gen, std::size_t r, std::size_t c, double lo, double hi ) + { + std::uniform_real_distribution< double > dist{ lo, hi }; + dmat m( r, c ); + for ( auto& v : m ) v = dist( gen ); + return m; + } + + cmat random_cmat( std::mt19937_64& gen, std::size_t r, std::size_t c ) + { + std::uniform_real_distribution< double > dist{ -1.0, 1.0 }; + cmat m( r, c ); + for ( auto& v : m ) v = std::complex< double >{ dist( gen ), dist( gen ) }; + return m; + } + + using op_type = std::function< void() >; + + struct workload + { + std::string name; + bool smoke; + std::function< op_type( std::mt19937_64& ) > setup; // builds the inputs, returns one operation + }; + + // The workloads of the parallel build (see the header comment); names without the par/ prefix. + bool in_par_set( std::string const& name ) + { + static char const* const fixed[] = { "gemm/8", "gemm/16", "gemm/17", "gemm/18", "gemm/32", "gemm/64", + "gemm/256", "gemm/1024", "gemm/tall", "gemm/wide", + "gemv/8", "gemv/32", "gemv/64", "gemv/1024" }; + for ( char const* f : fixed ) + if ( name == f ) return true; + static char const* const families[] = { "add/", "scale/", "map/", "sum/", "mean/", "variance/", "max/" }; + for ( char const* f : families ) + if ( name.rfind( f, 0 ) == 0 ) return true; + return false; + } + + std::vector< workload > make_workloads_all(); + + std::vector< workload > make_workloads() + { +#ifdef FENG_MATRIX_PARALLEL + std::vector< workload > par; + for ( auto& w : make_workloads_all() ) + if ( in_par_set( w.name ) ) par.push_back( { "par/" + w.name, w.smoke, std::move( w.setup ) } ); + return par; +#else + return make_workloads_all(); +#endif + } + + std::vector< workload > make_workloads_all() + { + std::vector< workload > w; + auto gemm = []( std::size_t m, std::size_t k, std::size_t n ) + { + return [m, k, n]( std::mt19937_64& gen ) -> op_type + { + auto a = std::make_shared< dmat >( random_dmat( gen, m, k, -1.0, 1.0 ) ); + auto b = std::make_shared< dmat >( random_dmat( gen, k, n, -1.0, 1.0 ) ); + return [a, b] { touch_matrix( ( *a ) * ( *b ) ); }; + }; + }; + for ( std::size_t n : { 8, 16, 17, 18, 32, 64, 128, 256, 512, 1024 } ) + w.push_back( { "gemm/" + std::to_string( n ), n == 32, gemm( n, n, n ) } ); + w.push_back( { "gemm/tall", false, gemm( 4096, 64, 64 ) } ); + w.push_back( { "gemm/wide", false, gemm( 64, 64, 4096 ) } ); + for ( std::size_t n : { 8, 16, 32, 64, 128, 256, 512, 1024 } ) + w.push_back( { "gemv/" + std::to_string( n ), n == 32, gemm( n, n, 1 ) } ); + + auto shaped = []( std::size_t n ) { return std::to_string( n ) + "x" + std::to_string( n ); }; + for ( std::size_t n : { 4, 32, 100, 1000 } ) + { + bool const sm = n == 32; + auto const sz = shaped( n ); + w.push_back( { "add/" + sz, sm, [n]( std::mt19937_64& gen ) -> op_type { + auto a = std::make_shared< dmat >( random_dmat( gen, n, n, 0.5, 1.5 ) ); + auto b = std::make_shared< dmat >( random_dmat( gen, n, n, 0.5, 1.5 ) ); + return [a, b] { touch_matrix( ( *a ) + ( *b ) ); }; } } ); + w.push_back( { "scale/" + sz, sm, [n]( std::mt19937_64& gen ) -> op_type { + auto a = std::make_shared< dmat >( random_dmat( gen, n, n, 0.5, 1.5 ) ); + return [a] { touch_matrix( ( *a ) * 1.0001 ); }; } } ); + w.push_back( { "map/" + sz, sm, [n]( std::mt19937_64& gen ) -> op_type { + auto a = std::make_shared< dmat >( random_dmat( gen, n, n, 0.5, 1.5 ) ); + return [a] { touch_matrix( feng::sqrt( *a ) ); }; } } ); + w.push_back( { "sum/" + sz, sm, [n]( std::mt19937_64& gen ) -> op_type { + auto a = std::make_shared< dmat >( random_dmat( gen, n, n, 0.5, 1.5 ) ); + return [a] { touch( feng::sum( *a ) ); }; } } ); + w.push_back( { "mean/" + sz, sm, [n]( std::mt19937_64& gen ) -> op_type { + auto a = std::make_shared< dmat >( random_dmat( gen, n, n, 0.5, 1.5 ) ); + return [a] { touch( feng::mean( *a ) ); }; } } ); + w.push_back( { "variance/" + sz, sm, [n]( std::mt19937_64& gen ) -> op_type { + auto a = std::make_shared< dmat >( random_dmat( gen, n, n, 0.5, 1.5 ) ); + return [a] { touch( feng::variance( *a ) ); }; } } ); + w.push_back( { "max/" + sz, sm, [n]( std::mt19937_64& gen ) -> op_type { + auto a = std::make_shared< dmat >( random_dmat( gen, n, n, 0.5, 1.5 ) ); + return [a] { touch( feng::max( *a ) ); }; } } ); + } + + for ( std::size_t n : { 64, 1024 } ) + { + bool const sm = n == 64; + auto const sz = std::to_string( n ); + w.push_back( { "transpose/" + sz, sm, [n]( std::mt19937_64& gen ) -> op_type { + auto a = std::make_shared< dmat >( random_dmat( gen, n, n, -1.0, 1.0 ) ); + return [a] { touch_matrix( a->transpose() ); }; } } ); + // a view of the n x n block at (n/4, n/3) of a 2n x 2n owner + w.push_back( { "view_copy/" + sz, sm, [n]( std::mt19937_64& gen ) -> op_type { + auto a = std::make_shared< dmat >( random_dmat( gen, 2 * n, 2 * n, -1.0, 1.0 ) ); + return [a, n] { + dmat const c{ feng::make_view( *a, { n / 4, n / 4 + n }, { n / 3, n / 3 + n } ) }; + touch_matrix( c ); }; } } ); + w.push_back( { "view_sum/" + sz, sm, [n]( std::mt19937_64& gen ) -> op_type { + auto a = std::make_shared< dmat >( random_dmat( gen, 2 * n, 2 * n, -1.0, 1.0 ) ); + return [a, n] { + auto const v = feng::make_view( *a, { n / 4, n / 4 + n }, { n / 3, n / 3 + n } ); + touch( std::accumulate( v.begin(), v.end(), 0.0 ) ); }; } } ); + } + + for ( std::size_t n : { 128, 256, 125, 250 } ) + w.push_back( { "fft/" + std::to_string( n ), n == 128 || n == 125, [n]( std::mt19937_64& gen ) -> op_type { + auto x = std::make_shared< cmat >( random_cmat( gen, n, n ) ); + return [x] { touch_matrix( feng::fft( *x ) ); }; } } ); + + auto conv = []( std::size_t n, std::size_t k ) + { + return [n, k]( std::mt19937_64& gen ) -> op_type + { + auto a = std::make_shared< dmat >( random_dmat( gen, n, n, -1.0, 1.0 ) ); + auto b = std::make_shared< dmat >( random_dmat( gen, k, k, -1.0, 1.0 ) ); + return [a, b] { touch_matrix( feng::conv( *a, *b ) ); }; + }; + }; + w.push_back( { "conv/64k5", true, conv( 64, 5 ) } ); + w.push_back( { "conv/256k7", false, conv( 256, 7 ) } ); + return w; + } + + std::string cpu_model() + { + std::ifstream in{ "/proc/cpuinfo" }; + std::string line; + while ( std::getline( in, line ) ) + if ( line.rfind( "model name", 0 ) == 0 ) + { + auto const p = line.find( ':' ); + if ( p != std::string::npos ) + { + auto s = line.substr( p + 1 ); + s.erase( 0, s.find_first_not_of( " \t" ) ); + return s; + } + } + return "unknown"; + } + + void run_workload( workload const& w, int samples, double target ) + { + // par/ workloads hash the unprefixed name, so they get the same inputs as their serial twins + std::string const base_name = w.name.rfind( "par/", 0 ) == 0 ? w.name.substr( 4 ) : w.name; + std::mt19937_64 gen{ 20261001ULL + name_hash( base_name ) }; + op_type const op = w.setup( gen ); + auto const batch = [&]( long reps ) + { + auto const t0 = std::chrono::steady_clock::now(); + for ( long r = 0; r != reps; ++r ) op(); + auto const t1 = std::chrono::steady_clock::now(); + return std::chrono::duration< double >( t1 - t0 ).count(); + }; + long reps = 1; + long const reps_cap = 1L << 24; + while ( batch( reps ) < target && reps < reps_cap ) reps *= 2; + std::printf( "SAMPLES %s reps %ld", w.name.c_str(), reps ); + for ( int s = 0; s != samples; ++s ) std::printf( " %.6e", batch( reps ) / static_cast< double >( reps ) ); + std::printf( "\n" ); + std::fflush( stdout ); + } +} + +int main( int argc, char** argv ) +{ + bool list = false, smoke = false; + int samples = -1; + std::vector< std::string > only; + for ( int i = 1; i < argc; ++i ) + { + if ( std::strcmp( argv[i], "--list" ) == 0 ) list = true; + else if ( std::strcmp( argv[i], "--smoke" ) == 0 ) smoke = true; + else if ( std::strcmp( argv[i], "--only" ) == 0 && i + 1 < argc ) only.emplace_back( argv[++i] ); + else if ( std::strcmp( argv[i], "--samples" ) == 0 && i + 1 < argc ) + { + samples = std::atoi( argv[++i] ); + if ( samples < 1 ) { std::fprintf( stderr, "bench: --samples needs a positive count\n" ); return 2; } + } + else { std::fprintf( stderr, "bench: unknown argument '%s'\nusage: bench [--list] [--only ]... [--samples ] [--smoke]\n", argv[i] ); return 2; } + } + if ( samples < 0 ) samples = smoke ? 3 : 10; + + auto const all = make_workloads(); + if ( list ) + { + for ( auto const& w : all ) + if ( !smoke || w.smoke ) std::printf( "%s\n", w.name.c_str() ); + return 0; + } + + std::vector< workload const* > chosen; + if ( only.empty() ) + { + for ( auto const& w : all ) + if ( !smoke || w.smoke ) chosen.push_back( &w ); + } + else + for ( auto const& name : only ) + { + workload const* found = nullptr; + for ( auto const& w : all ) + if ( w.name == name ) found = &w; + if ( !found ) { std::fprintf( stderr, "bench: unknown workload '%s'\n", name.c_str() ); return 2; } + chosen.push_back( found ); + } + + std::printf( "INFO cpu %s\n", cpu_model().c_str() ); + std::printf( "INFO cores %u\n", std::thread::hardware_concurrency() ); + std::printf( "INFO compiler %s\n", __VERSION__ ); + std::printf( "INFO flags %s\n", BENCH_FLAGS ); + std::fflush( stdout ); + double const target = smoke ? 0.001 : 0.005; + for ( auto const* w : chosen ) run_workload( *w, samples, target ); + std::printf( "CHECKSUM %.17g\n", sink ); + return 0; +} diff --git a/bench/kept.txt b/bench/kept.txt new file mode 100644 index 0000000..68cc7f1 --- /dev/null +++ b/bench/kept.txt @@ -0,0 +1,8 @@ +# bench/kept.txt: the kept S10 optimizations judged by `tools/check.sh bench compare` (S10-R3, D-008). +# One line per optimization: ` ...`, workload names as printed by `bench --list`. Each listed +# workload must be at least 10% faster (median) than the baseline. `#` starts a comment; blank lines are ignored. +gemm-blocked gemm/8 gemm/16 gemm/17 gemm/18 gemm/32 gemm/64 gemm/128 gemm/256 gemm/512 gemm/1024 gemm/tall gemm/wide gemv/8 gemv/16 gemv/32 gemv/64 gemv/128 gemv/256 gemv/512 gemv/1024 +# par-thresholds (S10-T7): work-based default worker counts in FENG_MATRIX_PARALLEL builds (matrix_details::work_workers). +par-thresholds par/add/4x4 par/add/32x32 par/add/100x100 par/add/1000x1000 par/map/100x100 par/map/1000x1000 par/sum/32x32 par/sum/100x100 par/sum/1000x1000 par/gemm/8 par/gemm/16 par/gemm/17 par/gemm/18 par/gemm/32 par/gemm/64 par/gemv/8 par/gemv/32 par/gemv/64 par/gemv/1024 +# fft-plans (S10-T8): one plan per length per fft2 call, reused across rows and columns (matrix_details::fft_plan). +fft-plans fft/128 fft/256 fft/125 fft/250 diff --git a/bench/results.md b/bench/results.md new file mode 100644 index 0000000..4b42000 --- /dev/null +++ b/bench/results.md @@ -0,0 +1,197 @@ +# S10 benchmark results + +The measured optimizations of stage S10 (S10-R3, D-008). Each kept optimization is listed in `bench/kept.txt` +under the same id and is judged by the compare lane on every full run. + +## How to run + +- `tools/check.sh bench compare`: full run, every workload of `bench --list`; judges kept optimizations and + regressions; ends with `LANE bench PASS: …` or `LANE bench FAIL: …`. About 145 s on the host below with the + parallel variant (95 s serial only). +- `tools/check.sh bench compare --smoke`: one small size per family plus the kept workloads; kept optimizations + are judged, regressions are printed but not judged. +- `BENCH_CPU=` sets the CPU the runs are pinned to (default 2); `BENCH_ONLY=` restricts the workloads, + for experiments only (a run that leaves out a kept workload cannot pass). +- Logs go to `build/logs/bench/`; binaries to `build/bench/compare/{base,head,base-par,head-par}/bench`. + +## Method + +- Baseline: `matrix.hpp` at the parent of the oldest commit whose subject is exactly `S10: spec and tasks`, + extracted with `git show`; head: the working tree. Both builds compile the same `bench/bench.cc` with the + first GCC and exactly `-std=c++20 -O2 -pthread` (no -O3, -march=native or fast-math, D-006). +- Inputs come from `std::mt19937_64` seeded with a constant plus a hash of the workload name. +- Each process warms up first: doubling a batch of operations until one batch lasts at least 5 ms (1 ms smoke). + Each sample is then one batch, in seconds per operation. +- Interleaving: per workload the lane runs 8 processes per build in the order ABBA BAAB ABBA BAAB (A base, + B head), 5 samples each, pinned with `taskset -c $BENCH_CPU`; the 40 samples of each build are pooled + (smoke: ABBA, 3 samples each, 6 pooled). +- Per build: median and rel. dispersion = median absolute deviation / median. Change = head median / base + median − 1. A workload is stable when its rel. dispersion is at most 5% in both builds (D-035). +- Judgement (D-008): every workload of a kept optimization has change ≤ −10%; no stable workload has change + > +5%; unstable workloads are printed with their numbers and not judged. + +## Host + +From the INFO lines of the full run below (head at commit c8b3206, baseline 78dc886, the parent of 3cbad44): + +- CPU: AMD Ryzen 9 7900X3D 12-Core Processor, 24 cores; pinned to cpu 2 (`taskset -c 2`); par/ workloads + unpinned +- Compiler: g++ 16.2.1 20260810 (all four builds); flags `-std=c++20 -O2 -pthread` (base, head) and + `-std=c++20 -O2 -pthread -DFENG_MATRIX_PARALLEL` (base-par, head-par) +- Mode: full, order ABBA BAAB ABBA BAAB, 8 runs x 5 samples = 40 pooled per build +- Result: `LANE bench PASS: 3 kept, 102 workloads, 6 unstable` (the six: conv/64k5, par/gemm/256, + par/gemm/1024, par/gemm/wide and the kept par/gemm/64 and par/gemv/1024); largest stable change outside the + kept workloads: par/map/4x4 +3.42% + +The tables below come from this run. Medians in seconds per operation; rel. dispersion base / head in %; label +`stable` when both are at most 5% (D-035). + +## Kept optimizations + +`gemm-blocked` (D-036): `operator*=`, `operator*` and `direct_multiply` use one cache-blocked row-major i-k-j +kernel (`matrix_details::gemm_blocked`) in serial and parallel builds, in place of the strided +`std::inner_product` per entry and, in serial builds, Strassen recursion for max dimension ≥ 17. Each entry is +the k-ascending sum from zero, bit-identical to the old kernel kept as `matrix_details::gemm_reference` +(`[S10][S10-R3]` tests). + +| id | workload | baseline median | head median | change % | dispersion % | label | +|---|---|---|---|---|---|---| +| gemm-blocked | gemm/8 | 2.5209e-07 | 1.5264e-07 | -39.45 | 1.34 / 2.07 | stable | +| gemm-blocked | gemm/16 | 1.5914e-06 | 9.7036e-07 | -39.02 | 0.38 / 0.40 | stable | +| gemm-blocked | gemm/17 | 3.4108e-06 | 1.1400e-06 | -66.58 | 0.81 / 0.39 | stable | +| gemm-blocked | gemm/18 | 6.5067e-06 | 2.2080e-06 | -66.07 | 0.67 / 0.55 | stable | +| gemm-blocked | gemm/32 | 1.4983e-05 | 7.2298e-06 | -51.75 | 0.24 / 0.42 | stable | +| gemm-blocked | gemm/64 | 1.7852e-04 | 5.7231e-05 | -67.94 | 0.92 / 0.19 | stable | +| gemm-blocked | gemm/128 | 1.2685e-03 | 4.5228e-04 | -64.34 | 0.57 / 0.18 | stable | +| gemm-blocked | gemm/256 | 8.4660e-03 | 3.9624e-03 | -53.20 | 0.26 / 0.48 | stable | +| gemm-blocked | gemm/512 | 5.7637e-02 | 3.0558e-02 | -46.98 | 0.51 / 0.36 | stable | +| gemm-blocked | gemm/1024 | 3.9512e-01 | 2.3764e-01 | -39.86 | 0.61 / 0.75 | stable | +| gemm-blocked | gemm/tall | 7.5342e-02 | 4.9836e-03 | -93.39 | 1.64 / 1.75 | stable | +| gemm-blocked | gemm/wide | 6.2296e-02 | 4.0113e-03 | -93.56 | 1.10 / 2.94 | stable | +| gemm-blocked | gemv/8 | 5.5061e-08 | 3.8638e-08 | -29.83 | 1.79 / 0.96 | stable | +| gemm-blocked | gemv/16 | 1.4597e-07 | 8.6724e-08 | -40.59 | 1.85 / 1.93 | stable | +| gemm-blocked | gemv/32 | 4.7085e-07 | 2.6251e-07 | -44.25 | 0.94 / 1.56 | stable | +| gemm-blocked | gemv/64 | 2.1021e-06 | 1.1171e-06 | -46.86 | 0.20 / 0.10 | stable | +| gemm-blocked | gemv/128 | 9.3135e-06 | 4.6953e-06 | -49.59 | 0.21 / 0.22 | stable | +| gemm-blocked | gemv/256 | 4.3272e-05 | 1.9614e-05 | -54.67 | 0.29 / 0.37 | stable | +| gemm-blocked | gemv/512 | 1.8907e-04 | 8.6784e-05 | -54.10 | 0.41 / 1.09 | stable | +| gemm-blocked | gemv/1024 | 7.8252e-04 | 3.4892e-04 | -55.41 | 0.47 / 1.40 | stable | + +### par-thresholds + +`par-thresholds` (S10-T7): in `FENG_MATRIX_PARALLEL` builds the default worker count is work-based, +`matrix_details::work_workers( work, grain )` = min( hardware_concurrency, work / grain ), at least 1, in place of +`hardware_concurrency` workers for every elementwise call (threshold 0), for `for_each` above 1024 elements and for +reductions above 32 elements. Grains: 2^16 elements for elementwise calls (`+=`, `-=`, `+`, `-`, unary minus, +`copy`, `clone`, `meshgrid`, `save_as_bmp` rows), callbacks (`for_each`, the maps, `apply`, colormaps, `pooling`) +and reductions (`sum`, `min`, `max`, `reduce`); 2^18 multiply-adds (M·K·N) for products, whose rows are split over +at most M workers. Explicit worker counts and serial builds are unchanged; the 1-worker run is the fallback, +and `[S10][S10-R3]` "thresholded parallel helpers equal the serial path" shows elementwise results bit-identical +to it, reductions equal to `reduce_range` with the same worker count and products bit-identical to +`gemm_reference`. + +How the grains were chosen: a scratch program timed `parallel_workers`, `reduce_range` and `gemm_blocked` at +1, 2, 3, 4, 6, 8, 12, 16 and 24 workers (median of 15 batches each, warm inputs, -O2). Starting and joining one +`std::jthread` costs about 19 µs, so at 24 workers any call costs about 0.45 ms. One worker is fastest up to +about 2^15 elements for sqrt via `for_each` (54 µs vs 58 µs on 2), 2^16 for sum (38 µs vs 49 µs) and 2^18 +multiply-adds for GEMM (64^3: 51 µs vs 58 µs on 3). At 2^17 elements sqrt is 211 µs on 1 worker and 111 µs on 4; +96^3 GEMM is 171 µs on 1 and 105 µs on 4. The elementwise grain comes from the workload itself: par/add/1000x1000 +(a + b into a fresh result), three interleaved runs each, gave 0.75–0.89 ms at grain 2^16 (15 workers), +0.84–0.90 ms at 2^17 (7), 0.98–1.06 ms at 2^18 (3) and 0.98–1.06 ms at the baseline's 24. For par/map/1000x1000 a +callback grain of 2^16 (15 workers, 0.92–0.95 ms) beat 2^13–2^15 (24 workers, 1.00–1.09 ms). + +par/gemm/256, par/gemm/1024, tall and wide are not listed: their worker count is 24 in both builds, so their gains +come from `gemm-blocked`. par/map/1000x1000 is closest to the limit (−14.00 %). par/gemm/64 and par/gemv/1024 are +unstable in this run (baseline dispersion 5.25 %); both are far beyond the 10 % bar. + +| id | workload | baseline median | head median | change % | dispersion % | label | +|---|---|---|---|---|---|---| +| par-thresholds | par/add/4x4 | 3.0267e-04 | 1.6460e-08 | -99.99 | 4.08 / 1.24 | stable | +| par-thresholds | par/add/32x32 | 4.7099e-04 | 2.7403e-07 | -99.94 | 2.50 / 1.25 | stable | +| par-thresholds | par/add/100x100 | 4.7990e-04 | 3.2316e-06 | -99.33 | 2.96 / 1.79 | stable | +| par-thresholds | par/add/1000x1000 | 8.4921e-04 | 6.7131e-04 | -20.95 | 2.16 / 2.08 | stable | +| par-thresholds | par/map/100x100 | 4.7906e-04 | 1.6933e-05 | -96.47 | 2.07 / 0.71 | stable | +| par-thresholds | par/map/1000x1000 | 9.6448e-04 | 8.2944e-04 | -14.00 | 1.82 / 1.20 | stable | +| par-thresholds | par/sum/32x32 | 4.6266e-04 | 5.5930e-07 | -99.88 | 3.38 / 0.70 | stable | +| par-thresholds | par/sum/100x100 | 4.5574e-04 | 5.8643e-06 | -98.71 | 2.82 / 1.35 | stable | +| par-thresholds | par/sum/1000x1000 | 4.7748e-04 | 3.0659e-04 | -35.79 | 1.81 / 1.94 | stable | +| par-thresholds | par/gemm/8 | 1.3027e-04 | 1.3194e-07 | -99.90 | 1.76 / 1.02 | stable | +| par-thresholds | par/gemm/16 | 2.8894e-04 | 8.9055e-07 | -99.69 | 0.98 / 0.81 | stable | +| par-thresholds | par/gemm/17 | 3.0637e-04 | 1.0283e-06 | -99.66 | 1.93 / 1.13 | stable | +| par-thresholds | par/gemm/18 | 3.3770e-04 | 1.1884e-06 | -99.65 | 2.70 / 1.35 | stable | +| par-thresholds | par/gemm/32 | 4.8804e-04 | 6.6837e-06 | -98.63 | 4.43 / 1.13 | stable | +| par-thresholds | par/gemm/64 | 5.0838e-04 | 5.2863e-05 | -89.60 | 5.25 / 1.06 | unstable | +| par-thresholds | par/gemv/8 | 1.3267e-04 | 3.3619e-08 | -99.97 | 1.72 / 2.87 | stable | +| par-thresholds | par/gemv/32 | 4.8237e-04 | 2.5033e-07 | -99.95 | 4.78 / 1.64 | stable | +| par-thresholds | par/gemv/64 | 4.8741e-04 | 1.0649e-06 | -99.78 | 4.60 / 1.66 | stable | +| par-thresholds | par/gemv/1024 | 9.5979e-04 | 5.7155e-04 | -40.45 | 5.25 / 3.93 | unstable | + +Rows of this run outside the kept workloads with change ≥ +2 %, all stable and in code paths the kept changes +leave serial or untouched: par/map/4x4 +3.42 % (16 elements, one worker in both builds), map/1000x1000 +2.31 %, +scale/4x4 +2.08 %. In the run before `par-thresholds` par/map/4x4 was +1.54 %; these rows move by a few percent +between runs (D-035). + +### fft-plans + +`fft-plans` (S10-T8, B-042): `matrix_details::fft2` builds one `fft_plan` per length per call and reuses it for +every row (length C) and every column (length R; the row plan when R == C). A radix-2 plan holds the twiddle +table for the sign used; a Bluestein plan for length n holds the chirp w (n), the forward radix-2 transform of b +(length M) and the length-M twiddle tables for -1 and +1, plus the length-M scratch. The per-row path that +rebuilt all of these for each row and column (three table builds and the chirp transform per Bluestein call) is +kept as `matrix_details::fft2_reference`; every plan value comes from the same `std::polar` expressions in the +same order, and `[S10][S10-R3]` "fft with reused plans equals the per-row transform" shows fft and ifft +bit-identical to it (int, float, double, complex; 0x0 to 250x3). + +| id | workload | baseline median | head median | change % | dispersion % | label | +|---|---|---|---|---|---|---| +| fft-plans | fft/128 | 3.4088e-04 | 2.0700e-04 | -39.28 | 1.67 / 2.14 | stable | +| fft-plans | fft/256 | 1.6610e-03 | 1.1714e-03 | -29.48 | 1.20 / 0.44 | stable | +| fft-plans | fft/125 | 2.4250e-03 | 8.7432e-04 | -63.95 | 0.87 / 3.62 | stable | +| fft-plans | fft/250 | 1.0221e-02 | 3.8516e-03 | -62.32 | 0.80 / 3.08 | stable | + +## Parallel variant + +The lane also builds both headers with `-std=c++20 -O2 -pthread -DFENG_MATRIX_PARALLEL` (`base-par`, `head-par`). +Their workloads are named `par/`: elementwise (add, scale, map) and statistics (sum, mean, variance, +max) at all four sizes, gemm 8/16/17/18/32/64/256/1024/tall/wide and gemv 8/32/64/1024. They use the same +inputs, interleave and rules as the serial rows but run unpinned (they need many cores). Starting point for the +threshold work (before `par-thresholds`), from one full run (head 9f75ba6 plus this lane, baseline 78dc886; `LANE bench PASS: 1 kept, 102 +workloads, 5 unstable`). Medians in seconds per operation; rel. dispersion base / head in %. + +| workload | baseline median | head median | change % | dispersion % | +|---|---|---|---|---| +| par/add/4x4 | 3.1543e-04 | 3.1329e-04 | -0.68 | 5.40 / 6.26 | +| par/scale/4x4 | 1.4137e-08 | 1.3950e-08 | -1.32 | 3.31 / 1.05 | +| par/map/4x4 | 2.7156e-08 | 2.7574e-08 | +1.54 | 2.10 / 1.34 | +| par/sum/4x4 | 4.4818e-09 | 4.2869e-09 | -4.35 | 1.81 / 1.39 | +| par/add/32x32 | 4.6794e-04 | 4.6743e-04 | -0.11 | 3.19 / 3.16 | +| par/sum/32x32 | 4.5935e-04 | 4.5610e-04 | -0.71 | 2.61 / 1.77 | +| par/add/100x100 | 4.7078e-04 | 4.6572e-04 | -1.07 | 1.91 / 2.21 | +| par/map/100x100 | 4.7225e-04 | 4.7453e-04 | +0.48 | 2.55 / 1.56 | +| par/add/1000x1000 | 9.4195e-04 | 9.7350e-04 | +3.35 | 7.26 / 6.49 | +| par/gemm/8 | 1.3303e-04 | 1.3789e-04 | +3.65 | 3.42 / 6.01 | +| par/gemm/64 | 4.9291e-04 | 4.7936e-04 | -2.75 | 3.51 / 2.76 | +| par/gemm/256 | 1.5610e-03 | 7.3545e-04 | -52.89 | 3.05 / 1.09 | +| par/gemm/1024 | 2.1967e-01 | 2.4593e-02 | -88.80 | 4.87 / 7.57 | +| par/gemv/8 | 1.3303e-04 | 1.3271e-04 | -0.24 | 3.09 / 3.54 | + +What the numbers show: a call that reaches the parallel loop costs 0.13–0.5 ms however small it is (add at every +size, map from 100x100, sum from 32x32, gemm and gemv at every size up to 64; par/add/4x4 is 3.1e-04 s, serial +add/32x32 3.2e-07 s against par/add/32x32 4.7e-04 s); scale, mean, variance, max and the 4x4 statistics stay at +serial cost. In the parallel build the head GEMM gains 53% at 256 and 89% at 1024. The full table is in the lane +output (`build/logs/bench/compare.log`). These are the numbers before `par-thresholds` (above), which is now +listed in `bench/kept.txt`. + +## Rejected candidates + +None yet. + +## Unstable workloads + +From the full run above, not judged (rel. dispersion base / head): + +- conv/64k5: baseline 4.2579e-05 s, head 4.0150e-05 s, change −5.70 %, 10.07 % / 5.13 % +- par/gemm/256: baseline 1.9032e-03 s, head 7.7015e-04 s, change −59.53 %, 14.46 % / 3.20 % +- par/gemm/1024: baseline 2.3209e-01 s, head 2.7134e-02 s, change −88.31 %, 7.17 % / 10.15 % +- par/gemm/wide: baseline 9.7876e-03 s, head 8.5125e-04 s, change −91.30 %, 5.52 % / 2.59 % +- par/gemm/64 and par/gemv/1024 (kept, `par-thresholds` table above) diff --git a/build/.gitkeep b/build/.gitkeep new file mode 100644 index 0000000..e69de29 diff --git a/docs/modern_cpp_optimization_plan.md b/docs/modern_cpp_optimization_plan.md new file mode 100644 index 0000000..d13534f --- /dev/null +++ b/docs/modern_cpp_optimization_plan.md @@ -0,0 +1,192 @@ +# Matrix library review and executable modernization plan + +**Recommendation:** repair storage, bounds, parsing, and numerical contracts before optimizing kernels or raising the language requirement. Keep a C++20 compatibility path during the repair release, introduce C++23 vocabulary types through adapters, and qualify C++26 features individually. The existing test suite is useful but does not justify a production-safety claim. + +Prepared **2026-10-01** against commit **`8f5489966f94c3d9356697312248ec4d86d663e8`**. The reviewed [matrix.hpp](../matrix.hpp) has 7,688 lines and SHA-256 `4840a67d1484d184f7cfba99f7b9555fbc1ae3560a5638503e56b649ea8976fa`. Line references below belong to that snapshot. Inputs include the [July review](opencode_sharded_review.md), [ReadMe](../ReadMe.md), [tests](../tests/test.cc), [Makefile](../Makefile), and GitHub workflows. Only this report is delivered; no library, test, example, or build source was changed or added. + +“Bug-free” should be an engineering objective with a release gate: **zero unresolved known correctness or memory-safety defects in the supported API, no findings in the specified analysis/test campaigns, and passing downstream acceptance tests**. Neither this review, sanitizers, nor adopting C++26 can prove that arbitrary C++ programs are free of bugs. Unsupported types, formats, and operations must fail explicitly instead of remaining accidental template instantiations. + +**Evidence and scope.** This is a source review plus builds and execution of existing tests. No new C++ probes or regression tests were written, respecting the research-only request. Findings marked **S** follow directly from source; **E** means observed in this review; **H** means empirical evidence recorded in the earlier review, not reproduced here. Proposed regression inputs are specifications for the implementation phase, not claims of execution. Numerical algorithms, optional integrations, and custom allocators are not exhaustively verified. + +| Check performed | Result and limits | +|---|---| +| GCC 16.2.1, existing suite, C++20, `-O2 -DPARALLEL -pthread` | **E: passed**, 49,216,592 assertions in 57 test cases. Uses ordinary floating-point optimization, rather than the Makefile's `-Ofast`. | +| GCC 16.2.1, existing suite, C++20, `-O1 -DNDEBUG -DPARALLEL -fsanitize=address,undefined -fno-omit-frame-pointer -pthread` | **E: passed**, 49,216,592 assertions in 57 cases; no sanitizer diagnostic in the captured run, with leak detection enabled. This covers only existing test paths. | +| Clang 22.1.8, existing suite, C++20, `-fsyntax-only -DPARALLEL -pthread` | **E: failed**, six incomplete-template errors at lines 3366–3368 and 3406–3408, plus two warnings. | +| Clang 22.1.8, ASan/UBSan build, C++20 release-check configuration | **E: compilation failed** at the same six locations; no Clang sanitizer executable ran. | +| GCC 16.2.1, existing suite, `-std=c++26 -fsyntax-only -DPARALLEL -pthread` | **E: passed syntax check**, with deprecated literal-operator spelling warnings at line 362. Does not establish C++26 library-feature availability or runtime correctness. | +| New defect probes, fuzzing, TSan, MSan, MSVC, optional OpenCV build, downstream projects, benchmarks | **Not run.** These require future harnesses, platforms, dependencies, or consumer access. No performance gain is claimed. | + +The exact commands and decisive output are recorded at the end. The large assertion count mainly reflects loops over existing happy paths; it is not an API-coverage measure. + +**The July findings still require action, but several recommendations need correction.** The current source retains the important defect patterns. The earlier report has some inconsistent line numbers and finding labels, so use the symbols and the fresh references here when implementing. + +| Earlier finding | Current assessment and implementation consequence | +|---|---| +| C1, C2: crop/pad and flip memory corruption | Confirmed in source at 3518–3535 and 4451–4486; historical ASan evidence remains relevant. Fix before structural refactoring. | +| C3: flipped aliases | Confirmed at 4491–4499. Correct axis semantics with migration notes. | +| C4, R1: pseudoinverse and SVD naming | `pinverse` still multiplies by singular values. `svd_inverse` compensates for its swapped output-variable names in subsequent expressions, so the naming alone is not a separate numerical bug. It still ignores SVD failure and leaves small nonzero singular values uninverted rather than setting them to zero. Do **not** use it as an already-qualified replacement. | +| C5, P2: determinant and decomposition | Confirmed; the same block-inversion assumption also breaks `inverse` for some invertible matrices. Unpivoted LU is not a safe replacement. A singular matrix has determinant zero, not automatically an error or NaN. | +| C6: matrix power | The bad expression is inside an ordinary runtime `if`. The entire function specialization must compile, including that branch, for **any exponent**, not only odd exponents. Test 0, 1, 2, 3, and larger powers. | +| C7 / S1 / S2: assertions, NPY, PNG | Confirmed and expanded below. With `NDEBUG`, `better_assert` still evaluates its expression but the failure reporter does nothing. Column indexing through `m[r][c]` is unchecked even in debug builds. | +| C8, C9, C10: statistics, convolution, RREF | More than local assertion fixes are needed. Statistics have type and denominator inconsistencies; full convolution also has an orientation bug; RREF needs independent pivot-row and pivot-column tracking for rank-deficient or tall inputs. Merely relaxing `row < col` can expose out-of-bounds access. | +| C11: global random state | Global reseeding is confirmed. A data race is not proven on every implementation; thread-safety of the C random facility is implementation-dependent. Use caller-owned engines for isolation and reproducibility. | +| C12: zero hardware concurrency | The iterator-based reduction at 1146–1177 has the zero-divisor path. The separate reduction at 4036–4040 already guards values ≤1; the earlier report's proposed second zero guard was misplaced. That second helper has other defects. | +| C13, P1: Fourier transforms | Correct total work is Θ(R²C²), or Θ(N⁴) for an N×N image, not Θ(R⁴C⁴). At 256×256 this is about 4.29 billion innermost iterations, not trillions. Both transforms also use the wrong input indices. Shift functions perform a transform as well as a shift and mishandle odd sizes. | +| T1, T2 | Coverage gaps remain. Add API-instantiation and adversarial tests, not simply more iterations of existing tests. | +| A1, A2, A3, R2, R3 | Simplify CRTP and boilerplate later. `matrix_view` does reuse some mixins, so “exactly one concrete class” overstates the case. Forwarding aliases and ADL-friendly math names are not inherently defects; preserve compatible aliases and constrain overloads. Top-level `const` value returns can inhibit moves; they are not merely cosmetic. | +| Earlier “verified non-issues” | Do not carry forward a blanket clearance for BMP, full convolution, or pooling. BMP validation is incomplete; full convolution is incorrect for asymmetric inputs; pooling mishandles negative maxima and invalid mode names. Large/nonfinite `expm` inputs also need explicit domain and shift-range tests rather than an “unreachable” assumption. | + +**Release-blocking source findings.** Severity describes potential impact; priority reflects the order of work. P0 covers memory corruption, invalid object state, and unsafe external-input boundaries. P1 covers numerical correctness, portability, and failure handling. Source-derived examples below must become regression tests before their fixes are accepted. + +| ID | Priority / evidence | Location in matrix.hpp | Finding, trigger, and required outcome | +|---|---|---|---| +| F01 | P0 / S,H | `shrink_to_size`, 3518–3535; `flipdim`, 4451–4486 | Crop/pad copies the row count as the number of columns. A 5×5→5×3 resize can write past storage. Dimension-2 flip swaps a column with a row; non-square matrices can corrupt memory. Empty flips also subtract one from a zero unsigned extent. Use correct extents and define empty behavior. | +| F02 | P0 / S | constructors, 3790–3843; `resize`/`reshape`, 2834–2868; public fields, 3780–3783 | Extent multiplication is unchecked. Signed initializer-list dimensions are cast to `unsigned long`; other dimensions convert to an unsigned size type. On this 64-bit target, extents 2⁶³×2 wrap the element count to zero while retaining huge extents. Release `reshape` can claim more elements than allocated. Public `row_`, `col_`, `dat_`, and `allocator_` let callers bypass all invariants. Validate signed values, element/byte counts, allocator limits, and pointer-difference limits before mutation. | +| F03 | P0 / S | `matrix(matrix_view const&)`, 3899–3903 | This constructor does not initialize the scalar storage members before `resize` reads them. It also copies from the first source row to an end in a later row, ignores column offsets, and can overrun the destination. Repair initialization and copy precisely the selected rectangle. | +| F04 | P0 / S | constructors, 3810–3843; `clear`, 1853–1863; `copy`, 1992–1998; `swap`, 3589–3596 | Raw allocation is followed by assignment, without general element construction or destruction. This is not valid support for arbitrary nontrivial `T`; implicit-lifetime rules for suitable scalar types do not justify the generic API. Copy assignment installs the source allocator before releasing destination storage, so unequal stateful allocators can free memory through the wrong resource. Allocator propagation traits and exception guarantees are not honored consistently. | +| F05 | P0 / S | `stride_iterator`, 1192–1310; column ends, 1928–1954; diagonal/anti-diagonal ranges, 1531–1760, 2087–2275 | A column end for column c>0 becomes `data + rows*cols + c`, beyond one-past the array. Similar diagonal ends exceed the allocation. Anti-diagonals of a one-column matrix have zero stride, including division by zero in iterator distance. Represent end positions logically and form pointers only for dereferenceable positions. | +| F06 | P0 / S | `make_view`, 4094–4112; `matrix_view`, 3709–3719; `clone`, 1871–1912 | Views borrow a matrix through `const&` and accept temporary owners. Stored views can therefore dangle. Resizing or destroying the owner invalidates extents/lifetime assumptions. Range normalization silently changes invalid requests; cloning lacks source upper-bound validation. Reject temporary-owner borrowing and validate slices in release builds. | +| F07 | P0 / S,H | `load_npy`, 2508–2564 | Reads the version before checking buffer length, trusts header offsets, never validates magic or dtype, and copies typed values from raw bytes without establishing alignment/representation preconditions. Foreign endianness is ignored; Fortran order changes shape instead of preserving the logical matrix. Malformed numeric fields can throw through `noexcept`. Parse, validate and convert into a temporary before committing. | +| F08 | P0 / S | text/binary loading, 2421–2495; BMP loading, 6643–6689; PNG writer, 3096–3348 | Empty text dereferences `rbegin`; ragged text can be flattened into a different rectangular matrix. Binary loading allocates and changes the destination before validating the payload, and accepts unchecked native object representations. BMP assumes a 54-byte, 24-bit layout without validating signature, offset, compression or signed height; arithmetic can overflow. PNG uses an unchecked `FILE*` and does not report write/close failure. | +| F09 | P0 / S | access, 1796–1821; multi-input maps, 3954–3980; vector product, 4281–4290; pooling, 6738–6742 | Public shape checks disappear as enforcement in release builds. Left-vector multiplication checks vector length against columns instead of rows: a length-3 vector times a 2×3 matrix passes the debug check and reads a third column element beyond storage. Multi-input internal maps trust the first input's size. Unknown pooling actions dereference `map::end()` after the ineffective release assertion. Validate before traversal. | +| F10 | P1 / E,S | image mixins, 3366–3368, 3406–3408; declaration 137; definition 3723 onward | Clang rejects nondependent local `matrix` objects while the class is incomplete. GCC acceptance does not make the header portable. Move these method definitions after the complete type or separate the image layer. | +| F11 | P1 / S | `noexcept` constructors and operations throughout; `parallel`, 267–326 | Allocation, I/O, parsing, user callbacks and thread creation can throw through unconditional `noexcept`. An escaping worker exception terminates the process; failure partway through creating a vector of joinable threads is also unsafe. Specify exception guarantees, join workers on every exit, and propagate errors after joining. | +| F12 | P1 / S | `parallel`, 317–322; reductions, 1146–1177 and 4024–4062; `map`, 4013–4018 | The final parallel chunk omits `dim_first`, so nonzero-origin ranges can overlap or execute outside the requested range. The iterator reduction divides by a possibly zero core count. Curried `reduce` captures its by-value `init` parameter by reference in the returned lambda, which dangles immediately; curried `map` can retain a temporary callable by reference. The second reduction repeats `init` per worker and can form invalid pointers when workers exceed elements. These internal helpers are not all used by the current public reductions. | +| F13 | P1 / S,H | determinant, 2055–2082; inverse, 2370–2407; LU, 6499–6530; SVD, 4921–5236 | Block determinant/inverse assumes invertible subblocks. The invertible 4×4 block-exchange matrix with zero 2×2 diagonal blocks is a counterexample. LU lacks pivoting and a final-pivot singularity check. `pinverse` omits reciprocation; both inverse wrappers ignore convergence failure. Require reliable factorization, singular/rank policy, and residual checks. | +| F14 | P1 / S | `fft`/`ifft`, 6313–6337, 6446–6476; shifts, 6349–6366, 6480–6496 | Inner sums use `x[r][c]`, not `x[r_][c_]`; an off-origin impulse therefore does not transform correctly. The inverse has no normalization. `fftshift`/`ifftshift` unexpectedly compute transforms and use incompatible odd-size swaps. Specify transform sign, scaling, axes, and pure permutation semantics before optimizing. | +| F15 | P1 / S | `conv`, 6573–6634; RREF, 6393–6429; statistics, 7634–7662; maxima, 3635–3675, 6720–6726 | Convolution does not reverse the kernel. Source tracing for 1×2 inputs [1,2] and [3,4] gives [6,11,4], whereas convolution is [3,10,8]. Maxima initialize with floating `min()` (smallest positive normal), so all-negative maxima are wrong. Integer mean truncates, and signed sums can convert to unsigned during division. Variance uses N while standard deviation uses N−1; integral instantiations also need return-type compilation checks. | +| F16 | P1 / S,H | matrix power, 5555–5570; valarray product, 4257–4265; clone overload, 1871–1876; OpenCV, 1432 onward | Public templates contain latent compile errors: power precedence, nonexistent `valarray::row()`, and indexing `initializer_list` with `[]`. The OpenCV branch calls `isContiguous()`; the documented `cv::Mat` API is `isContinuous()`. The OpenCV branch was not built here. Add explicit API instantiation coverage and adapter tests. | +| F17 | P1 / S | random, 5242–5253; scalar and mixed arithmetic throughout | Random generation reseeds global state, uses a floating formula for unconstrained types, and does not establish its claimed open interval across rounding/types. Mixed operations often keep the left element type, risking narrowing; signed element arithmetic can overflow independently of shape safety. Define supported scalar domains and promotion/overflow behavior. | +| F18 | P1 / S | Makefile; `.github/workflows/ci.yml`; header configuration, 4–66 | CI selects GCC 12 and `PARALLEL` only. Makefile uses `-Ofast` and `-march=native`; the former can invalidate NaN/Inf-based diagnostics, and the latter is inappropriate for portable release qualification. Macro-dependent definitions risk ODR violations if translation units disagree. The C++20 version check uses the draft value 201709L rather than the final 202002L, and the version constant remains 20240314. | + +F05 is a language-level issue even without dereferencing the end pointer: pointer arithmetic is limited to an array and its one-past position. Sanitizers do not necessarily report every invalid pointer formation. [C++ working draft, pointer addition](https://eel.is/c++draft/expr.add). + +For F07, dtype, byte order, shape, and Fortran layout are actual format metadata, not optional hints; rank-2 support can be deliberately narrower than NumPy if rejected cases are documented. [NumPy NPY specification](https://numpy.org/doc/stable/reference/generated/numpy.lib.format.html). F14/F15 use the conventional NumPy/SciPy contracts as the proposed future behavior, with a migration path for existing consumers. [Inverse FFT](https://numpy.org/doc/stable/reference/generated/numpy.fft.ifft2.html), [frequency shift](https://numpy.org/doc/stable/reference/generated/numpy.fft.fftshift.html), [2D convolution](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.convolve2d.html). F16's OpenCV comparison is against the official [cv::Mat reference](https://docs.opencv.org/4.x/d3/d63/classcv_1_1Mat.html). + +**Target design and contracts.** Keep the recognizable `feng::matrix` facade while shrinking the trusted implementation. Separate ownership, borrowing, algorithms, external-data parsing, and optional acceleration. The relevant Core Guidelines principles are RAII (R.1), avoiding explicit allocation management where possible (R.11), expressing invariants in the type (C.2), and using `noexcept` only where failure cannot escape (E.12). These principles guide the following project-specific decisions. [C++ Core Guidelines](https://isocpp.github.io/CppCoreGuidelines/CppCoreGuidelines). + +| Area | Recommended decision | Compatibility and safety requirement | +|---|---|---| +| Ownership | Private contiguous storage, preferably `std::vector`, plus validated extents. Use a bespoke allocator-aware buffer only if a measured requirement rules out vector. | Preserve row-major layout and allocator template parameter. Correct construction/destruction and allocator propagation are required. Exclude `bool` from a contiguous `T*` model or introduce a separately specified mask type; `vector` is not suitable storage. | +| Object invariants | Storage size equals checked rows×columns; extents and offsets fit the index/difference domain. Allow 0×N and N×0 shapes, with no accessible elements. | A moved-from object must remain consistent. Defaulting all moves after adding a vector can leave nonzero extents with moved-away storage; explicitly preserve the invariant. Assignment must commit metadata and storage together. | +| Checked access | Provide always-checked `at(r,c)` and checked slicing; make the safe facade's two-index access validate both coordinates. Validate shapes once before trusted internal loops. | Retain raw `data()` and legacy row-pointer access as documented escape hatches during migration. Safe access cannot be promised through arbitrary pointer arithmetic. Changing `m[r]` to a checked proxy is a versioned API change. | +| Views | Explicit mutable and const rank-2 views with extents and physical row stride. Add span for contiguous rows/storage and mdspan adapters. | Borrow only from lvalues; reject temporary owners. Document invalidation by destruction, reallocation, assignment and shape changes. Do not mark the owning matrix as a borrowed range. A view is not a lifetime guarantee. | +| Strided iteration | Store base, stride, logical index and count; derive addresses only for valid dereferences. | End/sentinel values must not require an out-of-array pointer. For subviews, physical row stride is the parent's width, not the subview's width. Test reverse iteration and standard iterator concepts. | +| Errors | Use exceptions for invalid construction/access and allocation failure; use structured recoverable results for parsing and numerical failure. | Suggested categories: invalid shape, overflow, unsupported dtype/format, truncation, I/O failure, singularity, nonconvergence. Preserve old bool/optional entry points as adapters. Remove false `noexcept`; `expected` does not itself make allocation nonthrowing. | +| Numeric types | Support integer storage/basic arithmetic and real/complex floating algorithms explicitly. Introduce operation-specific concepts and a documented promotion policy. | Do not constrain all algorithms to `std::floating_point`, which excludes complex. Inverse/decomposition on integers must be rejected or explicitly promoted. Checked integer operations are needed wherever the API promises overflow detection; otherwise document the representable-input precondition. | +| Statistics | Promote integral mean/variance to a floating result; use a stable accumulation algorithm and a shared degrees-of-freedom parameter. | Make population/sample selection explicit; empty mean/min/max and insufficient sample count return a defined error. Decide NaN handling and complex variance as separate contracts. | +| Numerics | Factorization objects carry pivots, rank/status and reusable factors; solve systems from factors. Implement determinant from pivot parity and U's diagonal. | For complex algorithms use conjugate transpose where required. Use relative, scale-aware tolerances, not one universal 1e−10 cutoff. Distinguish singularity from nonconvergence and nonfinite input. | +| Execution | Serial is the default for callbacks; parallel numeric operations opt in through a policy with bounded workers. | Guarantee every scheduled index is visited exactly once, join on exceptions, avoid nested backend oversubscription, document reduction ordering. `jthread` helps lifetime management but does not transport exceptions automatically. | +| Public API and packaging | Keep compatibility aliases forwarding to one implementation. Split maintained headers by responsibility and generate a single-header distribution. | Preserve simple inclusion. Name configuration macros with a project prefix and require one consistent configuration per program. Test generated and maintained headers against the same suite. | + +Adopt value returns without top-level `const`, `[[nodiscard]]` on results that must be checked, explicit conversions, and constrained overloads. Replace repetitive elementwise bodies with a small audited transform layer, preserving operation-specific result types. Prefer standard algorithms/ranges when they simplify traversal; do not introduce lazy expressions whose lifetimes are harder to prove. Directly include the standard headers actually used, including facilities currently obtained transitively. These changes follow the source defects and should not become a broad style rewrite in the safety release. + +**C++20, C++23 and C++26 adoption.** The core safety work does not need C++26. Freeze the minimum supported compiler/library pairs after the consumer inventory in WP01; the working recommendation is a C++20 maintenance line and an opt-in modern interface, followed by a C++23 baseline only when consumers can build it. Compiler language mode and standard-library support are separate capabilities. The official GCC page still describes experimental aspects of its C++26 support; vendor support tables are moving evidence, not a deployment guarantee. [GCC language support](https://gcc.gnu.org/projects/cxx-status.html), [libstdc++ library status](https://gcc.gnu.org/onlinedocs/libstdc++/manual/status.html), [libc++ C++26 status](https://libcxx.llvm.org/Status/Cxx26.html). + +| Facility | Adoption decision | Qualification gate | +|---|---|---| +| C++20 concepts, ranges, span, jthread, numbers | Use immediately where they remove bespoke machinery or clarify requirements. The project already uses concepts/ranges. | Compile the maintained C++20 path; retain explicit shape validation and worker-exception handling. | +| C++23 `std::expected` | Use for new parser/factorization APIs when available; expose an explicit adapter on C++20 rather than changing one public function's return type invisibly with language mode. | Check `__cpp_lib_expected`, compile the exact operations used, and test error propagation. [WG21 expected design](https://www.open-std.org/jtc1/sc22/wg21/docs/papers/2022/p0323r12.html). | +| C++23 `std::mdspan` | Use as the interoperability view for contiguous and strided matrices. It replaces neither ownership nor runtime input validation. | Check `__cpp_lib_mdspan`, test constness/layout/strides, and verify the backing storage remains alive. If needed, use an explicitly packaged backport behind the adapter. [WG21 mdspan](https://www.open-std.org/jtc1/sc22/wg21/docs/papers/2022/p0009r18.html), [Kokkos reference implementation](https://github.com/kokkos/mdspan). | +| C++26 `submdspan` and padded layouts | Prefer for slicing and padded backend buffers once the exact implementation is available. | Check `__cpp_lib_submdspan` and required layout support separately; verify rectangular offsets and required storage span. Keep the tested fallback. [WG21 submdspan](https://www.open-std.org/jtc1/sc22/wg21/docs/papers/2023/p2630r4.html), [padded layouts](https://open-std.org/JTC1/SC22/WG21/docs/papers/2023/p2642r5.html). | +| C++26 `` | Optional backend for BLAS-like operations on mdspan. It is not a replacement for the owning matrix, LU/SVD implementations, or FFT. | Check `__cpp_lib_linalg` and compile/link/run every selected operation; verify backend dispatch and benchmark it. [WG21 BLAS-based interface](https://www.open-std.org/jtc1/sc22/wg21/docs/papers/2023/p1673r13.html). | +| C++26 `` | Evaluate for contiguous elementwise kernels and reductions after scalar correctness and benchmarks exist. | Check `__cpp_lib_simd`, the exact supported revision/API, alignment and tail handling. Do not assume historical proposal spellings match the deployed API. Keep ABI-neutral dispatch and a scalar fallback. [WG21 SIMD proposal](https://www.open-std.org/jtc1/sc22/wg21/docs/papers/2024/p1928r15.pdf), [current draft SIMD interface](https://eel.is/c++draft/simd). | +| C++26 contracts | Optional expression of developer preconditions and invariants. Keep ordinary validation for files, dimensions and recoverable numerical errors. | Check `__cpp_contracts` and the selected evaluation semantics. Assertions that may be ignored or merely observed cannot be the sole barrier before unsafe access. Test with enforcement disabled as well. [GCC contracts support entry](https://gcc.gnu.org/projects/cxx-status.html). | +| C++26 library hardening | Enable a supported vendor hardening mode for qualification and production where appropriate. | Verify that the mode is active and what it checks. It does not repair dangling views, raw pointer writes, malformed formats, or incorrect algorithms. C++26 mode alone does not promise these protections. [WG21 hardening](https://www.open-std.org/jtc1/sc22/wg21/docs/papers/2025/p3471r4.html), [libc++ hardening modes](https://libcxx.llvm.org/Hardening.html). | +| Reflection, modules, newer execution frameworks | Defer unless a measured consumer need appears. | Reflection does not justify complicating forty wrappers; modules need a demonstrated build-time benefit and packaging support. A safety-profile or borrow-checking proposal is not a portable memory-safety guarantee supplied by C++26. | + +Use both feature-test macros and actual compile/link/runtime probes. Do not select capabilities solely from `__cplusplus` or a compiler version. Record the compiler, standard-library vendor/version, feature macro values, backend version, flags, and architecture in CI artifacts. The draft links above are research references; pin the applicable standard/proposal revision when implementing because the live working draft can move beyond C++26. + +**Dependency choices.** Preserve a standard-library-only storage and basic arithmetic core. Use mature libraries where they reduce the numerical maintenance burden, with adapters and pinned qualification versions. + +| Component | Recommended first choice | Decision boundary | +|---|---|---| +| Advanced decompositions | Evaluate Eigen as an optional header-only adapter and independent test oracle; use BLAS/LAPACK for consumers that already accept linked numerical dependencies. | Select after checking consumer packaging, license, supported scalar types and required sizes. Retain an audited pivoted-LU fallback if dependency-free solving is required. Prefer disabling an unqualified advanced operation to silently using a known-wrong fallback. [Eigen decomposition catalogue](https://libeigen.gitlab.io/eigen/docs-nightly/group__TopicLinearAlgebraDecompositions.html), [LAPACK factorization/solve model](https://www.netlib.org/lapack/lug/node38.html). | +| FFT | Evaluate pocketfft through a narrow optional adapter. | Validate normalization, arbitrary sizes, strides, complex types and thread policy. Keep a tiny correct direct DFT only as a test oracle or explicitly named small-input fallback. Confirm the chosen version's license and packaging. [Pocketfft project](https://github.com/mreineck/pocketfft). | +| Image I/O | Isolate existing codecs immediately; then evaluate a maintained codec dependency based on actual consumer formats. | Do not make image support a mandatory dependency of all matrix users. Replacing a parser does not remove the need for input limits and malformed-file tests. | +| Testing and packaging | Update the vendored Catch v2.0.1 through a separate test-only change; introduce CMake interface targets/CTest while retaining a Makefile convenience entry point. | Pin test dependencies and release artifacts. A build-system migration must preserve simple single-header consumers and must not inject `-Ofast` or `-march=native` into their targets. | + +**Executable work queue.** Open one issue per work package and split it into reviewable pull requests at the listed boundaries. Roles are suggested responsibilities, not assigned people: **L** library maintainer, **Q** test/build engineer, **N** numerical reviewer, **C** downstream project owner. Estimates are engineering days including review/tests, assume one primary implementer with specialist review, and are planning ranges rather than promises. No implementation is authorized by this report itself. + +| Package | Priority / owner / effort | Depends on | Concrete deliverable and acceptance criteria | +|---|---|---|---| +| **WP01 — consumer and contract inventory** | P0 / L+C / 2–3d | None | Record all consuming repositories, compiler/stdlib pairs, scalar types, allocator use, direct field access, exception policy, file formats, and numerical APIs used. Classify every public family as supported, experimental, deprecated, or removed. Freeze the error/empty/shape/aliasing contracts above and create migration decisions for behavior changes. Acceptance: each production consumer has an owner and a build/test command; unavailable consumers remain an explicit release risk. | +| **WP02 — build and evidence baseline** | P0 / Q+L / 2–4d | None | Fix F10's definition-order errors without changing arithmetic. Add GCC and Clang builds, debug/release, serial/PARALLEL, strict floating-point flags, ASan+UBSan, and public API compile tests. Start consumer-required MSVC coverage. Correct the standard check and direct includes. Acceptance: existing suite builds on selected compilers; all new reproductions fail for the intended reason before their fix. Track unresolved cases individually; no blanket expected-failure suite. | +| **WP03 — immediate bounds repairs** | P0 / L+Q / 3–5d | WP01 contract slice, WP02 | Fix F01/F02/F09: crop/pad counts, both flip axes and aliases, empty cases, extent/byte overflow, reshape invariant, clone source bounds, input shape checks, vector multiplication and pooling mode validation. Acceptance: R01–R03 and relevant R07 cases below pass with and without NDEBUG under ASan/UBSan; validation occurs before allocation or traversal. This package can land as small safety patches before storage replacement. | +| **WP04 — owning storage and exceptions** | P0 / L+Q / 4–7d | WP01, WP02, checked dimensions from WP03 | Replace raw ownership or implement complete allocator-aware lifetime handling; make fields private in the migration release; repair F03/F04. Audit allocation/callback/I/O `noexcept` specifications across callers. Acceptance: R04 passes for supported nontrivial and throwing elements and unequal stateful allocators; failure leaves the documented state, and moved-from extents/storage agree. No wrong-resource deallocation, leaks or unbalanced lifetimes. | +| **WP05 — views and iterators** | P0 / L+Q / 4–6d | WP03, WP04 | Repair logical stride ends, parent-stride view traversal and view-to-owner copying. Reject temporary-owner views; define range validation and invalidation. Repair or remove the unused dangling curried helpers. Acceptance: R05 passes, including compile-negative lifetime cases, nonzero slice offsets, one-column anti-diagonals and reverse iteration. Source review proves end construction stays within pointer rules; sanitizer silence alone is insufficient. | +| **WP06 — external data boundaries** | P0 / L+Q / 5–8d | WP03, WP04 | Make NPY, text, native binary and BMP parsing transactional, bounded and typed; report PNG/stream failures. Keep legacy binary explicitly native-format and restricted to supported trivially copyable scalar representations, or add a versioned portable format. Acceptance: R06 plus parser fuzzing pass; malformed input cannot alter the destination, trigger unbounded allocation, or terminate through `noexcept`. | +| **WP07 — execution and randomness** | P1 / L+Q / 3–5d | WP02, WP04 | Repair range partitioning, zero-core fallback, reduction identity rules and thread cleanup. Propagate worker failures after joining. Introduce caller-owned random engines with explicit real/integer distributions. Acceptance: R07 passes for injected worker counts and failures; TSan has no project findings in the supported concurrency scenarios; existing aliases share the implementation. | +| **WP08 — arithmetic and statistics contracts** | P1 / L+N / 3–5d | WP03–WP05, WP07 | Repair power, valarray/clone instantiation defects, floating maxima, mean promotion and shared variance/stddev degrees of freedom. Specify mixed-type results and integer-overflow expectations. Acceptance: R08 passes for signed/unsigned, float/double and applicable complex operations; unsupported instantiations fail with useful constraints. Negative-only tests must detect the old maxima defect. | +| **WP09 — reliable linear algebra** | P1 / N+L / 6–10d | WP01, WP04, WP05, WP08 | Select and qualify backend/fallback; implement pivoted factorization, solve, determinant and inverse from factors. Qualify SVD/pseudoinverse, Cholesky and RREF separately. Propagate singularity/nonconvergence and remove invalid fixed-subblock assumptions. Acceptance: R09 identities/residuals pass across square/tall/wide/rank-deficient inputs in the supported domains; all retained iterative/eigen routines receive convergence and breakdown tests or are marked experimental. | +| **WP10 — convolution and Fourier semantics** | P1 / N+L / 4–7d | WP03, WP05, WP08 | Correct asymmetric-kernel convolution, same/valid sizing, transform indexing and normalization, and pure shifts. Then integrate the selected FFT adapter. Acceptance: R10 passes against independent fixtures, including prime and odd dimensions, impulses and non-symmetric kernels; complexity/performance evidence accompanies the FFT replacement. | +| **WP11 — modern API and maintainable layout** | P2 / L / 4–6d | WP04–WP10 correctness gates | Consolidate transform helpers, constrain concepts, remove top-level const value returns, add nodiscard results, namespaced configuration, structured errors and mdspan/span adapters. Split maintained headers and generate the umbrella distribution. Acceptance: compatible consumers build; generated header passes the same suite and a multi-translation-unit link test; declared feature combinations are tested. Avoid unrelated renaming. | +| **WP12 — optional adapters and packaging** | P1 for used adapters, otherwise P2 / L+Q / 2–4d | WP02, WP05, WP06, WP11 | Repair OpenCV continuity/layout handling and integer dimension conversion. Add enabled/disabled adapter tests, interface build targets/install/export checks, and release automation that publishes only tested artifacts. Acceptance: non-square multi-channel ROI roundtrips pass; a consumer can build from both installed headers and the generated header. | +| **WP13 — measured optimization** | P2 / L+N / 4–8d | WP07–WP12 | Establish pinned benchmark workloads; optimize allocation counts, matrix multiplication locality, thread thresholds and selected SIMD kernels. Qualify optional BLAS/linalg dispatch. Acceptance: performance policy below passes with numerical and sanitizer gates intact; each enabled optimization has measured benefit and a fallback. | +| **WP14 — downstream migration and release** | P0 release gate / L+Q+C / 3–5d | All applicable packages | Run consumer suites, migration tests, fuzz campaigns, release-mode checks and artifact verification. Publish contracts, deprecations, corrected numerical behavior and residual limitations. Acceptance: all release gates below pass, every production consumer signs off, and rollback to the previous pinned version is documented. | + +The listed packages total approximately **49–83 engineering days** before contingency. Plan 20–30% additional capacity for downstream migration and numerical surprises. The critical dependency chain is WP01/WP02 → WP03/WP04 → WP05/WP06 → numerical qualification → WP14. WP07 and independent test/fixture preparation can overlap when staffing permits. Do not defer P0 repairs until a large modernization branch is ready. WP12 must be brought forward if an affected integration is used in production. + +**Regression specifications for those packages.** Use deterministic hand-checkable cases first, then properties and independent reference fixtures. Test expectations must describe the contract rather than repeat the implementation. Compile-only cases are essential because template bodies absent from the current test suite can remain broken indefinitely. + +| Suite | Inputs and properties | Required result | +|---|---|---| +| **R01: shapes and overflow** | 0×0, 0×5, 5×0, 1×1; negative signed dimensions; near-size-limit extents; element/byte overflow; allocator max-size limits; incompatible reshape; out-of-range row and column. | Empty operations obey documented identities/errors; invalid values fail before allocation/traversal and preserve the destination. Run in release and debug. | +| **R02: crop/pad/flip** | 5×5→5×3, 3×10→5×2, 2×3→4×5; 3×5 and 5×3 labeled grids; square and empty flips; both axis aliases. | Exact expected content/zero padding; applying each flip twice restores input. Bounds repairs do not alter unrelated elements. | +| **R03: multi-input and overlap** | Left vector of length 2 times a 2×3 matrix; invalid length 3; wrong-shaped binary/ternary inputs; self-assignment; overlapping slice copy; self-multiplication. | Valid output shape/content; deterministic rejection of mismatches; an explicit overlap policy, using a temporary where necessary. | +| **R04: ownership/failure injection** | Tracked nontrivial element lifetimes; throwing copies; fail-on-N allocation; allocators with different resource IDs and propagation traits; copy/move/swap/self-move; supported over-aligned types. | Balanced construction/destruction; each resource frees its allocations; no leaks; stated strong/basic guarantees; consistent moved-from objects. | +| **R05: borrowed storage** | Copy an offset 2×3 slice out of a larger matrix; stride-aware rows/columns; every valid diagonal of small rectangular shapes; one-column anti-diagonals; forward/reverse traversal; temporary-owner construction. | Exact logical ranges with valid pointer formation. Temporary-owner borrowing fails to compile. Invalidation rules are documented; diagnostic lifetime tests run under sanitizers where appropriate. | +| **R06: file formats** | Empty/truncated files at each header/payload boundary; huge dimensions and header lengths; bad magic/version/dtype; endian variations; NPY C/Fortran order and unsupported object/structured dtypes; ragged text; BMP unsupported compression/offset/top-down height; write/open/close failure. | Either correct values/shape or a specific error, bounded resource use, and unchanged destination on failure. Reject unsupported cases intentionally. Do not deserialize Python objects. | +| **R07: scheduling/randomness** | Inject worker counts 0,1,2 and greater than element count; nonzero-start ranges; throw in a worker and during thread creation; reductions with nonzero init; stored curried callables; invalid pooling action; simultaneous independent seeded generators. | Every index executes exactly once; no joinable threads escape; errors propagate; init appears exactly once; stored callables remain valid; deterministic engine behavior with documented distribution portability limits. | +| **R08: arithmetic/statistics/API** | Matrix powers 0,1,2,3,13; left/right vector and valarray products; every clone overload; max of −3 and −1; mean of 1 and 2, and −1 and −2; population/sample variance; empty inputs; nonfinite values. | Correct values and result types. Integer and complex support is explicit. Variance and stddev agree for the same degrees of freedom. Each retained overload compiles or is intentionally constrained out. | +| **R09: linear algebra** | Identity, diagonal, row-swap, block-exchange, singular/rank-deficient, nearly singular, tall/wide matrices; float/double and supported complex; non-SPD Cholesky; forced nonconvergence. | Pivoted factorization reconstructs input; solve/inverse residuals meet scale-aware limits; determinant has correct sign and is zero for singular input. Pseudoinverse passes all four Moore–Penrose identities, including conjugate-transpose identities for complex. | +| **R10: signal operations** | Asymmetric convolution [1,2] with [3,4]; 1×1 kernel; all modes; empty/oversized kernel policy; off-origin impulse and sinusoid FFT; 1×N, 3×5, 4×6 and prime-size inputs; odd/even shift roundtrips. | Convolution [3,10,8] for the example; documented output alignment; inverse(transform(x))≈x under selected normalization; shifts only permute values and invert one another. | +| **R11: portability/integration** | One translation unit per public API family; multiple translation units using identical configuration; maintained/generated headers; GCC/Clang/MSVC as supported; C++20/23/26 capability lanes; OpenCV ROI/non-square/channel cases. | No latent template errors, unintended mandatory dependencies or ODR defects; feature-disabled fallbacks pass equivalent tests. Detect/document mixed-configuration misuse rather than relying on linker luck. | + +For well-conditioned floating-point fixtures, start with residual tolerances scaled by dimension, machine epsilon, and operand norms, then justify constants per algorithm. For example, assess a solve using ‖AX−B‖/(‖A‖‖X‖+‖B‖), handling a zero denominator separately. Near-singular tests should check failure/rank behavior and backward error; they should not demand small forward error regardless of condition number. Use independent Eigen/LAPACK/NumPy/SciPy fixtures where suitable, recording versions and conventions. A backend cannot be its own sole oracle. + +WP09 also owns qualification of retained matrix-exponential and eigen/iterative APIs: include zero and diagonal matrix exponentials, extreme finite and nonfinite inputs, scaling-shift limits, eigenpair residuals, iterative-solver breakdown, and forced iteration exhaustion. Any family not qualified by these tests stays explicitly experimental. This closes the earlier review's unsupported clearance of extreme `expm` inputs without asserting a new runtime reproduction. + +**Verification and release gates.** Add the following lanes incrementally in WP02 and make the relevant lanes blocking before WP14: + +1. **Ordinary builds:** every supported compiler/stdlib pair, strict C++ mode, debug and release, serial and parallel paths. Retain a C++20 lane until explicitly retired; C++26-only facilities require their own enabled/disabled tests. Test the Strassen path as well as direct multiplication: the current `PARALLEL` configuration always selects direct multiplication. +2. **Memory/undefined behavior:** ASan+UBSan with debug and NDEBUG, leak checking where supported, and optimized builds without fast-math. Run TSan separately on parallel/lifetime scenarios; use MSan only with a correctly instrumented dependency environment. Sanitizers complement source reasoning and compile tests. [AddressSanitizer documentation](https://clang.llvm.org/docs/AddressSanitizer.html). +3. **Static checks:** compiler warnings including conversions/shadowing on project headers; clang-tidy lifetime, bounds and owning-memory checks where applicable; review all suppressions. Scope warning enforcement to project code rather than treating the old vendored test framework as the library. +4. **Fuzzing:** byte-buffer parsers for NPY/BMP/native binary/text and a bounded shape/slice operation-sequence harness. Fix seeds, timeouts and input/allocation caps. On a fixed runner, budget at least one CPU-hour per target nightly and 24 CPU-hours per target before release; retain minimized failures as permanent regressions. Record coverage and executed corpus, not only elapsed time. +5. **Coverage:** every supported public family must have at least one instantiation and behavioral test; every documented error category needs an exercised negative path. Target at least 90% branch coverage of new validation/parser/ownership code, but never use coverage percentage as an exception to a missing known-defect test. Each F01–F18 item needs closure evidence or an explicit supported-scope decision. +6. **Consumers and artifacts:** build and run all inventoried production consumers against the release candidate; test installed and amalgamated headers; publish only after those checks succeed. The existing auto-release workflow does not depend on the test workflow and must be changed before trusting releases. + +Release acceptance requires zero unresolved P0/P1 findings in the supported surface, all specified regression suites green, no unsuppressed project sanitizer/static-analysis findings, completed fuzz budgets without open crashes, and accepted consumer migration results. Suspected defects are not closed by renaming the API or disabling an assertion. Unsupported experimental features must be labeled, excluded from the production guarantee, and unavailable through an apparently safe fallback. + +**Performance work starts from a correct baseline.** Benchmark serial kernels before selecting thread counts or ISA-specific paths. The current code creates threads repeatedly and switches multiplication algorithms based on the global `PARALLEL` macro; it also materializes many matrices in Strassen and block inversion. These are optimization hypotheses, not measured bottleneck rankings. + +| Workload | Sizes/data | Measurements and candidate change | +|---|---|---| +| GEMM and matrix-vector | Square 8,16,32,64,128,256,512,1024; tall/skinny and wide cases; dimensions 16/17/18 around the current threshold | Latency, throughput, allocations, temporary bytes, thread count, numerical error. Compare cache-blocked traversal and optional BLAS/linalg against current direct/Strassen paths. | +| Elementwise/statistics | Tiny matrices through 10⁶ elements; aligned/misaligned and tail lengths; mixed signs | Per-call overhead, effective bandwidth, rounding behavior. Reduce allocations and temporary matrices before adding SIMD. | +| Views/transposes | Whole matrix and offset narrow slices, rectangular shapes | Copy count, stride costs and bytes touched. Use views internally where lifetime/aliasing are explicit. | +| FFT/convolution | Power-of-two, prime and rectangular sizes; small/large kernels | Correctness first, then scaling, throughput, plan/setup cost and temporary memory. Compare a qualified FFT backend against the correct small-input reference. | +| Build footprint | Minimal matrix include, common consumer translation unit, image-enabled unit | Clean compile time, object/executable size and startup allocations. Separate optional I/O and per-translation-unit color-map initialization if measurements justify it. | + +Use pinned compiler/flags/backend versions, fixed seeds, warmups, at least 20 timed samples and medians plus dispersion. Record CPU, available cores, affinity policy and backend threads. Compare like-for-like scalar types, checking, numerical tolerances and result consumption. Portable release builds use normal `-O2`/`-O3`; native ISA and relaxed floating-point builds are separate opt-in measurements. GCC documents that `-Ofast` enables optimizations that do not preserve all standard guarantees. [GCC optimization options](https://gcc.gnu.org/onlinedocs/gcc/Optimize-Options.html). + +Proposed performance acceptance: no unexplained median regression greater than 5% on stable representative workloads; enable a new optional path only with a repeatable benefit (initial target ≥10% on its intended workload) and unchanged numerical/error guarantees. Review noisy microbenchmarks separately. A necessary safety check can exceed the performance budget only with an explicit documented decision; removing validation is not an optimization strategy. No universal speedup target should be promised before WP13 establishes the baseline. + +**Migration and rollback.** Ship bounded safety repairs early with release notes; reserve storage privacy, checked row proxies, changed return types and changed numerical conventions for a clearly versioned migration. Keep `pinv`/`pinverse`, `rand`/`random`, and member/free convenience names as forwarding aliases when behavior can be shared safely. Introduce corrected pure shift and convolution contracts explicitly so consumers can update golden results; do not preserve memory-unsafe behavior behind a compatibility switch. + +Inventory direct access to storage fields before making them private. Provide accessors and an upgrade guide, then rebuild all consumers: header-only libraries can still cross binary boundaries when instantiated objects are passed between separately built components. Keep one backend/configuration contract per program. Pin each consumer to the previous release until its acceptance tests pass, then upgrade progressively. Backend dispatch must be switchable back to the qualified scalar path without changing the mathematical API. Keep corrected fixture provenance and any changed numerical outputs in release notes. + +**Reproduction record.** Commands below were executed from the repository root through the required `rtk` proxy. They compile only existing source and place outputs outside the repository. `/tmp/matrix-review-20261001.gcCOxS` was the temporary evidence directory for this review; it is not a durable artifact dependency of this report. + +| Purpose | Command | Decisive output / exit | +|---|---|---| +| Baseline build | `rtk proxy g++ -std=c++20 -O2 -DPARALLEL -pthread tests/test.cc -o /tmp/matrix-review-20261001.gcCOxS/test-gcc` | Exit 0; build log empty. | +| Baseline run | `rtk proxy /tmp/matrix-review-20261001.gcCOxS/test-gcc` | `All tests passed (49216592 assertions in 57 test cases)`; exit 0. | +| Sanitizer build | `rtk proxy g++ -std=c++20 -O1 -DNDEBUG -DPARALLEL -pthread -fsanitize=address,undefined -fno-omit-frame-pointer tests/test.cc -o /tmp/matrix-review-20261001.gcCOxS/test-gcc-asan` | Build succeeded. | +| Sanitizer run | `rtk proxy env ASAN_OPTIONS=detect_leaks=1 /tmp/matrix-review-20261001.gcCOxS/test-gcc-asan` | `All tests passed (49216592 assertions in 57 test cases)`; no sanitizer diagnostic. | +| Clang portability | `rtk proxy clang++ -std=c++20 -fsyntax-only -DPARALLEL -pthread tests/test.cc` | Exit 1; `implicit instantiation of undefined template 'feng::matrix>'`; six errors, two warnings. | +| Clang sanitizer attempt | `rtk proxy clang++ -std=c++20 -O1 -DNDEBUG -DPARALLEL -pthread -fsanitize=address,undefined -fno-omit-frame-pointer tests/test.cc -o /tmp/matrix-review-20261001.gcCOxS/test-clang-asan` | Exit 1; same six incomplete-template errors. | +| C++26 syntax | `rtk proxy g++ -std=c++26 -fsyntax-only -DPARALLEL -pthread tests/test.cc` | Exit 0; deprecated whitespace before `_u8` literal suffix at line 362, emitted twice. | + +Compiler version strings were `g++ (GCC) 16.2.1 20260810` and `clang version 22.1.8`, targeting x86_64 Linux. Captured commands used shell redirection for logs and ran build/run pairs sequentially; the table separates the equivalent invocations for readability. All online references were consulted on 2026-10-01. Estimates, priority assignments, target architecture, test budgets, and performance thresholds are recommendations from this review, not guarantees made by the cited standards or libraries. diff --git a/docs/opencode_sharded_review.md b/docs/opencode_sharded_review.md new file mode 100644 index 0000000..79d072a --- /dev/null +++ b/docs/opencode_sharded_review.md @@ -0,0 +1,364 @@ +# Sharded Code Review — `matrix.hpp` + +- **Date:** 2026-07-13 +- **Scope:** `matrix.hpp` (7,689 lines, single-header C++20 matrix library, `namespace feng`), plus `tests/` and `ReadMe.md` for contract evidence. +- **Method:** Manual review along six axes (correctness, readability, security/safety, tests, architecture, performance) followed by empirical verification with GCC 16.2 (C++20, `-DPARALLEL`): + - Full test suite built via `make test` and executed: **All tests passed (49,216,592 assertions in 57 test cases)**. + - AddressSanitizer probes compiled with `-DNDEBUG -DPARALLEL -fsanitize=address` (so `better_assert` is a silent no-op and the *actual* out-of-bounds behavior is observable rather than aborted on a precondition). +- **Line numbers** refer to `matrix.hpp` at review time. + +## Findings summary + +| # | Severity | Axis | Finding | Evidence verified | +|---|----------|------|---------|-------------------| +| C1 | **Critical** | Correctness | `shrink_to_size` copies the wrong column count → heap OOB write + silent corruption | ASan-confirmed | +| C2 | **Critical** | Correctness | `flipdim(m, 2)` swaps a *column* with a *row* → heap OOB (non-square) / silent corruption (square) | ASan-confirmed | +| C3 | High | Correctness | `fliplr`/`flipud` aliases are swapped vs. conventional semantics | Code-verified | +| C4 | High | Correctness | `pinverse`/`pinv` never inverts the singular values | Probe-confirmed (returns 2.0 where 0.5 expected) | +| C5 | High | Correctness | `det()` Schur complement uses `P.inverse()` with no singularity handling → silent `NaN` | Probe-confirmed | +| C6 | High | Correctness | `operator^` does not compile for any odd exponent ≥ 3 (precedence bug) | Compile probe-confirmed | +| S1 | High | Security/Correctness | `load_npy` performs no buffer-size validation → OOB read on truncated files; `stoul` can throw from `noexcept` | ASan-confirmed | +| P1 | High | Performance | `fft`/`ifft` are naive O(N⁴) direct DFTs despite the FFT name | Code-verified | +| S2 | Medium | Security/Safety | `save_png` dereferences unchecked `fopen` result (null `FILE*`) | Code-verified | +| C7 | Medium | Correctness | `better_assert` silently no-ops under `NDEBUG`, turning all boundary checks into UB paths in release builds | Code-verified | +| C8 | Medium | Correctness | `mean`/`variance`/`standard_deviation` truncate for integer matrices | Probe-confirmed | +| C9 | Medium | Correctness | `conv` "same" mode: second assert checks `rb` instead of `cb`; both reject valid 1×1 kernel | Code-verified | +| C10 | Medium | Correctness | `rref`/`gauss_jordan_elimination` precondition `row < col` rejects square systems the algorithm handles | Code-verified | +| C11 | Medium | Correctness | `rand` uses global `srand`/`rand`: re-seeds every call, not thread-safe, low quality | Code-verified | +| T1 | Medium | Tests | Test suite is green but the five most buggy code paths (shrink_to_size, flipdim, pinverse, det, `^`) have zero test coverage | Verified by listing `tests/cases/` | +| A1 | Medium | Architecture | ~30 CRTP mixins each re-derive identical typedefs via `type_proxy_type`; high indirection for a single concrete class | Code-verified | +| R1 | Medium | Readability | `svd_inverse` calls `singular_value_decomposition(a, u, v, w)` with swapped argument order vs. the signature `(a, u, w, v)` | Code-verified | +| R2 | Low | Readability | ~40 nearly identical 6-line elementwise templates (unary/binary/complex math, ~1,200 lines of boilerplate) | Code-verified | +| R3 | Low | Readability | Stray double semicolon in `save_png`; typo in `det` precondition message ("the row and matrix are supposed to be same") | Code-verified | +| C12 | Low | Correctness | `matrix_details::reduce` divides by `hardware_concurrency()` which may be 0 → SIGFPE | Code-verified (unreachable on typical hosts) | +| A2 | Low | Architecture | API duplication: `random`↔`rand`, `random_like`↔`rand_like`, `pinv`↔`pinverse`; free `det(m)` + member `m.det()` | Code-verified | +| P2 | Low | Performance | `lu_decomposition` has no partial pivoting (stability), and `cholesky` has no positive-definiteness guard | Code-verified | +| C13 | Low | Correctness | `fftshift`/`ifftshift` are wrong for odd dimensions (pair-swap, not circular rotation) | Derived (not executed) | + +--- + +## Correctness + +### C1 — `shrink_to_size` copies the wrong column count (Critical) + +- **Severity:** Critical (memory corruption) +- **Evidence:** `matrix.hpp:3528-3532` + ```cpp + size_type const the_rows_to_copy = std::min( zen.row(), new_row ); + size_type const the_cols_to_copy = std::min( zen.col(), new_col ); + + for ( size_type r = 0; r != the_rows_to_copy; ++r ) + std::copy( zen.row_begin( r ), zen.row_begin( r ) + the_rows_to_copy, other.row_begin( r ) ); + ``` + The loop copies `the_rows_to_copy` **columns per row** instead of `the_cols_to_copy`. +- **Violated contract:** the documented behavior ("if new row or col are larger than the original, padding with zero; otherwise, drop these elements", comment at `matrix.hpp:3515-3517`). +- **Impact (empirically verified):** + - `matrix{5,5,1.0}.shrink_to_size(5,3)` → AddressSanitizer: `heap-buffer-overflow` at `matrix.hpp:3532`. + - `matrix{3,10}.shrink_to_size(5,2)` → no crash but **silent corruption**: last row becomes `(21, 22, 23)` instead of the documented zero padding. +- **Smallest safe fix:** `std::copy( zen.row_begin( r ), zen.row_begin( r ) + the_cols_to_copy, other.row_begin( r ) );` +- **Confidence:** 100% (reproduced). + +### C2 — `flipdim(m, 2)` swaps a column with a row (Critical) + +- **Severity:** Critical (memory corruption) +- **Evidence:** `matrix.hpp:4476-4481` + ```cpp + std::swap_ranges( ans.col_begin( index_left ), ans.col_end( index_left ), ans.row_begin( index_right ) ); + ``` + The third argument of `swap_ranges` must be the start of the *second column*, i.e. `ans.col_begin( index_right )`. As written, it swaps a column (length `row()`) against a *row* (length `col()`). +- **Violated contract:** `flipdim` must flip along dimension 2 (left/right flip), per the parallel structure of the `dim == 1` branch and the public `fliplr`/`flipud` API. +- **Impact (empirically verified):** + - Square 4×4: result **does not equal** a left-right flip (silent data corruption). + - Non-square 3×5: AddressSanitizer `heap-buffer-overflow` at `matrix.hpp:4479`. +- **Smallest safe fix:** use `ans.col_begin( index_right )` as the third argument. +- **Confidence:** 100% (reproduced). + +### C3 — `fliplr` / `flipud` aliases are swapped (High) + +- **Severity:** High (wrong semantics; compounds C2) +- **Evidence:** `matrix.hpp:4491-4499` + ```cpp + matrix const fliplr( matrix const& m ) { return flipdim( m, 1 ); } // dim 1 flips up/down + matrix const flipud( matrix const& m ) { return flipdim( m, 2 ); } // dim 2 flips left/right + ``` +- **Violated contract:** MATLAB/NumPy convention, which this library follows elsewhere (`meshgrid`, `conv`, pooling): `fliplr` = left-right (column) flip, `flipud` = up-down (row) flip. +- **Impact:** users get the transpose-axis flip they didn't ask for; silent, no error. +- **Smallest safe fix:** `fliplr → flipdim(m, 2)`, `flipud → flipdim(m, 1)`. +- **Confidence:** High (semantics by convention; the flipdim body itself is broken anyway). + +### C4 — `pinverse` / `pinv` never inverts the singular values (High) + +- **Severity:** High (silently wrong numerical results) +- **Evidence:** `matrix.hpp:5226-5230` + ```cpp + Matrix const pinverse( const Matrix& m ) + { + Matrix u, w, v; + singular_value_decomposition( m, u, w, v ); + return v * w * u.transpose(); // W is the diagonal of singular values, NOT inverted + } + ``` + The pseudoinverse is `V · Σ⁺ · Uᵀ`; this returns `V · Σ · Uᵀ`. Compare `svd_inverse` (`matrix.hpp:5216-5224`), which *does* invert the diagonal with a 1e-10 threshold and produces correct results. +- **Violated contract:** a function named `pinverse` must compute the Moore–Penrose pseudoinverse. +- **Impact (empirically verified):** `pinverse(diag(1,2))` returns `diag(1, 2)`; expected `diag(1, 0.5)`. `svd_inverse(diag(1,2))` correctly returns `diag(1, 0.5)`. +- **Smallest safe fix:** `return svd_inverse( m );` (delete the body), or apply the same diagonal-inversion loop as `svd_inverse`. +- **Confidence:** 100% (reproduced). + +### C5 — `det()` uses `P.inverse()` without handling a singular P (High) + +- **Severity:** High (silent `NaN`/wrong results) +- **Evidence:** `matrix.hpp:2063-2067` + ```cpp + zen_type const& tmp = S - ( R * ( P.inverse() ) * Q ); + return P.det() * tmp.det(); + ``` + The Schur-complement identity `det = det(P)·det(S − R·P⁻¹·Q)` requires `P` nonsingular. There is no check; `inverse()` on a singular block yields `inf`/`NaN`, which propagates silently. +- **Violated contract:** `det` must return the determinant for any square matrix (ReadMe §"det -- matrix determinant"); for singular input the answer is `0`, not `NaN`. +- **Impact (empirically verified):** + ```cpp + // P block [[1,2],[2,4]] is singular; true determinant is 0 + det(m) == -nan + ``` +- **Additional evidence:** the precondition message at `matrix.hpp:2056` has a typo ("the row and matrix are supposed to be same"). +- **Smallest safe fix:** compute the determinant via the existing `lu_decomposition` (`matrix.hpp:~6700`): `det = ±∏U_ii` with a singularity check, and return `NaN`/`std::optional` on pivot zero. This also removes the O(n³)-per-level `inverse()` (see P2). +- **Confidence:** 100% (reproduced). + +### C6 — `operator^` does not compile for odd exponents ≥ 3 (High) + +- **Severity:** High (public API member unusable) +- **Evidence:** `matrix.hpp:5567` + ```cpp + if ( n & 1 ) + return lhs ^ ( n - 1 ) * lhs; // parses as lhs ^ ((n-1) * lhs) — `*` binds tighter than `^` + ``` +- **Violated contract:** `m ^ n` (integer power) is a documented public operation. +- **Impact (empirically verified):** + ``` + matrix.hpp:5567:36: error: no match for ‘operator*’ + (operand types are ‘uint_least64_t’ and ‘const feng::matrix’) + return lhs ^ ( n - 1 ) * lhs; + ``` + `m ^ 3` fails to instantiate; only `n == 0, 1` and even powers compile. +- **Smallest safe fix:** + ```cpp + auto const& half = lhs ^ ( n >> 1 ); + return half * half * lhs; + ``` +- **Confidence:** 100% (reproduced). + +### C7 — `mean`/`variance`/`standard_deviation` truncate for integer matrices (Medium) + +- **Severity:** Medium (wrong numerical results for integer types) +- **Evidence:** `matrix.hpp:7640-7641` + ```cpp + auto mean( Mat const& m ) { return sum( m ) / m.size(); } + ``` + For `matrix` this is integer division. +- **Violated contract:** "mean" is the arithmetic mean; ReadMe documents `mean` for numeric matrices generally. +- **Impact (empirically verified):** `mean(matrix{1,2, {1,2}}) == 1` (expected 1.5). Also `variance` of `{1,2}` is `0.25 → 0`, so `standard_deviation` of a 2-element int matrix is `0`. +- **Smallest safe fix:** promote the divisor/accumulator to `double` (or the matrix's floating-point promotion type) in the reduce helpers, or document integer truncation explicitly in the ReadMe. +- **Confidence:** 100% (reproduced); severity is a contract judgment. + +### C9 — `conv` "same" mode asserts are wrong (Medium) + +- **Severity:** Medium +- **Evidence:** `matrix.hpp:6620-6621` + ```cpp + better_assert( rb > 1, " ... the row of the second matrix is at least 1, but now has ", rb ); + better_assert( rb > 1, " ... the column of the second matrix is at least 1, but now has ", cb ); + ``` + Two problems: (1) the second assert re-checks `rb` instead of `cb`; (2) the message says "at least 1" but the condition `> 1` rejects a valid 1×1 kernel (for which "same" mode is well-defined and the code below handles it: `(rb-1)>>1 == 0`). +- **Violated contract:** the documented "same" mode (matches NumPy/Matlab `conv(...,'same')`, ReadMe §pooling/conv region). +- **Impact:** in debug builds a valid 1×1-kernel "same" convolution aborts; in release builds the column bound is never enforced. +- **Smallest safe fix:** `better_assert( rb >= 1 && cb >= 1, ... )` (or drop, since the slicing below already requires positive dims), and fix the copy-pasted condition. +- **Confidence:** High. + +### C10 — `rref`/`gauss_jordan_elimination` precondition `row < col` (Medium) + +- **Severity:** Medium (overly restrictive documented precondition) +- **Evidence:** `matrix.hpp:6396` + ```cpp + better_assert( row < col && "matrix row must be less than colum to execut a Gauss-Jordan Elimination" ); + ``` + The algorithm (partial-pivoting Gauss–Jordan, `matrix.hpp:6398-6420`) is fully defined for square and even over-determined systems; only the assert is restrictive. In debug builds `rref(square)` aborts; in release the same call succeeds — inconsistent behavior across build modes. +- **Violated contract:** `rref` (Matlab alias, comment at `matrix.hpp:6427`) is expected to work on square systems. +- **Smallest safe fix:** relax to `row > 0 && col > 0`; keep the pivot-magnitude early exit (`1.0e-10`) as the singularity signal. +- **Confidence:** High (code-level; not executed in debug mode to avoid the intended abort). + +### C11 — `rand` uses the global `srand`/`rand` (Medium) + +- **Severity:** Medium +- **Evidence:** `matrix.hpp:5244-5250` + ```cpp + if ( 0 == seed ) + std::srand( static_cast< unsigned int >( ... std::time(nullptr) + reinterpret_cast<...>( &ans ) ) ); + else + std::srand( seed ); + auto const& generator = []() noexcept + { return ( static_cast( std::rand() ) + 1 ) / ( static_cast( RAND_MAX ) + 2 ); }; + ``` +- **Violated contract / invariant:** `rand` is a public API of a library whose own algorithms run on multiple threads; C++11+ `rand()`/`srand()` are not required to be thread-safe (concurrent `rand()` calls are a data race → UB), and re-seeding the single global generator from a time+address value on every call makes repeated calls within the same second highly correlated. +- **Impact:** low-quality, potentially correlated randomness; UB if users fill matrices concurrently (e.g., inside a `std::async`/thread pool). +- **Smallest safe fix:** use a local `std::mt19937` (seeded as today) and `std::uniform_real_distribution(0.0, 1.0)`; drop `noexcept` if the allocation can throw. +- **Confidence:** High (code-level; concurrency impact is latent). + +### C12 — `reduce` divides by `hardware_concurrency()` which may be 0 (Low) + +- **Severity:** Low +- **Evidence:** `matrix.hpp:1152-1161` — `cache.resize( total_cores ); auto block_size = total_elements / total_cores;` with `total_cores = std::thread::hardware_concurrency()`, which is permitted to return `0` ("cannot determine"). +- **Impact:** integer division by zero (SIGFPE) on hosts where it returns 0. The `parallel` helper at `matrix.hpp:276` guards with `total_cores <= 1`; this `reduce` path does not. +- **Smallest safe fix:** `if ( total_cores < 1 ) total_cores = 1;` (also applies to `matrix.hpp:4036`). +- **Confidence:** High. + +### C13 — `fftshift`/`ifftshift` wrong for odd dimensions (Low) + +- **Severity:** Low +- **Evidence:** `matrix.hpp:6340-6355` (and mirror at 6470-6485). For odd `R`, `row_starter = R/2 + 1` and the loop swaps rows `i` with `R/2+1+i` only, leaving the middle row fixed — a pair-swap, not the circular rotation by `floor(R/2)` that `fftshift` is defined as. E.g. `R=3`: produces `[2,1,0]` instead of `[1,2,0]`. +- **Smallest safe fix:** implement as a two-block move (`std::rotate` of row indices), or `row r → (r + (R-1)>>1) % R`. +- **Confidence:** Medium (derived by hand; not executed because the DFT around it makes a probe slow). + +--- + +## Security / Safety + +### S1 — `load_npy` performs no size validation on untrusted file input (High) + +- **Severity:** High (out-of-bounds reads on malformed external input) +- **Evidence:** `matrix.hpp:2508-2560`. After `std::ifstream` succeeds the code dereferences fixed offsets with no length checks: + - `buffer.data()+6` (version), `buffer.data()+8..11` (header length), `buffer.data()+10/12 + header_length` (header string), and finally `std::copy_n( buffer.data()+data_offset, row*col, ... )` where `row*col` comes from *parsed file contents*. +- **Violated contract / invariant:** "data from external sources is treated as untrusted; external data flows are validated at system boundaries before use." `load_npy` is a file-input boundary. +- **Impact (empirically verified):** a 3-byte file → AddressSanitizer `heap-buffer-overflow` read at `matrix.hpp:2520`. Additional issues: + - `std::stoul` on a malformed header **throws** from a `noexcept` member → `std::terminate`. + - No `dtype` check: a `float32`/complex `.npy` loaded into `matrix` silently copies misinterpreted bytes. + - In `NDEBUG` builds the only guard (`better_assert( ifs, ... )`) is a no-op, so even open failures fall through into the OOB path. +- **Smallest safe fix:** validate before any dereference: + ```cpp + if ( buffer.size() < 12 ) return false; + // after parsing header_length: + if ( buffer.size() < data_offset + header_length + std::size_t{row} * col * sizeof( value_type ) ) + return false; + // after parsing dtype: + if ( header.find( expected_dtype_string ) == std::string::npos ) return false; + ``` + and either drop `noexcept` or catch `stoul` exceptions. +- **Confidence:** 100% (OOB reproduced); dtype issue verified by reading the code. + +### S2 — `save_png` dereferences unchecked `fopen` result (Medium) + +- **Severity:** Medium +- **Evidence:** `matrix.hpp:3100-3105` + ```cpp + FILE* fp = fopen( file_name, "wb" ); + for ( i = 0; i < 8; i++ ) + fputc( ( "\x89PNG\r\n\32\n" )[i], fp );; // also a stray double semicolon + ``` + No `if ( !fp )` check before the first `fputc` (null-pointer UB on open failure, e.g. bad path/permissions); the function is `noexcept`. +- **Smallest safe fix:** `if ( !fp ) return;` immediately after `fopen`; remove the stray `;`. Contrast with `save_as_bmp` (`matrix.hpp:~6830`), which correctly checks the stream and reports via `better_assert`. +- **Confidence:** High. + +--- + +## Readability / Simplicity + +### R1 — `svd_inverse` swaps argument order against the function signature (Medium) + +- **Severity:** Medium (comprehensibility trap; one of the direct causes of C4) +- **Evidence:** `matrix.hpp:5216-5224` calls `singular_value_decomposition( a, u, v, w )` while the signature is `( A, u, w, v )`. Local variable names then match the *call site*, not the function's parameters, so reading the body (`for_each( v.begin(), ... ) 1.0/val`) requires knowing the swap. +- **Smallest safe fix:** keep names consistent with the signature: `matrix u, w, v; singular_value_decomposition( a, u, w, v ); ... invert w ...; return v * w.transpose()*... ` (i.e., stop transposing the V matrix into the "w" slot), or better, delete `svd_inverse` and fix `pinverse` (C4) to be the single correct implementation. +- **Confidence:** High. + +### R2 — ~40 near-identical elementwise templates (Low–Medium) + +- **Severity:** Low (no correctness impact; maintainability cost) +- **Evidence:** the "unary functions" block (`matrix.hpp:6495-6930`) and "binary functions" block (`matrix.hpp:6940-7545`) contain ~40 functions that are the same 6-line shape: `zeros_like` + `matrix_details::for_each` + `std::`. E.g. `exp`, `exp2`, `expm1`, `log`, `log10`, `log1p`, `log2`, `sqrt`, … `abs`, `exp`, `imag` each differ only in the standard function and (for complex) the result type. +- **Contract clause:** "Could this be done in fewer lines?" — this is ~1,200 lines of copy-paste. +- **Smallest safe fix:** one macro or a small `apply_unary(m)`/`apply_binary(a,b)` helper; or keep the explicit list but generate it via a single macro that lists the function names. Low urgency. +- **Confidence:** High. + +### R3 — Dead/broken artifacts (Low) + +- Stray `;;` at `matrix.hpp:3105` (save_png). +- `better_assert` typo in `det` message at `matrix.hpp:2056`: "the row and matrix are supposed to be same". +- `matrix const` return type (top-level `const` on returned prvalues) is used across the free-function API (e.g. `magic`, `flipdim`, `rand`); harmless but non-idiomatic and signals confusion with `const&` returns. +- **Confidence:** High. + +--- + +## Tests + +### T1 — Test suite is green, but coverage avoids the buggy paths (Medium) + +- **Severity:** Medium +- **Evidence:** `tests/cases/` contains 59 small files (1,158 lines total), dominated by elementwise unary-math cases (`sin.hpp`, `cos.hpp`, …). There is **no test** for: `shrink_to_size`, `flipdim`/`fliplr`/`flipud`, `pinverse`/`svd_inverse`, `det` (member or free), `operator^`/`pow` on matrices, `operator*(valarray, matrix)`, `conv` modes, `fft`, or file save/load except a single happy-path `load_npy`. Examples (`examples/cases/0005_det.hpp`, `0018_conv.hpp`, `0021_singular_value_decomposition.hpp`, …) exercise some of these, but the maintained Catch2 suite does not. +- **Violated contract clause:** "Are all error paths covered? Do the tests actually assert the right things?" — every Critical/High finding above (C1, C2, C4, C5, C6, S1) is in a path with no test, which is why a fully green suite (57 cases / 49.2M assertions) coexists with heap corruption. +- **Smallest safe fix:** add one regression case each: `shrink_to_size(5,5→5,3)` content+shape check; `flipdim` on 3×5 vs. expected; `pinverse(diag(1,2))` ≈ `diag(1,0.5)`; `det` of the singular-P matrix ≈ 0; `m ^ 3` vs. `m*m*m`; `load_npy` on a truncated file expecting `false` (needs S1 fix first). +- **Confidence:** High (file listing + `make test` run). + +### T2 — Happy-path-only assertions; error paths untested (Medium) + +- **Severity:** Medium +- **Evidence:** e.g. `tests/cases/ones.hpp` (shown above) only checks well-formed shapes; `load_npy.hpp` loads a valid file; no test expects `{}` from `lu_solver` on a singular matrix or `nullopt` from `gauss_jordan_elimination`. +- **Smallest safe fix:** after fixing S1/C5/C10, add negative-path cases (singular det, singular LU, truncated npy, `rref` on a square matrix). +- **Confidence:** High. + +--- + +## Architecture + +### A1 — CRTP mixin sprawl for a single concrete class (Medium) + +- **Severity:** Medium (design debt; no behavior bug) +- **Evidence:** `matrix.hpp:3760` — `matrix` inherits ~30 `crtp_*` structs (`crtp_typedef`, `crtp_inverse`, `crtp_det`, `crtp_clone`, `crtp_shrink_to_size`, `crtp_load_npy`, …). Every mixin re-derives the same typedefs through `crtp_typedef`/`type_proxy_type` and casts back with `static_cast(*this)`. +- **Contract clause:** "Are abstractions earning their complexity?" — CRTP pays a real comprehension cost (a reader must jump mixin → typedef → cast to see what a method does) but buys no reuse: there is exactly one class template, and no second derived type exists. Regular member functions (grouped in sections) would delete the `zen`/`zen_type` indirection layer entirely. +- **Caveat:** this is a *refactor* recommendation, not a fix; do it after the correctness fixes land and are covered by tests (T1). +- **Confidence:** High (structural observation). + +### A2 — Duplicated public API (Low) + +- **Evidence:** `random`→`rand` (`matrix.hpp:5260-5268`), `random_like`→`rand_like` (`5276-5280`), `pinv`→`pinverse` (`5232-5236`), free `det(m)`→`m.det()` (`4319-4321`). Each alias is one line, but doubling the surface means every fix must be applied/verified twice (C4 shows the two SVD-inversion paths already diverged). +- **Smallest safe fix:** keep one canonical name per operation; delete or `static_assert` the duplicates. +- **Confidence:** High. + +### A3 — Hostile-to-ADL name collisions (Low) + +- **Evidence:** `namespace feng` defines free `abs`, `exp`, `sqrt`, `log`, `pow`, `norm`, `real`, `imag`, `conj`, `det`, `diag`, `fft`, `meshgrid` (e.g. `matrix.hpp:6505`, `7560-7630`). With `using namespace feng;` in a translation unit that also uses `std::` or third-party code, overload sets merge and unqualified calls can change meaning (e.g. `abs(x)` for a scalar now also sees `feng::abs(Mat)` — usually SFINAE'd away, but `norm` has *both* a complex-matrix version and the commented-out scalar version at `6160-6190`, which shows the drift risk). +- **Smallest safe fix:** namespace the elementwise layer (e.g. `feng::elem::`) or rename the colliding few (`norm` → `cmplx_norm`). +- **Confidence:** Medium. + +--- + +## Performance + +### P1 — `fft` / `ifft` are naive O(N⁴) direct DFTs (High) + +- **Severity:** High (misleading complexity; unusable for real image sizes) +- **Evidence:** `matrix.hpp:6313-6335` (and `ifft` at `6446-6468`): quadruple-nested loops with the definition `X[r][c] = Σ_r' Σ_c' x[r'][c'] · ω…`, i.e. O(R²C²) per output element → O(R⁴C⁴)-ish per matrix, plus two `cos`/`sin` evaluations (`make_omege`) per multiply. +- **Violated contract / invariant:** the name (`fft`, and `fftshift` matching the FFT convention) implies O(N log N) behavior; a 256×256 input costs trillions of operations here. +- **Smallest safe fix:** (a) rename to `dft` and document the complexity, or (b) implement a real radix-2 FFT row-wise + column-wise (the standard separable 2-D FFT) and keep trig precomputation per row. +- **Confidence:** High (algorithm is plainly the direct sum). + +### P2 — `det`/`inverse`-level routines avoid the library's own LU (Low–Medium) + +- **Severity:** Low–Medium +- **Evidence:** `det` recurses through Schur complements built with `P.inverse()` (`matrix.hpp:2063-2067`), i.e. O(n³) work per recursion level instead of O(n³/3) once via the existing `lu_decomposition` (`matrix.hpp:~6700`). `lu_decomposition` itself performs no partial pivoting, so stability depends on the input; `forward_substitution` masks failure with an `isinf`/`isnan` check (`matrix.hpp:6379-6383`), and `cholesky_decomposition` has no positive-definiteness guard (sqrt of a negative silently yields `NaN`). +- **Smallest safe fix:** implement `det` via LU (folds into C5); add pivot selection to `lu_decomposition`; return `std::optional` from `cholesky` on `sum < 0`. +- **Confidence:** High. + +--- + +## Verified non-issues (checked and found acceptable) + +- `load_bmp` validates header/size consistency before parsing (`matrix.hpp:6760-6770`) — good boundary handling; the model S1 should follow. +- `save_as_bmp` checks stream construction and shape equality of the three channels. +- `expm` scaling matches the standard `A/s2` reduction (the `s == 0` case reduces to the identity scaling); the only edge is `1 << s` at `s ≥ 64`, unreachable in practice for double inputs. +- `conv` padding and `mode == "full"` path are correct; only the `"same"` asserts are wrong (C9). +- `pooling` correctly ignores leftover rows/cols (`row/dim_r` truncation) and validates the action name. +- Full test suite passes as-is (`make test`, 57 cases, 49,216,592 assertions); `examples/` builds the remaining 2 cases gated behind missing optional data. + +## Suggested fix order + +1. C1, C2 (memory corruption, one-line fixes each) + T1 regression tests. +2. S1 (`load_npy` validation) — unblocks negative-path tests. +3. C4 (make `pinverse` = `svd_inverse`), C5 (`det` via LU), C6 (parenthesize `operator^`), C3 (swap aliases). +4. S2, C7 (document or convert the `NDEBUG` policy), C8–C11. +5. P1 (FFT rename or real implementation), then A1/A2/R2 refactors behind the new tests. diff --git a/examples/cases/0000_create.hpp b/examples/cases/0000_create.hpp index c13df04..69a7bcd 100644 --- a/examples/cases/0000_create.hpp +++ b/examples/cases/0000_create.hpp @@ -2,7 +2,7 @@ void _0000_create() { feng::matrix m{ 64, 256 }; std::generate( m.begin(), m.end(), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0000_create.bmp" ); + (void)m.save_as_bmp( "./images/0000_create.bmp" ); } void _0001_create() @@ -10,13 +10,13 @@ void _0001_create() feng::matrix m{ 64, 256 }; std::generate( m.row_begin(17), m.row_end(17), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0001_create.bmp" ); + (void)m.save_as_bmp( "./images/0001_create.bmp" ); } void _0002_create() { feng::matrix m{ 64, 256 }; - m.save_as_bmp( "./images/0002_create.bmp" ); + (void)m.save_as_bmp( "./images/0002_create.bmp" ); assert( m.row() == 64 ); assert( m.col() == 256 ); assert( m.size() == m.row() * m.col() ); @@ -33,7 +33,7 @@ void _0003_create() feng::matrix m{ 64, 256 }; std::generate( m.row_rbegin(17), m.row_rend(17), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0003_create.bmp" ); + (void)m.save_as_bmp( "./images/0003_create.bmp" ); } void _0004_create() @@ -41,7 +41,7 @@ void _0004_create() feng::matrix m{ 64, 256 }; std::generate( m.col_begin(17), m.col_end(17), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0004_create.bmp" ); + (void)m.save_as_bmp( "./images/0004_create.bmp" ); } void _0005_create() @@ -49,14 +49,14 @@ void _0005_create() feng::matrix m{ 64, 256 }; std::generate( m.col_rbegin(17), m.col_rend(17), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0005_create.bmp" ); + (void)m.save_as_bmp( "./images/0005_create.bmp" ); } void _0006_create() { feng::matrix m{ 64, 256 }; std::generate( m.rbegin(), m.rend(), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0006_create.bmp" ); + (void)m.save_as_bmp( "./images/0006_create.bmp" ); } void _0007_create() @@ -64,7 +64,7 @@ void _0007_create() feng::matrix m{ 64, 256 }; std::generate( m.upper_diag_begin(17), m.upper_diag_end(17), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0007_create.bmp" ); + (void)m.save_as_bmp( "./images/0007_create.bmp" ); } void _0008_create() @@ -72,7 +72,7 @@ void _0008_create() feng::matrix m{ 64, 256 }; std::generate( m.upper_diag_rbegin(17), m.upper_diag_rend(17), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0008_create.bmp" ); + (void)m.save_as_bmp( "./images/0008_create.bmp" ); } void _0009_create() @@ -80,7 +80,7 @@ void _0009_create() feng::matrix m{ 64, 256 }; std::generate( m.lower_diag_begin(17), m.lower_diag_end(17), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0009_create.bmp" ); + (void)m.save_as_bmp( "./images/0009_create.bmp" ); } void _0010_create() @@ -88,7 +88,7 @@ void _0010_create() feng::matrix m{ 64, 256 }; std::generate( m.lower_diag_rbegin(17), m.lower_diag_rend(17), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0010_create.bmp" ); + (void)m.save_as_bmp( "./images/0010_create.bmp" ); } void _0011_create() @@ -96,7 +96,7 @@ void _0011_create() feng::matrix m{ 64, 256 }; std::generate( m.diag_begin(), m.diag_end(), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0011_create.bmp" ); + (void)m.save_as_bmp( "./images/0011_create.bmp" ); } void _0012_create() @@ -104,7 +104,7 @@ void _0012_create() feng::matrix m{ 64, 256 }; std::generate( m.diag_rbegin(), m.diag_rend(), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0012_create.bmp" ); + (void)m.save_as_bmp( "./images/0012_create.bmp" ); } void _0013_create() @@ -112,7 +112,7 @@ void _0013_create() feng::matrix m{ 64, 256 }; std::generate( m.upper_anti_diag_begin(17), m.upper_anti_diag_end(17), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0013_create.bmp" ); + (void)m.save_as_bmp( "./images/0013_create.bmp" ); } void _0014_create() @@ -120,7 +120,7 @@ void _0014_create() feng::matrix m{ 64, 256 }; std::generate( m.upper_anti_diag_rbegin(17), m.upper_anti_diag_rend(17), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0014_create.bmp" ); + (void)m.save_as_bmp( "./images/0014_create.bmp" ); } void _0015_create() @@ -128,7 +128,7 @@ void _0015_create() feng::matrix m{ 64, 256 }; std::generate( m.lower_anti_diag_begin(17), m.lower_anti_diag_end(17), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0015_create.bmp" ); + (void)m.save_as_bmp( "./images/0015_create.bmp" ); } void _0016_create() @@ -136,7 +136,7 @@ void _0016_create() feng::matrix m{ 64, 256 }; std::generate( m.lower_anti_diag_rbegin(17), m.lower_anti_diag_rend(17), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0016_create.bmp" ); + (void)m.save_as_bmp( "./images/0016_create.bmp" ); } void _0017_create() @@ -144,7 +144,7 @@ void _0017_create() feng::matrix m{ 64, 256 }; std::generate( m.anti_diag_begin(), m.anti_diag_end(), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0017_create.bmp" ); + (void)m.save_as_bmp( "./images/0017_create.bmp" ); } void _0018_create() @@ -152,7 +152,7 @@ void _0018_create() feng::matrix m{ 64, 256 }; std::generate( m.anti_diag_rbegin(), m.anti_diag_rend(), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0018_create.bmp" ); + (void)m.save_as_bmp( "./images/0018_create.bmp" ); } void _0019_create() @@ -167,7 +167,7 @@ void _0019_create() for ( auto c = 123; c != 234; ++c ) m(r, c) = -1.0; - m.save_as_bmp( "./images/0019_create.bmp" ); + (void)m.save_as_bmp( "./images/0019_create.bmp" ); } void _0020_create() @@ -176,16 +176,16 @@ void _0020_create() for ( auto r = 12; r != 34; ++r ) for ( auto c = 12; c != 34; ++c ) m[r][c] = 1.0; - m.save_as_bmp( "./images/0020_create.bmp" ); + (void)m.save_as_bmp( "./images/0020_create.bmp" ); feng::matrix n = m; //copying - n.save_as_bmp( "./images/0021_create.bmp" ); + (void)n.save_as_bmp( "./images/0021_create.bmp" ); n.resize( 63, 244 ); - n.save_as_bmp( "./images/0022_create.bmp" ); + (void)n.save_as_bmp( "./images/0022_create.bmp" ); m.reshape( m.col(), m.row() ); - m.save_as_bmp( "./images/0023_create.bmp" ); + (void)m.save_as_bmp( "./images/0023_create.bmp" ); } diff --git a/examples/cases/0001_apply.hpp b/examples/cases/0001_apply.hpp index 1aaf0ea..1e94c4b 100644 --- a/examples/cases/0001_apply.hpp +++ b/examples/cases/0001_apply.hpp @@ -2,8 +2,8 @@ void _0000_apply() { feng::matrix m{ 64, 256 }; std::generate( m.begin(), m.end(), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0000_apply.bmp" ); + (void)m.save_as_bmp( "./images/0000_apply.bmp" ); m.apply( [](auto& x) { x = std::sin(x); } ); - m.save_as_bmp( "./images/0001_apply.bmp" ); + (void)m.save_as_bmp( "./images/0001_apply.bmp" ); } diff --git a/examples/cases/0002_access.hpp b/examples/cases/0002_access.hpp index fcaa35f..63c9014 100644 --- a/examples/cases/0002_access.hpp +++ b/examples/cases/0002_access.hpp @@ -11,5 +11,5 @@ void _0000_access() int val = starter++ & 0x7; x = keys[val]; } - m.save_as_bmp( "./images/0000_access.bmp" ); + (void)m.save_as_bmp( "./images/0000_access.bmp" ); } diff --git a/examples/cases/0003_clone.hpp b/examples/cases/0003_clone.hpp index 4917b5a..99477b1 100644 --- a/examples/cases/0003_clone.hpp +++ b/examples/cases/0003_clone.hpp @@ -3,11 +3,11 @@ void _0000_clone() feng::matrix m{ 64, 256 }; //std::generate( m.begin(), m.end(), [](){ double x = 0.0; return [x]()mutable{ x+=0.1; return std::sin(x);}; }() ); std::fill( m.diag_begin(), m.diag_end(), 1.1 ); - m.save_as_bmp( "./images/0000_clone.bmp" ); + (void)m.save_as_bmp( "./images/0000_clone.bmp" ); auto n = m.clone( 0, 32, 0, 64 ); - n.save_as_bmp( "./images/0001_clone.bmp" ); + (void)n.save_as_bmp( "./images/0001_clone.bmp" ); n.clone( m, 32, 64, 0, 64 ); - n.save_as_bmp( "./images/0002_clone.bmp" ); + (void)n.save_as_bmp( "./images/0002_clone.bmp" ); } diff --git a/examples/cases/0004_data.hpp b/examples/cases/0004_data.hpp index 3c0a917..049d6aa 100644 --- a/examples/cases/0004_data.hpp +++ b/examples/cases/0004_data.hpp @@ -1,11 +1,11 @@ void _0000_data() { feng::matrix m{ 64, 256 }; - m.save_as_bmp( "./images/0000_data.bmp" ); + (void)m.save_as_bmp( "./images/0000_data.bmp" ); auto ptr = m.data(); for ( auto idx = 0UL; idx != m.size(); ++idx ) ptr[idx] = std::sin( idx*idx*0.1 ); - m.save_as_bmp( "./images/0001_data.bmp" ); + (void)m.save_as_bmp( "./images/0001_data.bmp" ); } diff --git a/examples/cases/0006_devide_equal.hpp b/examples/cases/0006_devide_equal.hpp index e617cff..3970469 100644 --- a/examples/cases/0006_devide_equal.hpp +++ b/examples/cases/0006_devide_equal.hpp @@ -5,6 +5,6 @@ void _0000_divide_equal() n /= 2.0; m /= n; - m.save_as_bmp( "images/0000_divide_equal.bmp" ); + (void)m.save_as_bmp( "images/0000_divide_equal.bmp" ); } diff --git a/examples/cases/0007_slicing.hpp b/examples/cases/0007_slicing.hpp index 124898d..9827628 100644 --- a/examples/cases/0007_slicing.hpp +++ b/examples/cases/0007_slicing.hpp @@ -4,12 +4,12 @@ void _0000_slicing() std::fill( m.upper_diag_begin(1), m.upper_diag_end(1), 1.0 ); std::fill( m.diag_begin(), m.diag_end(), 1.0 ); std::fill( m.lower_diag_begin(1), m.lower_diag_end(1), 1.0 ); - m.save_as_bmp( "./images/0000_slicing.bmp" ); + (void)m.save_as_bmp( "./images/0000_slicing.bmp" ); feng::matrix n{ m, 0, 32, 0, 64 }; - n.save_as_bmp( "./images/0001_slicing.bmp" ); + (void)n.save_as_bmp( "./images/0001_slicing.bmp" ); feng::matrix p{ m, {16, 48}, {0, 64} }; - p.save_as_bmp( "./images/0002_slicing.bmp" ); + (void)p.save_as_bmp( "./images/0002_slicing.bmp" ); } diff --git a/examples/cases/0008_inverse.hpp b/examples/cases/0008_inverse.hpp index 92270e2..60ac871 100644 --- a/examples/cases/0008_inverse.hpp +++ b/examples/cases/0008_inverse.hpp @@ -3,6 +3,6 @@ void _0000_inverse() auto const& m = feng::rand( 128, 128 ); auto const& n = m.inverse(); auto const& identity = m * n; - identity.save_as_bmp( "./images/0000_inverse.bmp" ); + (void)identity.save_as_bmp( "./images/0000_inverse.bmp" ); } diff --git a/examples/cases/0009_save_load.hpp b/examples/cases/0009_save_load.hpp index 883c533..40b2350 100644 --- a/examples/cases/0009_save_load.hpp +++ b/examples/cases/0009_save_load.hpp @@ -2,31 +2,31 @@ void _0000_save_load() { //auto const& m = feng::rand( 128, 128 ); feng::matrix m; - m.load_txt( "./images/Lenna.txt" ); - m.save_as_txt( "./images/0000_save_load.txt" ); - m.save_as_binary( "./images/0000_save_load.bin" ); - m.save_as_bmp( "./images/0000_save_load.bmp" ); + (void)m.load_txt( "./images/Lenna.txt" ); + (void)m.save_as_txt( "./images/0000_save_load.txt" ); + (void)m.save_as_binary( "./images/0000_save_load.bin" ); + (void)m.save_as_bmp( "./images/0000_save_load.bmp" ); feng::matrix n; - n.load_txt( "./images/0000_save_load.txt" ); - n.save_as_bmp( "./images/0001_save_load.bmp" ); - n.load_binary( "./images/0000_save_load.bin" ); - n.save_as_pgm( "./images/0002_save_load.pgm" ); + (void)n.load_txt( "./images/0000_save_load.txt" ); + (void)n.save_as_bmp( "./images/0001_save_load.bmp" ); + (void)n.load_binary( "./images/0000_save_load.bin" ); + (void)n.save_as_pgm( "./images/0002_save_load.pgm" ); } void _0001_save_load() { { feng::matrix m; - m.load_txt( "./images/Lenna.txt" ); - m.save_as_bmp( "./images/Lenna.bmp", "gray" ); + (void)m.load_txt( "./images/Lenna.txt" ); + (void)m.save_as_bmp( "./images/Lenna.bmp", "gray" ); } auto const& mat_3 = feng::load_bmp( "./images/Lenna.bmp" ); if ( mat_3 ) { - (*mat_3)[0].save_as_bmp( "./images/0001_save_load_julia_red.bmp", "gray" ); - (*mat_3)[1].save_as_bmp( "./images/0001_save_load_julia_green.bmp", "gray" ); - (*mat_3)[2].save_as_bmp( "./images/0001_save_load_julia_blue.bmp", "gray" ); + (void)(*mat_3)[0].save_as_bmp( "./images/0001_save_load_julia_red.bmp", "gray" ); + (void)(*mat_3)[1].save_as_bmp( "./images/0001_save_load_julia_green.bmp", "gray" ); + (void)(*mat_3)[2].save_as_bmp( "./images/0001_save_load_julia_blue.bmp", "gray" ); } else { diff --git a/examples/cases/0010_minus_equal.hpp b/examples/cases/0010_minus_equal.hpp index 4b6d228..057e02b 100644 --- a/examples/cases/0010_minus_equal.hpp +++ b/examples/cases/0010_minus_equal.hpp @@ -1,16 +1,16 @@ void _0000_minus_equal() { feng::matrix image; - image.load_txt( "images/Lenna.txt" ); - image.save_as_bmp("images/0000_minus_equal.bmp", "gray"); + (void)image.load_txt( "images/Lenna.txt" ); + (void)image.save_as_bmp("images/0000_minus_equal.bmp", "gray"); double const min = *std::min_element( image.begin(), image.end() ); image -= min; - image.save_as_bmp("images/0001_minus_equal.bmp", "jet"); + (void)image.save_as_bmp("images/0001_minus_equal.bmp", "jet"); image -= image; - image.save_as_bmp("images/0002_minus_equal.bmp"); + (void)image.save_as_bmp("images/0002_minus_equal.bmp"); } diff --git a/examples/cases/0011_multiply_equal.hpp b/examples/cases/0011_multiply_equal.hpp index ea87f3a..cb2f1f7 100644 --- a/examples/cases/0011_multiply_equal.hpp +++ b/examples/cases/0011_multiply_equal.hpp @@ -3,6 +3,6 @@ void _0000_multiply_equal() auto m = feng::rand( 127, 127 ); m *= m.inverse(); - m.save_as_bmp("images/0001_multiply_equal.bmp"); + (void)m.save_as_bmp("images/0001_multiply_equal.bmp"); } diff --git a/examples/cases/0012_plus_equal.hpp b/examples/cases/0012_plus_equal.hpp index b086b54..6c9399a 100644 --- a/examples/cases/0012_plus_equal.hpp +++ b/examples/cases/0012_plus_equal.hpp @@ -1,8 +1,8 @@ void _0000_plus_equal() { feng::matrix image; - image.load_txt( "images/Lenna.txt" ); - image.save_as_bmp("images/0000_plus_equal.bmp", "gray"); + (void)image.load_txt( "images/Lenna.txt" ); + (void)image.save_as_bmp("images/0000_plus_equal.bmp", "gray"); double const mn = *std::min_element( image.begin(), image.end() ); double const mx = *std::max_element( image.begin(), image.end() ); @@ -10,6 +10,6 @@ void _0000_plus_equal() auto const& noise = feng::rand( image.row(), image.col(), 1 ); //setting random seed to 1 image += 0.1*noise; - image.save_as_bmp("images/0001_plus_equal.bmp", "gray"); + (void)image.save_as_bmp("images/0001_plus_equal.bmp", "gray"); } diff --git a/examples/cases/0013_prefix.hpp b/examples/cases/0013_prefix.hpp index 98e3dcb..2ceb26a 100644 --- a/examples/cases/0013_prefix.hpp +++ b/examples/cases/0013_prefix.hpp @@ -4,6 +4,6 @@ void _0000_prefix() auto const& pp = +m; auto const& pm = -m; auto const& shoule_be_zero = pp + pm; - shoule_be_zero.save_as_bmp("images/0000_prefix.bmp"); + (void)shoule_be_zero.save_as_bmp("images/0000_prefix.bmp"); } diff --git a/examples/cases/0014_sin.hpp b/examples/cases/0014_sin.hpp index f748eb0..dfb9135 100644 --- a/examples/cases/0014_sin.hpp +++ b/examples/cases/0014_sin.hpp @@ -2,8 +2,8 @@ void _0000_sin() { feng::matrix m{ 64, 256 }; std::generate( m.begin(), m.end(), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0000_sin.bmp" ); + (void)m.save_as_bmp( "./images/0000_sin.bmp" ); m = feng::sin(m); - m.save_as_bmp( "./images/0001_sin.bmp" ); + (void)m.save_as_bmp( "./images/0001_sin.bmp" ); } diff --git a/examples/cases/0015_sinh.hpp b/examples/cases/0015_sinh.hpp index 934b1b4..460ea43 100644 --- a/examples/cases/0015_sinh.hpp +++ b/examples/cases/0015_sinh.hpp @@ -2,8 +2,8 @@ void _0000_sinh() { feng::matrix m{ 64, 256 }; std::generate( m.begin(), m.end(), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init/500.0; }; }() ); - m.save_as_bmp( "./images/0000_sinh.bmp" ); + (void)m.save_as_bmp( "./images/0000_sinh.bmp" ); m = feng::sinh(m); - m.save_as_bmp( "./images/0001_sinh.bmp" ); + (void)m.save_as_bmp( "./images/0001_sinh.bmp" ); } diff --git a/examples/cases/0016_eye.hpp b/examples/cases/0016_eye.hpp index 571e6f2..d9595a0 100644 --- a/examples/cases/0016_eye.hpp +++ b/examples/cases/0016_eye.hpp @@ -1,5 +1,5 @@ void _0000_eye() { auto const& m = feng::eye( 128, 128 ); - m.save_as_bmp( "./images/0000_eye.bmp" ); + (void)m.save_as_bmp( "./images/0000_eye.bmp" ); } diff --git a/examples/cases/0017_make_view.hpp b/examples/cases/0017_make_view.hpp index 6d17498..a9301f0 100644 --- a/examples/cases/0017_make_view.hpp +++ b/examples/cases/0017_make_view.hpp @@ -1,21 +1,21 @@ void _0000_make_view() { feng::matrix m; - m.load_txt( "./images/Lenna.txt" ); - m.save_as_bmp( "./images/0000_make_view.bmp" ); + (void)m.load_txt( "./images/Lenna.txt" ); + (void)m.save_as_bmp( "./images/0000_make_view.bmp" ); auto const[r,c] = m.shape(); auto const& v = feng::make_view( m, {r>>2, (r>>2)*3}, {c>>2, (c>>2)*3} ); - v.save_as_bmp( "./images/0001_make_view.bmp" ); + (void)v.save_as_bmp( "./images/0001_make_view.bmp" ); auto new_matrix{v}; - new_matrix.save_as_bmp( "./images/0002_make_view.bmp" ); + (void)new_matrix.save_as_bmp( "./images/0002_make_view.bmp" ); feng::matrix n{ v.row(), v.col() }; // row() and col() of a matrix view for ( auto r = 0UL; r != n.row(); ++r ) for ( auto c = 0UL; c != n.col(); ++c ) n[r][c] = v[r][c]; // accessing matrix elements using operator [], read-only - n.save_as_bmp( "./images/0003_make_view.bmp", "gray" ); + (void)n.save_as_bmp( "./images/0003_make_view.bmp", "gray" ); } diff --git a/examples/cases/0018_conv.hpp b/examples/cases/0018_conv.hpp index c1ddc25..79da6fc 100644 --- a/examples/cases/0018_conv.hpp +++ b/examples/cases/0018_conv.hpp @@ -1,24 +1,31 @@ void _0000_conv() { feng::matrix m; - m.load_txt( "./images/Lenna.txt" ); - m.save_as_bmp( "./images/0000_conv.bmp", "gray" ); + (void)m.load_txt( "./images/Lenna.txt" ); + (void)m.save_as_bmp( "./images/0000_conv.bmp", "gray" ); feng::matrix filter{3, 3, {0.0, 1.0, 0.0, 1.0,-4.0, 1.0, 0.0, 1.0, 0.0}}; auto const& edge = feng::conv( m, filter ); - edge.save_as_bmp( "./images/0001_conv.bmp", "gray" ); - edge.save_as_txt( "./images/0001_conv.txt" ); - edge.save_as_pgm( "./images/0001_conv.pgm" ); + (void)edge.save_as_bmp( "./images/0001_conv.bmp", "gray" ); + (void)edge.save_as_txt( "./images/0001_conv.txt" ); + (void)edge.save_as_pgm( "./images/0001_conv.pgm" ); auto const& edge_valid = feng::conv( m, filter, "valid" ); - edge_valid.save_as_bmp( "./images/0001_conv_valid.bmp", "gray" ); + (void)edge_valid.save_as_bmp( "./images/0001_conv_valid.bmp", "gray" ); auto const& edge_same = feng::conv( m, filter, "same" ); - edge_same.save_as_bmp( "./images/0001_conv_same.bmp", "gray" ); + (void)edge_same.save_as_bmp( "./images/0001_conv_same.bmp", "gray" ); auto const& edge_full = feng::conv( m, filter, "full" ); - edge_full.save_as_bmp( "./images/0001_conv_full.bmp", "gray" ); + (void)edge_full.save_as_bmp( "./images/0001_conv_full.bmp", "gray" ); + + // asymmetric kernel: conv reverses it (scipy.signal.convolve2d), so this is the horizontal Sobel derivative + feng::matrix sobel{3, 3, {-1.0, 0.0, 1.0, + -2.0, 0.0, 2.0, + -1.0, 0.0, 1.0}}; + auto const& edge_sobel = feng::conv( m, sobel, "same" ); + (void)edge_sobel.save_as_bmp( "./images/0001_conv_sobel.bmp", "gray" ); } diff --git a/examples/cases/0019_lu_decomposition.hpp b/examples/cases/0019_lu_decomposition.hpp index eaaa5a3..d06e727 100644 --- a/examples/cases/0019_lu_decomposition.hpp +++ b/examples/cases/0019_lu_decomposition.hpp @@ -2,8 +2,8 @@ void _0000_lu_decomposition() { // initial matrix feng::matrix m; - m.load_txt( "./images/Lenna.txt" ); - m.save_as_bmp( "./images/0000_lu_decomposition.bmp", "gray" ); + (void)m.load_txt( "./images/Lenna.txt" ); + (void)m.save_as_bmp( "./images/0000_lu_decomposition.bmp", "gray" ); // adding noise double mn = *std::min_element( m.begin(), m.end() ); @@ -11,18 +11,18 @@ void _0000_lu_decomposition() m = (m-mn) / (mx - mn + 1.0e-10); auto const& [row, col] = m.shape(); m += feng::rand( row, col, 1 ); // set random seed to 1 - m.save_as_bmp( "./images/0001_lu_decomposition.bmp", "gray" ); + (void)m.save_as_bmp( "./images/0001_lu_decomposition.bmp", "gray" ); // lu decomposition auto const& lu = feng::lu_decomposition( m ); if (lu) { auto const& [l, u] = lu.value(); - l.save_as_bmp( "./images/0002_lu_decomposition.bmp", "jet" ); - u.save_as_bmp( "./images/0003_lu_decomposition.bmp", "jet" ); + (void)l.save_as_bmp( "./images/0002_lu_decomposition.bmp", "jet" ); + (void)u.save_as_bmp( "./images/0003_lu_decomposition.bmp", "jet" ); auto const& reconstructed = l * u; - reconstructed.save_as_bmp( "./images/0004_lu_decomposition.bmp", "gray" ); + (void)reconstructed.save_as_bmp( "./images/0004_lu_decomposition.bmp", "gray" ); } else { diff --git a/examples/cases/0020_gauss_jordan_elimination.hpp b/examples/cases/0020_gauss_jordan_elimination.hpp index 9597b03..73b68ea 100644 --- a/examples/cases/0020_gauss_jordan_elimination.hpp +++ b/examples/cases/0020_gauss_jordan_elimination.hpp @@ -1,12 +1,12 @@ void _0000_gauss_jordan_elimination() { auto const& m = feng::rand( 64, 128, 1 ); // setting random seed to 1 - m.save_as_bmp( "./images/0000_gauss_jordan_elimination.bmp", "gray" ); + (void)m.save_as_bmp( "./images/0000_gauss_jordan_elimination.bmp", "gray" ); auto const& n = feng::gauss_jordan_elimination( m ); //<- also `feng::rref(m);`, alias name from Matlab if (n) - (*n).save_as_bmp( "./images/0001_gauss_jordan_elimination.bmp", "gray" ); + (void)(*n).save_as_bmp( "./images/0001_gauss_jordan_elimination.bmp", "gray" ); else std::cout << "Failed to execute Gauss-Jordan Elimination for matrix m.\n"; } diff --git a/examples/cases/0021_singular_value_decomposition.hpp b/examples/cases/0021_singular_value_decomposition.hpp index 4ac21c4..646d59e 100644 --- a/examples/cases/0021_singular_value_decomposition.hpp +++ b/examples/cases/0021_singular_value_decomposition.hpp @@ -2,18 +2,18 @@ void _0000_singular_value_decomposition() { // load feng::matrix m; - m.load_txt( "./images/Teacher.txt" ); + (void)m.load_txt( "./images/Teacher.txt" ); // normalize auto const mx = *std::max_element( m.begin(), m.end() ); auto const mn = *std::min_element( m.begin(), m.end() ); m = ( m - mn ) / ( mx - mn + 1.0e-10 ); // take a snapshot - m.save_as_bmp( "./images/0000_singular_value_decomposition.bmp", "gray" ); + (void)m.save_as_bmp( "./images/0000_singular_value_decomposition.bmp", "gray" ); // adding noise auto const[r, c] = m.shape(); m += feng::rand( r, c, 2 ); // record noisy matrix - m.save_as_bmp( "./images/0001_singular_value_decomposition.bmp", "gray" ); + (void)m.save_as_bmp( "./images/0001_singular_value_decomposition.bmp", "gray" ); // execute svd auto const& svd = feng::singular_value_decomposition( m ); // check svd result @@ -24,7 +24,7 @@ void _0000_singular_value_decomposition() // try to reconstruct matrix using u * v * w' auto const& m_ = u * v * (w.transpose()); // record reconstructed matrix - m_.save_as_bmp( "./images/0002_singular_value_decomposition.bmp", "gray" ); + (void)m_.save_as_bmp( "./images/0002_singular_value_decomposition.bmp", "gray" ); auto dm = std::min( r, c ); auto factor = 2UL; @@ -38,7 +38,7 @@ void _0000_singular_value_decomposition() auto const& new_m = new_u * new_v * new_w.transpose(); - new_m.save_as_bmp( "./images/0003_singular_value_decomposition_"+std::to_string(new_dm)+".bmp", "gray" ); + (void)new_m.save_as_bmp( "./images/0003_singular_value_decomposition_"+std::to_string(new_dm)+".bmp", "gray" ); factor *= 2UL; } @@ -53,18 +53,18 @@ void _0001_singular_value_decomposition() { // load feng::matrix m; - m.load_txt( "./images/frame_1.txt" ); + (void)m.load_txt( "./images/frame_1.txt" ); // normalize auto const mx = *std::max_element( m.begin(), m.end() ); auto const mn = *std::min_element( m.begin(), m.end() ); m = ( m - mn ) / ( mx - mn + 1.0e-10 ); // take a snapshot - m.save_as_bmp( "./images/1_0000_singular_value_decomposition.bmp", "gray" ); + (void)m.save_as_bmp( "./images/1_0000_singular_value_decomposition.bmp", "gray" ); // adding noise auto const[r, c] = m.shape(); m += feng::rand( r, c, 2 ); // record noisy matrix - m.save_as_bmp( "./images/1_0001_singular_value_decomposition.bmp", "gray" ); + (void)m.save_as_bmp( "./images/1_0001_singular_value_decomposition.bmp", "gray" ); // execute svd auto const& svd = feng::singular_value_decomposition( m ); // check svd result @@ -75,7 +75,7 @@ void _0001_singular_value_decomposition() // try to reconstruct matrix using u * v * w' auto const& m_ = u * v * (w.transpose()); // record reconstructed matrix - m_.save_as_bmp( "./images/1_0002_singular_value_decomposition.bmp", "gray" ); + (void)m_.save_as_bmp( "./images/1_0002_singular_value_decomposition.bmp", "gray" ); auto dm = std::min( r, c ); auto factor = 2UL; @@ -89,7 +89,7 @@ void _0001_singular_value_decomposition() auto const& new_m = new_u * new_v * new_w.transpose(); - new_m.save_as_bmp( "./images/1_0003_singular_value_decomposition_"+std::to_string(new_dm)+".bmp", "gray" ); + (void)new_m.save_as_bmp( "./images/1_0003_singular_value_decomposition_"+std::to_string(new_dm)+".bmp", "gray" ); factor *= 2UL; } diff --git a/examples/cases/0022_save_with_colromap.hpp b/examples/cases/0022_save_with_colromap.hpp index 45d82b5..db58214 100644 --- a/examples/cases/0022_save_with_colromap.hpp +++ b/examples/cases/0022_save_with_colromap.hpp @@ -1,38 +1,38 @@ void _0000_save_with_colormap() { feng::matrix m; - m.load_txt( "./images/Lenna.txt" ); - m.save_as_bmp( "./images/0000_save_with_colormap_default.bmp" ); - m.save_as_bmp( "./images/0000_save_with_colormap_parula.bmp", "parula" ); - m.save_as_bmp( "./images/0000_save_with_colormap_hotblue.bmp", "hotblue" ); - m.save_as_bmp( "./images/0000_save_with_colormap_bluehot.bmp", "bluehot" ); - m.save_as_bmp( "./images/0000_save_with_colormap_jet.bmp", "jet" ); - m.save_as_bmp( "./images/0000_save_with_colormap_obscure.bmp", "obscure" ); - m.save_as_bmp( "./images/0000_save_with_colormap_gray.bmp", "gray" ); + (void)m.load_txt( "./images/Lenna.txt" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_default.bmp" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_parula.bmp", "parula" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_hotblue.bmp", "hotblue" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_bluehot.bmp", "bluehot" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_jet.bmp", "jet" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_obscure.bmp", "obscure" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_gray.bmp", "gray" ); - m.save_as_bmp( "./images/0000_save_with_colormap_hsv.bmp", "hsv" ); - m.save_as_bmp( "./images/0000_save_with_colormap_spring.bmp", "spring" ); - m.save_as_bmp( "./images/0000_save_with_colormap_summer.bmp", "summer" ); - m.save_as_bmp( "./images/0000_save_with_colormap_autumn.bmp", "autumn" ); - m.save_as_bmp( "./images/0000_save_with_colormap_winter.bmp", "winter" ); - m.save_as_bmp( "./images/0000_save_with_colormap_pink.bmp", "pink" ); - m.save_as_bmp( "./images/0000_save_with_colormap_hot.bmp", "hot" ); - m.save_as_bmp( "./images/0000_save_with_colormap_cool.bmp", "cool" ); - m.save_as_bmp( "./images/0000_save_with_colormap_bone.bmp", "bone" ); - m.save_as_bmp( "./images/0000_save_with_colormap_copper.bmp", "copper" ); - m.save_as_bmp( "./images/0000_save_with_colormap_lines.bmp", "lines" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_hsv.bmp", "hsv" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_spring.bmp", "spring" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_summer.bmp", "summer" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_autumn.bmp", "autumn" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_winter.bmp", "winter" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_pink.bmp", "pink" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_hot.bmp", "hot" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_cool.bmp", "cool" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_bone.bmp", "bone" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_copper.bmp", "copper" ); + (void)m.save_as_bmp( "./images/0000_save_with_colormap_lines.bmp", "lines" ); - m.save_as_png( "./images/0000_save_with_colormap_default.png" ); - m.save_as_png( "./images/0000_save_with_colormap_parula.png", "parula" ); + (void)m.save_as_png( "./images/0000_save_with_colormap_default.png" ); + (void)m.save_as_png( "./images/0000_save_with_colormap_parula.png", "parula" ); } void _0001_save_with_colormap() { feng::matrix m; - m.load_txt( "./images/star.txt" ); - m.save_as_bmp( "./images/0001_star_hotblue.bmp", "hotblue" ); - m.save_as_bmp( "./images/0001_star_bluehot.bmp", "bluehot" ); - m.save_as_bmp( "./images/0001_star_hotgreen.bmp", "hotgreen" ); - m.save_as_bmp( "./images/0001_star_greenhot.bmp", "greenhot" ); - m.save_as_bmp( "./images/0001_star_tealhot.bmp", "tealhot" ); + (void)m.load_txt( "./images/star.txt" ); + (void)m.save_as_bmp( "./images/0001_star_hotblue.bmp", "hotblue" ); + (void)m.save_as_bmp( "./images/0001_star_bluehot.bmp", "bluehot" ); + (void)m.save_as_bmp( "./images/0001_star_hotgreen.bmp", "hotgreen" ); + (void)m.save_as_bmp( "./images/0001_star_greenhot.bmp", "greenhot" ); + (void)m.save_as_bmp( "./images/0001_star_tealhot.bmp", "tealhot" ); } diff --git a/examples/cases/0023_magic.hpp b/examples/cases/0023_magic.hpp index 0296090..7515b90 100644 --- a/examples/cases/0023_magic.hpp +++ b/examples/cases/0023_magic.hpp @@ -23,6 +23,6 @@ void _0001_magic() for ( auto cc = 0UL; cc != pixs; ++cc ) v_mat[r*pixs+rr][c*pixs+cc] = mat[r][c]; - v_mat.save_as_bmp("./images/0001_magic.bmp"); + (void)v_mat.save_as_bmp("./images/0001_magic.bmp"); } diff --git a/examples/cases/0024_pooling.hpp b/examples/cases/0024_pooling.hpp index 9716a4a..ca490c5 100644 --- a/examples/cases/0024_pooling.hpp +++ b/examples/cases/0024_pooling.hpp @@ -1,22 +1,22 @@ void _0000_pooling() { feng::matrix m; - m.load_txt( "./images/Lenna.txt" ); - m.save_as_bmp( "./images/0000_pooling.bmp", "gray" ); + (void)m.load_txt( "./images/Lenna.txt" ); + (void)m.save_as_bmp( "./images/0000_pooling.bmp", "gray" ); auto const& pooling_2 = feng::pooling( m, 2 ); - pooling_2.save_as_bmp( "./images/0000_pooling_2.bmp", "gray" ); + (void)pooling_2.save_as_bmp( "./images/0000_pooling_2.bmp", "gray" ); auto const& pooling_4 = feng::pooling( m, 4 ); - pooling_4.save_as_bmp( "./images/0000_pooling_4.bmp", "gray" ); + (void)pooling_4.save_as_bmp( "./images/0000_pooling_4.bmp", "gray" ); auto const& pooling_2_4 = feng::pooling( m, 2, 4 ); - pooling_2_4.save_as_bmp( "./images/0000_pooling_2_4.bmp", "gray" ); + (void)pooling_2_4.save_as_bmp( "./images/0000_pooling_2_4.bmp", "gray" ); auto const& pooling_4_2 = feng::pooling( m, 4, 2 ); - pooling_4_2.save_as_bmp( "./images/0000_pooling_4_2.bmp", "gray" ); + (void)pooling_4_2.save_as_bmp( "./images/0000_pooling_4_2.bmp", "gray" ); auto const& pooling_min = feng::pooling( m, 2, "min" ); - pooling_min.save_as_bmp( "./images/0000_pooling_2_min.bmp", "gray" ); + (void)pooling_min.save_as_bmp( "./images/0000_pooling_2_min.bmp", "gray" ); auto const& pooling_max = feng::pooling( m, 2, "max" ); - pooling_max.save_as_bmp( "./images/0000_pooling_2_max.bmp", "gray" ); + (void)pooling_max.save_as_bmp( "./images/0000_pooling_2_max.bmp", "gray" ); } diff --git a/examples/cases/0025_global_save_as_bmp.hpp b/examples/cases/0025_global_save_as_bmp.hpp index a9b67bf..cf96595 100644 --- a/examples/cases/0025_global_save_as_bmp.hpp +++ b/examples/cases/0025_global_save_as_bmp.hpp @@ -36,6 +36,6 @@ void _0000_global_save_as_bmp() blue[r][c] = BL( r, c ); } - feng::save_as_bmp( "./images/0000_global_save_as_bmp.bmp", red, green, blue ); + (void)feng::save_as_bmp( "./images/0000_global_save_as_bmp.bmp", red, green, blue ); } diff --git a/examples/cases/0026_mandelbrot.hpp b/examples/cases/0026_mandelbrot.hpp index 0210256..c44ab01 100644 --- a/examples/cases/0026_mandelbrot.hpp +++ b/examples/cases/0026_mandelbrot.hpp @@ -27,7 +27,7 @@ auto make_mandelbrot( std::complex const& lower_left, std::complex{-2.25, -1.5}, std::complex{0.75, 1.5} ); - mat.save_as_bmp( "./images/0000_mandelbrot.bmp", "gray" ); + (void)mat.save_as_bmp( "./images/0000_mandelbrot.bmp", "gray" ); } void _0001_mandelbrot() @@ -35,7 +35,7 @@ void _0001_mandelbrot() unsigned long const dims = 1024; unsigned long const iterations = 1024; auto&& mat = make_mandelbrot( std::complex{-2.0, -1.25}, std::complex{0.5, 1.25}, dims, iterations ); - mat.save_as_bmp( "./images/0001_mandelbrot.bmp", "bluehot" ); + (void)mat.save_as_bmp( "./images/0001_mandelbrot.bmp", "bluehot" ); mat /= static_cast( iterations ); //std::cout << "Var(mandelbrot) = " << feng::variance( mat ); } @@ -68,7 +68,7 @@ void _0002_mandelbrot() std::string const file_name = std::string{ "./images/mandelbrot_5/0002_mandel_brot_" } + std::to_string(r) + std::string{"-"} + std::to_string(c) + std::string{".bmp"}; //mat.save_as_bmp( file_name, "bluehot" ); //mat.save_as_bmp( file_name, "gray" ); - mat.save_as_bmp( file_name, "tealhot" ); + (void)mat.save_as_bmp( file_name, "tealhot" ); } } diff --git a/examples/cases/0027_julia_set.hpp b/examples/cases/0027_julia_set.hpp index 0f6cd2e..a57b050 100644 --- a/examples/cases/0027_julia_set.hpp +++ b/examples/cases/0027_julia_set.hpp @@ -59,7 +59,7 @@ void _0000_julia_set() { auto&& mat = make_julia_set( std::complex{-1.5, -1.0}, std::complex{1.5, 1.0}, std::complex{-0.4, 0.6} ); mat.apply( [](auto& x){ x = std::log(1.0+x); } ); - mat.save_as_bmp( "./images/0000_julia_set.bmp", "tealhot" ); + (void)mat.save_as_bmp( "./images/0000_julia_set.bmp", "tealhot" ); } void _0001_julia_set() @@ -76,7 +76,7 @@ void _0001_julia_set() //std::string file_name = std::string{"./images/julia_set/0001_julia_set_"} + std::to_string(r) + std::string{"-"} + std::to_string(c) + std::string{".bmp"}; //std::string file_name = std::string{"./images/julia_set_2/0001_julia_set_"} + std::to_string(r) + std::string{"-"} + std::to_string(c) + std::string{".bmp"}; std::string file_name = std::string{"./images/julia_set_3/0001_julia_set_"} + std::to_string(r) + std::string{"-"} + std::to_string(c) + std::string{".bmp"}; - mat.save_as_bmp( file_name, "bluehot" ); + (void)mat.save_as_bmp( file_name, "bluehot" ); } } @@ -91,7 +91,7 @@ void _0002_julia_set() std::complex zc{ double(r)/n*1.8-0.9, double(c)/n*1.8-0.9 }; auto&& mat = make_julia_set( std::complex{-1.5, -1.0}, std::complex{1.5, 1.0}, zc, 1024, 1024, 4 ); std::string file_name = std::string{"./images/julia_set_4/0001_julia_set_"} + std::to_string(r) + std::string{"-"} + std::to_string(c) + std::string{".bmp"}; - mat.save_as_bmp( file_name, "tealhot" ); + (void)mat.save_as_bmp( file_name, "tealhot" ); } } diff --git a/examples/cases/0028_plot.hpp b/examples/cases/0028_plot.hpp index 6f4b0fe..1800074 100644 --- a/examples/cases/0028_plot.hpp +++ b/examples/cases/0028_plot.hpp @@ -1,15 +1,15 @@ void _0000_plot() { feng::matrix m; - m.load_txt( "./images/Lenna.txt" ); - m.plot( "./images/0000_plot_default.bmp" ); - m.plot( "./images/0000_plot_jet.bmp", "jet" ); + (void)m.load_txt( "./images/Lenna.txt" ); + (void)m.plot( "./images/0000_plot_default.bmp" ); + (void)m.plot( "./images/0000_plot_jet.bmp", "jet" ); } void _0001_plot() { feng::matrix m; - m.load_txt( "./images/star.txt" ); - m.plot( "./images/0001_plot_star_hotblue.bmp", "hotblue" ); - m.plot( "./images/0001_plot_star_bluehot.bmp", "bluehot" ); + (void)m.load_txt( "./images/star.txt" ); + (void)m.plot( "./images/0001_plot_star_hotblue.bmp", "hotblue" ); + (void)m.plot( "./images/0001_plot_star_bluehot.bmp", "bluehot" ); } diff --git a/examples/cases/0029_meshgrid.hpp b/examples/cases/0029_meshgrid.hpp index 2d448a8..e232c0d 100644 --- a/examples/cases/0029_meshgrid.hpp +++ b/examples/cases/0029_meshgrid.hpp @@ -1,8 +1,8 @@ void _0000_meshgrid() { auto const& [X, Y] = feng::meshgrid( 384, 512 ); - X.save_as_bmp( "./images/0000_meshgrid_x.bmp", "grey" ); - Y.save_as_bmp( "./images/0000_meshgrid_y.bmp", "grey" ); + (void)X.save_as_bmp( "./images/0000_meshgrid_x.bmp", "grey" ); + (void)Y.save_as_bmp( "./images/0000_meshgrid_y.bmp", "grey" ); { auto const& [X, Y] = feng::meshgrid( 3, 5 ); diff --git a/examples/cases/0030_arange.hpp b/examples/cases/0030_arange.hpp index a1aad09..975e77d 100644 --- a/examples/cases/0030_arange.hpp +++ b/examples/cases/0030_arange.hpp @@ -2,6 +2,6 @@ void _0000_arange() { auto m = feng::arange( 256*256 ); m.reshape( 256, 256 ); - m.save_as_bmp( "./images/0000_arange.bmp" ); + (void)m.save_as_bmp( "./images/0000_arange.bmp" ); } diff --git a/examples/cases/0031_clip.hpp b/examples/cases/0031_clip.hpp index 4301d20..450763d 100644 --- a/examples/cases/0031_clip.hpp +++ b/examples/cases/0031_clip.hpp @@ -2,14 +2,14 @@ void _0000_clip() { feng::matrix m{ 64, 256 }; std::generate( m.begin(), m.end(), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return init; }; }() ); - m.save_as_bmp( "./images/0000_clip.bmp" ); + (void)m.save_as_bmp( "./images/0000_clip.bmp" ); m = feng::sin(m); - m.save_as_bmp( "./images/0001_clip.bmp" ); + (void)m.save_as_bmp( "./images/0001_clip.bmp" ); auto const& cm0 = feng::clip( 0.1, 0.9 )( m ); - cm0.save_as_bmp( "./images/0002_clip.bmp" ); + (void)cm0.save_as_bmp( "./images/0002_clip.bmp" ); auto const& cm1 = feng::clip( 0.4, 0.6 )( m ); - cm1.save_as_bmp( "./images/0003_clip.bmp" ); + (void)cm1.save_as_bmp( "./images/0003_clip.bmp" ); } diff --git a/examples/cases/0032_empty.hpp b/examples/cases/0032_empty.hpp index 819da96..9cbe8a1 100644 --- a/examples/cases/0032_empty.hpp +++ b/examples/cases/0032_empty.hpp @@ -1,5 +1,5 @@ void _0000_empty() { auto const& m = feng::empty( 128, 128 ); - m.save_as_bmp( "./images/0000_empty.bmp" ); + (void)m.save_as_bmp( "./images/0000_empty.bmp" ); } diff --git a/examples/cases/0034_astype.hpp b/examples/cases/0034_astype.hpp index 77d0889..eb8690a 100644 --- a/examples/cases/0034_astype.hpp +++ b/examples/cases/0034_astype.hpp @@ -3,8 +3,8 @@ void _0000_astype() feng::matrix m{ 64, 256 }; std::generate( m.begin(), m.end(), [](){ double init = 0.0; return [init]() mutable { init += 0.1; return std::sin(init); }; }() ); m = m * 2.0; - m.save_as_bmp( "./images/0000_astype.bmp" ); + (void)m.save_as_bmp( "./images/0000_astype.bmp" ); auto const& mm = m.astype(); - mm.save_as_bmp( "./images/0001_astype.bmp" ); + (void)mm.save_as_bmp( "./images/0001_astype.bmp" ); } diff --git a/images/0000_gauss_jordan_elimination.bmp b/images/0000_gauss_jordan_elimination.bmp index d735e7d..0522e3f 100644 Binary files a/images/0000_gauss_jordan_elimination.bmp and b/images/0000_gauss_jordan_elimination.bmp differ diff --git a/images/0000_global_save_as_bmp.bmp b/images/0000_global_save_as_bmp.bmp index 94a0964..5348acf 100644 Binary files a/images/0000_global_save_as_bmp.bmp and b/images/0000_global_save_as_bmp.bmp differ diff --git a/images/0000_mandelbrot.bmp b/images/0000_mandelbrot.bmp index f82f53d..66ac13d 100644 Binary files a/images/0000_mandelbrot.bmp and b/images/0000_mandelbrot.bmp differ diff --git a/images/0001_conv_sobel.bmp b/images/0001_conv_sobel.bmp new file mode 100644 index 0000000..20cf46a Binary files /dev/null and b/images/0001_conv_sobel.bmp differ diff --git a/images/0001_gauss_jordan_elimination.bmp b/images/0001_gauss_jordan_elimination.bmp index 7b411c1..2e68f15 100644 Binary files a/images/0001_gauss_jordan_elimination.bmp and b/images/0001_gauss_jordan_elimination.bmp differ diff --git a/images/0001_lu_decomposition.bmp b/images/0001_lu_decomposition.bmp index bbda048..b273d62 100644 Binary files a/images/0001_lu_decomposition.bmp and b/images/0001_lu_decomposition.bmp differ diff --git a/images/0001_mandelbrot.bmp b/images/0001_mandelbrot.bmp index 3b61d71..07fd459 100644 Binary files a/images/0001_mandelbrot.bmp and b/images/0001_mandelbrot.bmp differ diff --git a/images/0001_plus_equal.bmp b/images/0001_plus_equal.bmp index cd96bb1..643ef7a 100644 Binary files a/images/0001_plus_equal.bmp and b/images/0001_plus_equal.bmp differ diff --git a/images/0001_singular_value_decomposition.bmp b/images/0001_singular_value_decomposition.bmp index 6a179e5..f82cc9a 100644 Binary files a/images/0001_singular_value_decomposition.bmp and b/images/0001_singular_value_decomposition.bmp differ diff --git a/images/0002_lu_decomposition.bmp b/images/0002_lu_decomposition.bmp index e5ebf5a..1ae7ee0 100644 Binary files a/images/0002_lu_decomposition.bmp and b/images/0002_lu_decomposition.bmp differ diff --git a/images/0002_singular_value_decomposition.bmp b/images/0002_singular_value_decomposition.bmp index 6a179e5..f82cc9a 100644 Binary files a/images/0002_singular_value_decomposition.bmp and b/images/0002_singular_value_decomposition.bmp differ diff --git a/images/0003_lu_decomposition.bmp b/images/0003_lu_decomposition.bmp index d3814a8..59ea16a 100644 Binary files a/images/0003_lu_decomposition.bmp and b/images/0003_lu_decomposition.bmp differ diff --git a/images/0003_singular_value_decomposition_1.bmp b/images/0003_singular_value_decomposition_1.bmp index 3952a13..50d61da 100644 Binary files a/images/0003_singular_value_decomposition_1.bmp and b/images/0003_singular_value_decomposition_1.bmp differ diff --git a/images/0003_singular_value_decomposition_128.bmp b/images/0003_singular_value_decomposition_128.bmp index 07f0fc2..49075be 100644 Binary files a/images/0003_singular_value_decomposition_128.bmp and b/images/0003_singular_value_decomposition_128.bmp differ diff --git a/images/0003_singular_value_decomposition_16.bmp b/images/0003_singular_value_decomposition_16.bmp index 4919817..86ad86d 100644 Binary files a/images/0003_singular_value_decomposition_16.bmp and b/images/0003_singular_value_decomposition_16.bmp differ diff --git a/images/0003_singular_value_decomposition_2.bmp b/images/0003_singular_value_decomposition_2.bmp index e7cf613..a404e5c 100644 Binary files a/images/0003_singular_value_decomposition_2.bmp and b/images/0003_singular_value_decomposition_2.bmp differ diff --git a/images/0003_singular_value_decomposition_256.bmp b/images/0003_singular_value_decomposition_256.bmp index 127600f..19d9047 100644 Binary files a/images/0003_singular_value_decomposition_256.bmp and b/images/0003_singular_value_decomposition_256.bmp differ diff --git a/images/0003_singular_value_decomposition_32.bmp b/images/0003_singular_value_decomposition_32.bmp index 4d9726f..7986b0d 100644 Binary files a/images/0003_singular_value_decomposition_32.bmp and b/images/0003_singular_value_decomposition_32.bmp differ diff --git a/images/0003_singular_value_decomposition_4.bmp b/images/0003_singular_value_decomposition_4.bmp index 512cbb8..a32d75c 100644 Binary files a/images/0003_singular_value_decomposition_4.bmp and b/images/0003_singular_value_decomposition_4.bmp differ diff --git a/images/0003_singular_value_decomposition_64.bmp b/images/0003_singular_value_decomposition_64.bmp index f6d4002..30bb57a 100644 Binary files a/images/0003_singular_value_decomposition_64.bmp and b/images/0003_singular_value_decomposition_64.bmp differ diff --git a/images/0003_singular_value_decomposition_8.bmp b/images/0003_singular_value_decomposition_8.bmp index 095f6fa..b8133bf 100644 Binary files a/images/0003_singular_value_decomposition_8.bmp and b/images/0003_singular_value_decomposition_8.bmp differ diff --git a/images/0004_lu_decomposition.bmp b/images/0004_lu_decomposition.bmp index bbda048..b273d62 100644 Binary files a/images/0004_lu_decomposition.bmp and b/images/0004_lu_decomposition.bmp differ diff --git a/images/1_0001_singular_value_decomposition.bmp b/images/1_0001_singular_value_decomposition.bmp index 09b5857..3dd40b4 100644 Binary files a/images/1_0001_singular_value_decomposition.bmp and b/images/1_0001_singular_value_decomposition.bmp differ diff --git a/images/1_0002_singular_value_decomposition.bmp b/images/1_0002_singular_value_decomposition.bmp index 09b5857..3dd40b4 100644 Binary files a/images/1_0002_singular_value_decomposition.bmp and b/images/1_0002_singular_value_decomposition.bmp differ diff --git a/images/1_0003_singular_value_decomposition_1.bmp b/images/1_0003_singular_value_decomposition_1.bmp index 1f8d26c..02b09ab 100644 Binary files a/images/1_0003_singular_value_decomposition_1.bmp and b/images/1_0003_singular_value_decomposition_1.bmp differ diff --git a/images/1_0003_singular_value_decomposition_128.bmp b/images/1_0003_singular_value_decomposition_128.bmp index 1f930c9..8de7266 100644 Binary files a/images/1_0003_singular_value_decomposition_128.bmp and b/images/1_0003_singular_value_decomposition_128.bmp differ diff --git a/images/1_0003_singular_value_decomposition_16.bmp b/images/1_0003_singular_value_decomposition_16.bmp index b0d9d2b..ca6bd80 100644 Binary files a/images/1_0003_singular_value_decomposition_16.bmp and b/images/1_0003_singular_value_decomposition_16.bmp differ diff --git a/images/1_0003_singular_value_decomposition_2.bmp b/images/1_0003_singular_value_decomposition_2.bmp index e9fafa6..ddff667 100644 Binary files a/images/1_0003_singular_value_decomposition_2.bmp and b/images/1_0003_singular_value_decomposition_2.bmp differ diff --git a/images/1_0003_singular_value_decomposition_256.bmp b/images/1_0003_singular_value_decomposition_256.bmp index 06a1c32..73c0d14 100644 Binary files a/images/1_0003_singular_value_decomposition_256.bmp and b/images/1_0003_singular_value_decomposition_256.bmp differ diff --git a/images/1_0003_singular_value_decomposition_32.bmp b/images/1_0003_singular_value_decomposition_32.bmp index 2663924..b455047 100644 Binary files a/images/1_0003_singular_value_decomposition_32.bmp and b/images/1_0003_singular_value_decomposition_32.bmp differ diff --git a/images/1_0003_singular_value_decomposition_4.bmp b/images/1_0003_singular_value_decomposition_4.bmp index e67b12b..eb260a0 100644 Binary files a/images/1_0003_singular_value_decomposition_4.bmp and b/images/1_0003_singular_value_decomposition_4.bmp differ diff --git a/images/1_0003_singular_value_decomposition_64.bmp b/images/1_0003_singular_value_decomposition_64.bmp index 92a5d8b..253f78e 100644 Binary files a/images/1_0003_singular_value_decomposition_64.bmp and b/images/1_0003_singular_value_decomposition_64.bmp differ diff --git a/images/1_0003_singular_value_decomposition_8.bmp b/images/1_0003_singular_value_decomposition_8.bmp index 1043a67..b152a72 100644 Binary files a/images/1_0003_singular_value_decomposition_8.bmp and b/images/1_0003_singular_value_decomposition_8.bmp differ diff --git a/matrix.hpp b/matrix.hpp index 603a33c..534bf2b 100644 --- a/matrix.hpp +++ b/matrix.hpp @@ -1,16 +1,22 @@ #ifndef FENG_MATRIX_HPP_INCLUDED_ #define FENG_MATRIX_HPP_INCLUDED_ -static_assert( __cplusplus >= 201709L, "C++20 is a must for this library, please update your compiler, or enable corresponding option such as -std=c++2a" ); +static_assert( __cplusplus >= 202002L, "C++20 is a must for this library, please update your compiler, or enable corresponding option such as -std=c++2a" ); #include #include +#include +#include +#include +#include #include +#include #include #include #include #include #include +#include #include #include #include @@ -32,23 +38,43 @@ static_assert( __cplusplus >= 201709L, "C++20 is a must for this library, please #include #include #include +#include #include #include #include #include +#include #include #include +#include +#if defined( __cpp_lib_mdspan ) +#include +#endif +#if defined( __cpp_lib_expected ) +#include +#endif + +// S9-R3 (D-033): configuration macros. FENG_MATRIX_PARALLEL runs elementwise loops in parallel, FENG_MATRIX_OPENCV +// enables the cv::Mat interface, FENG_MATRIX_CHECKED_ITERATORS (D-020) bounds-checks view iterators. The old names +// PARALLEL and OPENCV are deprecated and still accepted: each maps to its FENG_MATRIX_ name here, and the library +// below tests only the new names. +#if defined( PARALLEL ) && !defined( FENG_MATRIX_PARALLEL ) +#define FENG_MATRIX_PARALLEL +#endif +#if defined( OPENCV ) && !defined( FENG_MATRIX_OPENCV ) +#define FENG_MATRIX_OPENCV +#endif -#ifdef OPENCV // Interfacing cv::Mat. Enable this feature by passing `-DOPENCV` option to g++. +#ifdef FENG_MATRIX_OPENCV // Interfacing cv::Mat. Enable this feature by passing `-DFENG_MATRIX_OPENCV` to the compiler. #include #include -#endif//OPENCV +#endif//FENG_MATRIX_OPENCV namespace feng { constexpr std::uint_least64_t matrix_version = 20240314ULL; - #ifdef PARALLEL + #ifdef FENG_MATRIX_PARALLEL constexpr std::uint_least64_t parallel_mode = 1; #else constexpr std::uint_least64_t parallel_mode = 0; @@ -60,39 +86,176 @@ namespace feng constexpr std::uint_least64_t debug_mode = 1; #endif - #ifdef OPENCV + #ifdef FENG_MATRIX_OPENCV constexpr std::uint_least64_t enable_cv_mat = 1; #else constexpr std::uint_least64_t enable_cv_mat = 0; #endif + // S9-R3 (D-033): the configuration (parallel, checked iterators, OpenCV) names a namespace holding one inline + // variable, kept even in a TU that uses nothing from the library. On ELF, unless FENG_MATRIX_NO_CONFIG_CHECK, an + // extern "C" alias to that variable sits in the variable's COMDAT group: TUs with the same configuration fold + // into one group, TUs with different ones each keep theirs and the link fails with a multiple definition of + // feng_matrix_configuration_mismatch_between_translation_units. NDEBUG is not part of the configuration. + #ifdef FENG_MATRIX_PARALLEL + #define FENG_MATRIX_CONFIG_P_ 1 + #else + #define FENG_MATRIX_CONFIG_P_ 0 + #endif + #ifdef FENG_MATRIX_CHECKED_ITERATORS + #define FENG_MATRIX_CONFIG_C_ 1 + #else + #define FENG_MATRIX_CONFIG_C_ 0 + #endif + #ifdef FENG_MATRIX_OPENCV + #define FENG_MATRIX_CONFIG_O_ 1 + #else + #define FENG_MATRIX_CONFIG_O_ 0 + #endif + // matrix_config_p

_c_o has 22 characters, hence the Itanium name _ZN4feng22matrix_config_p

_c_o3tagE. + #define FENG_MATRIX_CONFIG_NS2_( P, C, O ) matrix_config_p ## P ## _c ## C ## _o ## O + #define FENG_MATRIX_CONFIG_NS_( P, C, O ) FENG_MATRIX_CONFIG_NS2_( P, C, O ) + #define FENG_MATRIX_CONFIG_MANGLED2_( P, C, O ) _ZN4feng22matrix_config_p ## P ## _c ## C ## _o ## O ## 3tagE + #define FENG_MATRIX_CONFIG_MANGLED_( P, C, O ) FENG_MATRIX_CONFIG_MANGLED2_( P, C, O ) + #define FENG_MATRIX_CONFIG_STR2_( X ) #X + #define FENG_MATRIX_CONFIG_STR_( X ) FENG_MATRIX_CONFIG_STR2_( X ) + + namespace FENG_MATRIX_CONFIG_NS_( FENG_MATRIX_CONFIG_P_, FENG_MATRIX_CONFIG_C_, FENG_MATRIX_CONFIG_O_ ) + { + [[gnu::used]] inline char tag = 0; + } +}//namespace feng + +#if defined( __ELF__ ) && !defined( FENG_MATRIX_NO_CONFIG_CHECK ) +extern "C" char feng_matrix_configuration_mismatch_between_translation_units + __attribute__(( alias( FENG_MATRIX_CONFIG_STR_( FENG_MATRIX_CONFIG_MANGLED_( FENG_MATRIX_CONFIG_P_, FENG_MATRIX_CONFIG_C_, FENG_MATRIX_CONFIG_O_ ) ) ) )); +#endif + +#undef FENG_MATRIX_CONFIG_STR_ +#undef FENG_MATRIX_CONFIG_STR2_ +#undef FENG_MATRIX_CONFIG_MANGLED_ +#undef FENG_MATRIX_CONFIG_MANGLED2_ +#undef FENG_MATRIX_CONFIG_NS_ +#undef FENG_MATRIX_CONFIG_NS2_ +#undef FENG_MATRIX_CONFIG_O_ +#undef FENG_MATRIX_CONFIG_C_ +#undef FENG_MATRIX_CONFIG_P_ + +namespace feng +{ + namespace matrix_private - { // for macro `better_assert` + { + // S2-R1 (D-003, D-011): the one contract-violation path. Formats a single line, writes it to stderr with + // one fwrite, flushes and aborts; used in every build mode, never throws. template< typename... Args > - void print_assertion(std::ostream& out, Args&&... args) + [[noreturn]] void contract_violation( char const* expr, char const* file, int line, Args const&... detail ) noexcept { - if constexpr( debug_mode ) + std::ostringstream os; + os.precision( 20 ); + os << "feng::matrix: contract violation: " << expr << " (" << file << ":" << line << ")"; + if constexpr( sizeof...( Args ) > 0 ) { - out.precision( 20 ); - (out << ... << args) << std::endl; - abort(); + os << ' '; + ( os << ... << detail ); } + os << '\n'; + std::string const msg = os.str(); + std::fwrite( msg.data(), 1, msg.size(), stderr ); + std::fflush( stderr ); + std::abort(); } } + #ifdef FENG_MATRIX_EXPECTS + #undef FENG_MATRIX_EXPECTS + #endif + // FENG_MATRIX_EXPECTS( expr, detail... ): when `expr` is false, report it and abort, in debug and NDEBUG builds. + #define FENG_MATRIX_EXPECTS(EXPRESSION, ... ) ((EXPRESSION) ? (void)0 : ::feng::matrix_private::contract_violation( #EXPRESSION, __FILE__, __LINE__ __VA_OPT__(,) __VA_ARGS__ )) + #ifdef better_assert #undef better_assert #endif // - // enhancing 'assert' macro, usage: + // legacy name (D-002), forwards to FENG_MATRIX_EXPECTS, usage: // - // int a; - // ... - // better_assert( a > 0 ); //same as 'assert' - // better_assert( a > 0, "a is expected larger than 0, but now a = " a ); //with more info dumped to std::cerr + // better_assert( a > 0 ); + // better_assert( a > 0, "a is expected larger than 0, but now a = ", a ); // detail appended to the message // - #define better_assert(EXPRESSION, ... ) ((EXPRESSION) ? (void)0 : matrix_private::print_assertion(std::cerr, "[Assertion Failure]: '", #EXPRESSION, "' in File: ", __FILE__, " in Line: ", __LINE__ __VA_OPT__(,) __VA_ARGS__)) + #define better_assert(EXPRESSION, ... ) FENG_MATRIX_EXPECTS( EXPRESSION __VA_OPT__(,) __VA_ARGS__ ) + + namespace matrix_private + { + // S2-R2 (D-012): element count for a rows x cols allocation; aborts on rows*cols overflow, on a byte count + // that overflows or exceeds PTRDIFF_MAX (F02), or above allocator_traits::max_size, before anything + // is allocated. + template< typename Alloc > + std::size_t checked_count( Alloc const& alloc, std::size_t r, std::size_t c ) noexcept + { + typedef typename std::allocator_traits< Alloc >::value_type value_type; + constexpr std::size_t size_max = std::numeric_limits< std::size_t >::max(); + FENG_MATRIX_EXPECTS( r == 0 || c <= size_max / r, "matrix size: rows*cols overflows, rows = ", r, ", cols = ", c ); + std::size_t const n = r * c; + FENG_MATRIX_EXPECTS( n <= size_max / sizeof( value_type ), "matrix size: byte count overflows, rows = ", r, ", cols = ", c ); + FENG_MATRIX_EXPECTS( n * sizeof( value_type ) <= static_cast< std::size_t >( PTRDIFF_MAX ), "matrix size: byte count above PTRDIFF_MAX, rows = ", r, ", cols = ", c ); + FENG_MATRIX_EXPECTS( n <= static_cast< std::size_t >( std::allocator_traits< Alloc >::max_size( alloc ) ), "matrix size: rows*cols above the allocator's max_size, rows = ", r, ", cols = ", c ); + return n; + } + + // S3-R5: checks a rows x cols allocation against alloc, then returns alloc for the owning container. + template< typename Alloc > + Alloc checked_allocator( Alloc alloc, std::size_t r, std::size_t c ) noexcept + { + checked_count( alloc, r, c ); + return alloc; + } + + // S3-R1: the one access point to matrix's private storage and extents, used by the CRTP bases. + struct storage_access + { + template< typename M > + static auto& storage( M& m ) noexcept { return m.storage_; } + template< typename M > + static auto& rows( M& m ) noexcept { return m.row_; } + template< typename M > + static auto& cols( M& m ) noexcept { return m.col_; } + }; + + // S2-R2: a signed or unsigned dimension is usable when it is non-negative and fits std::size_t. + template< std::integral I > + constexpr bool dimension_fits( I v ) noexcept + { + if constexpr( std::is_signed_v< I > ) + if ( v < 0 ) return false; + return static_cast< std::uintmax_t >( v ) <= std::numeric_limits< std::size_t >::max(); + } + + // S2-R2: clone bounds, checked before the source is read or anything is allocated. + inline void check_clone_lists( std::size_t rows_listed, std::size_t cols_listed ) noexcept + { + FENG_MATRIX_EXPECTS( rows_listed == 2, "matrix clone: the row range needs exactly two values, got ", rows_listed ); + FENG_MATRIX_EXPECTS( cols_listed == 2, "matrix clone: the column range needs exactly two values, got ", cols_listed ); + } + inline void check_clone_range( std::size_t r0, std::size_t r1, std::size_t c0, std::size_t c1, std::size_t rows, std::size_t cols ) noexcept + { + FENG_MATRIX_EXPECTS( r0 < r1 && r1 <= rows, "matrix clone: row range [", r0, ", ", r1, ") outside a source with ", rows, " rows" ); + FENG_MATRIX_EXPECTS( c0 < c1 && c1 <= cols, "matrix clone: column range [", c0, ", ", c1, ") outside a source with ", cols, " columns" ); + } + + // S2-R2: the one two-index check behind at( r, c ) and m( r, c ); aborts with `index` when out of range. + inline void check_index( std::size_t r, std::size_t c, std::size_t rows, std::size_t cols ) noexcept + { + FENG_MATRIX_EXPECTS( r < rows && "Row index out of boundary!", "matrix index: (", r, ", ", c, ") on a ", rows, "x", cols, " matrix" ); + FENG_MATRIX_EXPECTS( c < cols && "Column index out of boundary!", "matrix index: (", r, ", ", c, ") on a ", rows, "x", cols, " matrix" ); + } + + // S3-R3 (D-018): true for std::complex specializations. + template< typename T > + inline constexpr bool is_std_complex_v = false; + template< typename T > + inline constexpr bool is_std_complex_v< std::complex< T > > = true; + } // // begin of concept allocators @@ -133,9 +296,26 @@ namespace feng // - template < typename Type, Allocator Alloc > + // S3-R3 (PR-5, D-012): element types of matrix and matrix_view; bool is excluded (no contiguous T* storage). + template < typename T > + concept matrix_element = std::is_object_v< T > && !std::is_const_v< T > && !std::is_volatile_v< T > && + !std::is_same_v< T, bool > && + // D-018: the size constructors default-construct elements. std::complex is carved out + // because libstdc++ declares complex() without noexcept, though it only value-initializes two arithmetic members. + ( std::is_nothrow_default_constructible_v< T > || matrix_private::is_std_complex_v< T > ) && + std::is_nothrow_copy_constructible_v< T > && std::is_nothrow_move_constructible_v< T > && + std::is_nothrow_copy_assignable_v< T > && std::is_nothrow_move_assignable_v< T > && + std::is_nothrow_destructible_v< T >; + + template < matrix_element Type, Allocator Alloc > struct matrix; + template < matrix_element Type, Allocator Alloc > + struct matrix_view; + + template < matrix_element Type, Allocator Alloc > + struct mutable_matrix_view; + namespace matrix_details { @@ -147,7 +327,7 @@ namespace feng typedef Integer_Type value_type; typedef value_type& reference; typedef value_type* pointer; - typedef const reference const_reference; + typedef reference const_reference; // same type as before: const on a reference is ignored typedef std::random_access_iterator_tag iterator_category; typedef std::ptrdiff_t difference_type; typedef integer_iterator self_type; @@ -188,7 +368,7 @@ namespace feng return *this; } - self_type const operator++(int) noexcept + self_type operator++(int) noexcept { self_type ans{*this}; ++(*this); @@ -200,12 +380,12 @@ namespace feng return lhs.value_ - rhs.value_; } - friend self_type const operator + ( self_type const& lhs, difference_type const& rhs ) noexcept + friend self_type operator + ( self_type const& lhs, difference_type const& rhs ) noexcept { return self_type{ lhs.value_ + rhs }; } - friend self_type const operator + ( difference_type const& lhs, self_type const& rhs ) noexcept + friend self_type operator + ( difference_type const& lhs, self_type const& rhs ) noexcept { return rhs + lhs; } @@ -243,93 +423,162 @@ namespace feng template< std::weakly_incrementable W > - constexpr auto range( W val_begin, W val_end ) + constexpr auto range( W val_begin, W val_end ) noexcept { return std::ranges::iota_view( val_begin, val_end ); } template< std::weakly_incrementable W > - constexpr auto range( W val_end ) + constexpr auto range( W val_end ) noexcept { return range( W{0}, val_end ); } template - void repeat( Function function, Integer_Type n ) + void repeat( Function function, Integer_Type n ) noexcept { while ( n-- ) function(); } + // S6-R1 (F11, F12): the one partition behind parallel and both reductions. For n = last - first > 0 the + // effective worker count is w = clamp( workers, 1, n ) (0 means serial); an empty or reversed range has + // w = 0. Chunk k holds n / w + ( k < n % w ) indices starting at first + k * ( n / w ) + min( k, n % w ), + // so the w chunks are contiguous, disjoint, non-empty and cover [first, last) in order. + template< std::integral Integer_Type > + constexpr std::size_t effective_workers( Integer_Type first, Integer_Type last, std::size_t workers ) noexcept + { + if ( !( first < last ) ) return 0; + std::size_t const n = static_cast( last - first ); + return std::clamp( workers, std::size_t{1}, n ); + } + + // [begin, end) of chunk k; an empty pair for an empty range or k >= effective_workers( first, last, workers ). + template< std::integral Integer_Type > + constexpr std::pair chunk_bounds( Integer_Type first, Integer_Type last, std::size_t workers, std::size_t k ) noexcept + { + std::size_t const w = effective_workers( first, last, workers ); + if ( k >= w ) return { first < last ? last : first, first < last ? last : first }; + std::size_t const n = static_cast( last - first ); + std::size_t const q = n / w; + std::size_t const r = n % w; + std::size_t const offset = k * q + std::min( k, r ); + Integer_Type const b = static_cast( first + static_cast( offset ) ); + Integer_Type const e = static_cast( b + static_cast( q + ( k < r ? 1 : 0 ) ) ); + return { b, e }; + } + + // Runs func( i ) once for every i in [first, last): chunk 0 on the calling thread, chunks 1..w-1 on + // std::jthreads that are all joined before return. The injected count is honoured in every build. A + // thread-creation failure terminates (noexcept, D-012); func must not throw. template< typename Function, std::integral Integer_Type > - void parallel( Function const& func, Integer_Type dim_first, Integer_Type dim_last, unsigned long threshold = 1024 ) // 1d parallel + void parallel_workers( Function const& func, Integer_Type first, Integer_Type last, std::size_t workers ) noexcept { - if constexpr( parallel_mode == 0 ) + std::size_t const w = effective_workers( first, last, workers ); + if ( w == 0 ) return; + auto const run_chunk = [&func, first, last, w]( std::size_t k ) noexcept + { + auto const [b, e] = chunk_bounds( first, last, w, k ); + for ( Integer_Type i = b; i != e; ++i ) + func( i ); + }; + if ( w == 1 ) { - for ( auto a : range( dim_first, dim_last ) ) - func( a ); + run_chunk( 0 ); return; } - else // <- this is constexpr-if, `else` is a must - { - unsigned int const total_cores = std::thread::hardware_concurrency(); - - // case of non-parallel or small jobs - if ( (total_cores <= 1) || ((dim_last - dim_first) <= threshold) ) - { - for ( auto a : range( dim_first, dim_last ) ) - func( a ); - return; - } - - // case of small job numbers - std::vector threads; - if ( dim_last - dim_first <= total_cores ) - { - for ( auto index = dim_first; index != dim_last; ++index ) - threads.emplace_back( std::thread{[&func, index](){ func( index ); }} ); - for ( auto& th : threads ) - th.join(); - return; - } + std::vector threads; + threads.reserve( w - 1 ); + for ( std::size_t k = 1; k != w; ++k ) + threads.emplace_back( run_chunk, k ); + run_chunk( 0 ); + for ( auto& th : threads ) + th.join(); + } - // case of more jobs than CPU cores - auto const& job_slice = [&func]( Integer_Type a, Integer_Type b ) - { - if ( a >= b ) return; - while ( a != b ) - func(a++); - }; + // 1 in serial builds or for n <= threshold, else hardware_concurrency (1 if it reports 0). + inline std::size_t default_workers( std::size_t n, std::size_t threshold ) noexcept + { + if constexpr( parallel_mode == 0 ) + { + (void)n; (void)threshold; + return 1; + } + else + { + if ( n <= threshold ) return 1; + unsigned int const cores = std::thread::hardware_concurrency(); + return cores == 0 ? std::size_t{1} : static_cast( cores ); + } + } - threads.reserve( total_cores-1 ); - std::uint_least64_t tasks_per_thread = ( dim_last - dim_first + total_cores - 1 ) / total_cores; + // S10-R3 (D-008, kept optimization par-thresholds): work-based default worker counts. In parallel builds a + // call doing `work` units runs on min( hardware_concurrency, work / grain ) workers, at least 1, so a call + // with less than two grains of work stays on the calling thread; serial builds always return 1. Explicit + // worker counts (parallel_workers, reduce_range, the trailing reduce argument) are not affected. + // Grains, measured on the S10 host (Ryzen 9 7900X3D, 24 threads, g++ -O2 -pthread; lane numbers in + // bench/results.md "par-thresholds"): starting and joining one std::jthread costs about 19 us, so a worker + // pays off only once it takes over more than that much work. Per-call medians, 1 worker against the best + // count, warm inputs: sqrt via for_each 2^15 doubles 54 us (1) vs 58 us (2), 2^17 211 us vs 111 us (4); sum + // 2^16 38 us (1) vs 49 us (2), 2^18 151 us vs 90 us (3); GEMM 64^3 (2^18 multiply-adds) 51 us (1) vs 58 us + // (3), 96^3 171 us vs 105 us (4), 1024x1024 by 1024x1 171 us vs 105 us (4). The elementwise grain comes from + // the bench workload par/add/1000x1000 (a + b into a fresh result): grain 2^16 (15 workers) 0.75-0.89 ms, + // 2^17 (7) 0.84-0.90 ms, 2^18 (3) 0.98-1.06 ms, 24 workers (pre-S10) 0.98-1.06 ms; for map/1000x1000 a + // callback grain of 2^16 (15 workers) beat 2^13-2^15 (24). + inline constexpr std::size_t elementwise_grain = std::size_t{ 1 } << 16; // elements: add, minus, negate, copy, clone + inline constexpr std::size_t callback_grain = std::size_t{ 1 } << 16; // elements: for_each, apply, colormap, pooling + inline constexpr std::size_t reduce_grain = std::size_t{ 1 } << 16; // elements folded: reduce, sum, min, max + inline constexpr std::size_t gemm_grain = std::size_t{ 1 } << 18; // multiply-adds M*K*N; workers <= M rows + + inline std::size_t work_workers( std::size_t work, std::size_t grain ) noexcept + { + if constexpr( parallel_mode == 0 ) + { + (void)work; (void)grain; + return 1; + } + else + { + std::size_t const want = work / std::max( grain, std::size_t{ 1 } ); + if ( want <= 1 ) return 1; + unsigned int const cores = std::thread::hardware_concurrency(); + std::size_t const cap = cores == 0 ? std::size_t{ 1 } : static_cast( cores ); + return std::min( want, cap ); + } + } - for ( auto index : range( total_cores-1 ) ) - { - Integer_Type first = tasks_per_thread * index + dim_first; - first = std::min( first, dim_last ); - Integer_Type last = first + tasks_per_thread; - last = std::min( last, dim_last ); - threads.emplace_back( std::thread{ job_slice, first, last } ); - } + // a * b, saturated at SIZE_MAX (work amounts only). + constexpr std::size_t saturating_work( std::size_t a, std::size_t b ) noexcept + { + if ( a != 0 && b > std::numeric_limits::max() / a ) return std::numeric_limits::max(); + return a * b; + } - job_slice( tasks_per_thread*(total_cores-1), dim_last ); + // Runs func( i ) for i in [first, last) on work_workers( ( last - first ) * work_per_index, grain ) workers. + template< typename Function, std::integral Integer_Type > + void parallel_work( Function const& func, Integer_Type first, Integer_Type last, std::size_t work_per_index, std::size_t grain ) noexcept + { + std::size_t const n = first < last ? static_cast( last - first ) : 0; + parallel_workers( func, first, last, work_workers( saturating_work( n, work_per_index ), grain ) ); + } - for ( auto& th : threads ) - th.join(); - } + template< typename Function, std::integral Integer_Type > + void parallel( Function const& func, Integer_Type dim_first, Integer_Type dim_last, unsigned long threshold = 1024 ) noexcept // 1d parallel + { + std::size_t const n = dim_first < dim_last ? static_cast( dim_last - dim_first ) : 0; + parallel_workers( func, dim_first, dim_last, default_workers( n, threshold ) ); } template< typename Function, typename Integer_Type > - void parallel( Function const& func, Integer_Type dim_last ) + void parallel( Function const& func, Integer_Type dim_last ) noexcept { parallel( func, Integer_Type{0}, dim_last ); } namespace bmp_details { - inline std::vector const generate_bmp_header( std::uint_least64_t const the_row, std::uint_least64_t const the_col ) + inline std::vector generate_bmp_header( std::uint_least64_t const the_row, std::uint_least64_t const the_col ) noexcept { auto const& ul_to_byte = []( std::uint_least64_t val ) { return static_cast< std::uint8_t >( val & 0xffUL ); }; std::uint8_t file[14] = { 0x42, 0x4D, 0, 0, 0, 0, 0, 0, 0, 0, 54, 0, 0, 0 }; @@ -359,19 +608,19 @@ namespace feng return header; } - inline std::uint8_t operator "" _u8(unsigned long long value) + inline std::uint8_t operator""_u8(unsigned long long value) noexcept { return static_cast(value); } static std::function(double)> make_transformation_function( std::tuple const& color_1, double value_1, - std::tuple const& color_2, double value_2 ) + std::tuple const& color_2, double value_2 ) noexcept { auto const [r1, g1, b1] = color_1; auto const [r2, g2, b2] = color_2; - return [r1=r1, g1=g1, b1=b1, r2=r2, g2=g2, b2=b2, value_1, value_2]( double x ) + return [r1=r1, g1=g1, b1=b1, r2=r2, g2=g2, b2=b2, value_1, value_2]( double x ) noexcept { double dr = static_cast(r1) - static_cast(r2); double dg = static_cast(g1) - static_cast(g2); @@ -392,13 +641,13 @@ namespace feng //static std::tuple static std::function(double)> - make_color_map( std::vector const& values, std::vector> const& colors ) + make_color_map( std::vector const& values, std::vector> const& colors ) noexcept { better_assert( (values.size() == colors.size()), "make_color_map::length of values and colors not match!" ); better_assert( std::abs(*(values.begin())) < 1.0e-10 && "make_color_map::value should start from 0!" ); better_assert( std::abs(*(values.rbegin())-1.0) < 1.0e-10 && "make_color_map::value should end at 1!" ); - return [=]( double x ) + return [=]( double x ) noexcept { for ( auto index : matrix_details::range( values.size() - 1 ) ) if ( x <= values[index+1] ) @@ -408,7 +657,7 @@ namespace feng }; } static std::function(double)> - make_color_map( std::initializer_list const& values, std::initializer_list> const& colors ) + make_color_map( std::initializer_list const& values, std::initializer_list> const& colors ) noexcept { return make_color_map( std::vector{values}, std::vector>{colors} ); } @@ -1065,7 +1314,7 @@ namespace feng *start_pos++ = channel_r[row_index][c]; } }; - matrix_details::parallel( fill_row, 0UL, the_row, 0UL ); + matrix_details::parallel_work( fill_row, 0UL, the_row, saturating_work( the_col, 3 ), elementwise_grain ); return {encoding}; } @@ -1100,15 +1349,15 @@ namespace feng }; template < typename Function, typename InputIterator1, typename... InputIteratorn > - Function _for_each_n( Function f, std::uint_least64_t n, InputIterator1 begin1, InputIteratorn... beginn ) + Function _for_each_n( Function f, std::uint_least64_t n, InputIterator1 begin1, InputIteratorn... beginn ) noexcept { auto const& func = [&]( std::uint_least64_t idx ) { f( *(begin1+idx), *(beginn+idx)... ); }; - parallel( func, 0UL, n ); + parallel_work( func, std::uint_least64_t{ 0 }, n, 1, callback_grain ); return f; } template < typename Function, typename InputIterator1, typename... InputIteratorn > - Function _for_each( Function f, InputIterator1 begin1, InputIterator1 end1, InputIteratorn... beginn ) + Function _for_each( Function f, InputIterator1 begin1, InputIterator1 end1, InputIteratorn... beginn ) noexcept { return _for_each_n( f, std::distance( begin1, end1 ), begin1, beginn... ); } @@ -1121,13 +1370,13 @@ namespace feng typedef typename extract_type_backward< 1, Types_N... >::result_type return_type; template < typename Predict, typename... Types > - Predict impl( Predict p, dummy, Types... types ) const + Predict impl( Predict p, dummy, Types... types ) const noexcept { return _for_each( p, types... ); } template < typename S, typename... Types > - return_type impl( S s, Types... types ) const + return_type impl( S s, Types... types ) const noexcept { return impl( types..., s ); } @@ -1136,180 +1385,977 @@ namespace feng template < typename... Types > typename for_each_impl_private::extract_type_backward< 1, Types... >::result_type - for_each( Types... types ) + for_each( Types... types ) noexcept { static_assert( sizeof...( types ) > 2, "f::for_each requires at least 3 arguments" ); return for_each_impl_private::for_each_impl_with_dummy< Types... >().impl( types..., for_each_impl_private::dummy() ); } + // S6-R1 (F11, F12): the one reduction. at( i ) reads element i of [0, n). Each chunk of the shared partition + // folds from its own first element; the result is init folded with the chunk results in order, so init + // appears exactly once and any associative func gives std::accumulate's result. Only at( i ) for i < n is + // read, and no count is ever divided by. + template< typename Result, typename At, typename Function > + Result reduce_range( At const& at, std::size_t n, Result init, Function const& func, std::size_t workers ) noexcept + { + std::size_t const w = effective_workers( std::size_t{0}, n, workers ); + Result ans = std::move( init ); + if ( w <= 1 ) + { + for ( std::size_t i = 0; i != n; ++i ) + ans = func( ans, at( i ) ); + return ans; + } + std::vector cache( w ); + auto const chunk_func = [&]( std::size_t k ) noexcept + { + auto const [b, e] = chunk_bounds( std::size_t{0}, n, w, k ); + Result acc = at( b ); + for ( std::size_t i = b + 1; i != e; ++i ) + acc = func( acc, at( i ) ); + cache[k] = std::move( acc ); + }; + parallel_workers( chunk_func, std::size_t{0}, w, w ); + for ( std::size_t k = 0; k != w; ++k ) + ans = func( ans, cache[k] ); + return ans; + } + + template< typename Iterator > + std::size_t iterator_range_size( Iterator begin, Iterator end ) noexcept + { + auto const d = std::distance( begin, end ); + return d > 0 ? static_cast( d ) : std::size_t{0}; + } + + // Iterator reduce over [begin, end) with an injected worker count; the overload below defaults it to + // work_workers( n, reduce_grain ) (S10-R3; before S10 default_workers( n, 32 )). template typename std::invoke_result::value_type, typename std::iterator_traits::value_type>::type - reduce( Iterator begin, Iterator end, typename std::iterator_traits::value_type init, Function const& func ) + reduce( Iterator begin, Iterator end, typename std::iterator_traits::value_type init, Function const& func, std::size_t workers ) noexcept { typedef typename std::iterator_traits::value_type value_type; typedef typename std::invoke_result::type result_type; + std::size_t const n = iterator_range_size( begin, end ); + auto const at = [begin]( std::size_t i ) noexcept -> decltype( auto ) { return *std::next( begin, static_cast::difference_type>( i ) ); }; + return reduce_range( at, n, static_cast( std::move( init ) ), func, workers ); + } - unsigned int const total_cores = std::thread::hardware_concurrency(); - unsigned long const total_elements = std::distance( begin, end ); - - // case of small size, reduce inplace - if (total_elements <= total_cores) - return std::accumulate( begin, end, init, func ); - - // parallel reduce to cache - //std::vector cache{ static_cast(total_cores), static_cast(init) };// of size `total_cores`, of same value `init` - std::vector cache; - cache.resize( total_cores ); - auto const block_size = total_elements / total_cores; - auto const& thread_reducer = [&]( unsigned long idx ) - { - auto const block_begin = idx * block_size; - auto const block_end = idx + 1 != total_cores ? block_begin + block_size : total_elements; - result_type temp = *(begin+block_begin); - for ( auto id : range( block_begin + 1, block_end ) ) - temp = func( temp, *(begin+id) ); - cache[idx] = temp; - }; - parallel( thread_reducer, static_cast(0), total_cores, 0UL ); - - // inplace reduce from cache - return std::accumulate( cache.begin(), cache.end(), init, func ); + template + typename std::invoke_result::value_type, typename std::iterator_traits::value_type>::type + reduce( Iterator begin, Iterator end, typename std::iterator_traits::value_type init, Function const& func ) noexcept + { + std::size_t const n = matrix_details::iterator_range_size( begin, end ); + return reduce( begin, end, std::move( init ), func, work_workers( n, reduce_grain ) ); } - static bool create_directory_if_not_present( std::string const& file_name ) + inline bool create_directory_if_not_present( std::string const& file_name ) noexcept { + // S5-R4 (D-011, D-012): error_code overloads only; false on failure, never throws + std::error_code ec; std::filesystem::path file_path{ file_name }; auto directory = file_path.parent_path(); if ( directory.empty() ) return true; - if ( ! std::filesystem::exists(directory) ) return std::filesystem::create_directories(directory); - return true; + if ( std::filesystem::is_directory( directory, ec ) ) return true; + ec.clear(); + if ( std::filesystem::exists( directory, ec ) || ec ) return false; // a non-directory, or not checkable + std::filesystem::create_directories( directory, ec ); + return !ec && std::filesystem::is_directory( directory, ec ); } - }//namespace matrix_details - - // for column, diagonal and anti-diagonal iteration - template < typename Iterator_Type > - struct stride_iterator - { - typedef stride_iterator self_type; - typedef typename std::iterator_traits< Iterator_Type >::value_type value_type; - typedef typename std::iterator_traits< Iterator_Type >::reference reference; - typedef typename std::iterator_traits< Iterator_Type >::difference_type difference_type; - typedef typename std::iterator_traits< Iterator_Type >::pointer pointer; - typedef typename std::iterator_traits< Iterator_Type >::iterator_category iterator_category; - - Iterator_Type iterator_; - difference_type step_; - - stride_iterator( const Iterator_Type& it, const difference_type& dt ) noexcept : iterator_( it ) , step_( dt ) { } - - stride_iterator() noexcept : iterator_( 0 ) , step_( 1 ) { } - stride_iterator( const self_type& ) noexcept = default; - stride_iterator( self_type&& ) noexcept = default; - self_type& operator=( const self_type& ) noexcept = default; - self_type& operator=( self_type&& ) noexcept = default; - - self_type& operator++() noexcept + // S5-R4 (D-011): prints the one stderr line of a failed writer, " -- : ", and returns false. + inline bool write_failed( char const* who, char const* reason, std::string const& path ) noexcept { - iterator_ += step_; - return *this; + std::cerr << who << " -- " << reason << ": " << path << "\n"; + return false; } - const self_type operator++( int ) noexcept + + // S5-R4: flushes and closes a written stream; true only if every write, the flush and the close succeeded. + inline bool checked_close( std::ofstream& ofs ) noexcept { - self_type ans( *this ); - operator++(); - return ans; + ofs.flush(); + ofs.close(); + return !ofs.fail(); } - self_type& operator+=( const difference_type dt ) noexcept + + // S5-R4: the common tail of a stream writer: a directory, open, write or close failure prints one line naming + // `who` and `path` and returns false. `body` writes into the open stream. + template < typename Body > + bool write_stream( char const* who, std::string const& path, std::ios_base::openmode mode, Body&& body ) noexcept { - iterator_ += dt * step_; - return *this; + if ( !create_directory_if_not_present( path ) ) return write_failed( who, "failed to create the parent directory", path ); + std::ofstream ofs( path, mode ); + if ( !ofs ) return write_failed( who, "failed to open file", path ); + body( ofs ); + if ( !ofs ) return write_failed( who, "failed to write file", path ); + if ( !checked_close( ofs ) ) return write_failed( who, "failed to write or close file", path ); + return true; } - friend const self_type operator+( const self_type& lhs, const difference_type rhs ) noexcept + + // S5-R1/S5-R2: overflow-checked size arithmetic for every external-data size computation; true when the + // result fits in `out` (D-012: no throwing, no wrap-around). + template < std::integral T > + constexpr bool checked_mul( T a, T b, T& out ) noexcept { - self_type ans( lhs ); - ans += rhs; - return ans; + return !__builtin_mul_overflow( a, b, &out ); } - friend const self_type operator+( const difference_type lhs, const self_type& rhs ) noexcept + + template < std::integral T > + constexpr bool checked_add( T a, T b, T& out ) noexcept { - return rhs + lhs; + return !__builtin_add_overflow( a, b, &out ); } - self_type& operator--() noexcept + + // S5-R4 (D-011): reads the whole file into `bytes`; on a directory, an open or a read failure prints one stderr + // line " -- : " and returns false, leaving `bytes` unchanged. Never aborts. + inline bool read_file( char const* path, std::vector< std::uint8_t >& bytes, char const* who = "read_file" ) noexcept { - iterator_ -= step_; - return *this; + if ( path == nullptr ) + { + std::cerr << who << " -- no file name given\n"; + return false; + } + std::error_code ec; + if ( std::filesystem::is_directory( path, ec ) ) + { + std::cerr << who << " -- is a directory: " << path << "\n"; + return false; + } + std::ifstream ifs( path, std::ios::binary ); + if ( !ifs ) + { + std::cerr << who << " -- failed to open file: " << path << "\n"; + return false; + } + std::vector< std::uint8_t > buffer; + std::array< char, 65536 > chunk; + while ( ifs ) + { + ifs.read( chunk.data(), static_cast< std::streamsize >( chunk.size() ) ); + std::streamsize const got = ifs.gcount(); + if ( got > 0 ) + buffer.insert( buffer.end(), reinterpret_cast< std::uint8_t const* >( chunk.data() ), + reinterpret_cast< std::uint8_t const* >( chunk.data() ) + got ); + } + if ( ifs.bad() || !ifs.eof() ) + { + std::cerr << who << " -- failed to read file: " << path << "\n"; + return false; + } + bytes.swap( buffer ); + return true; } - const self_type operator--( int ) noexcept + + // S5-R1 (D-021): the NPY dtype code of an element type; kind 0 means the type has no NPY dtype. + template < typename T > + struct npy_dtype + { + static constexpr char kind = std::is_same_v< T, float > && sizeof( T ) == 4 ? 'f' + : std::is_same_v< T, double > && sizeof( T ) == 8 ? 'f' + : std::is_same_v< T, std::complex< float > > && sizeof( T ) == 8 ? 'c' + : std::is_same_v< T, std::complex< double > > && sizeof( T ) == 16 ? 'c' + : std::is_integral_v< T > && !std::is_same_v< T, bool > && std::is_signed_v< T > + && ( sizeof( T ) == 1 || sizeof( T ) == 2 || sizeof( T ) == 4 || sizeof( T ) == 8 ) ? 'i' + : std::is_integral_v< T > && !std::is_same_v< T, bool > && std::is_unsigned_v< T > + && ( sizeof( T ) == 1 || sizeof( T ) == 2 || sizeof( T ) == 4 || sizeof( T ) == 8 ) ? 'u' + : '\0'; + static constexpr std::size_t component = kind == 'c' ? sizeof( T ) / 2 : sizeof( T ); // bytes swapped as one unit + }; + + // S5-R1: a cursor over the NPY header text; every read checks the bound first. + struct npy_header_cursor { - self_type ans( *this ); - operator--(); - return ans; - } - self_type& operator-=( const difference_type dt ) noexcept + char const* p; + char const* end; + + bool at_end() const noexcept { return p == end; } + char peek() const noexcept { return p == end ? '\0' : *p; } + void skip_ws() noexcept + { + while ( p != end && ( *p == ' ' || *p == '\t' || *p == '\n' || *p == '\r' ) ) ++p; + } + bool eat( char c ) noexcept + { + if ( p == end || *p != c ) return false; + ++p; + return true; + } + // A Python string literal in ' or " quotes without escapes; [b, e) receives its contents. + bool string_literal( char const*& b, char const*& e ) noexcept + { + if ( p == end || ( *p != '\'' && *p != '"' ) ) return false; + char const quote = *p++; + b = p; + while ( p != end && *p != quote ) + { + if ( *p == '\\' || *p == '\n' ) return false; + ++p; + } + if ( p == end ) return false; + e = p++; + return true; + } + bool word( char const* w ) noexcept + { + std::size_t const n = std::strlen( w ); + if ( static_cast< std::size_t >( end - p ) < n || std::memcmp( p, w, n ) != 0 ) return false; + char const* const after = p + n; + if ( after != end && ( std::isalnum( static_cast< unsigned char >( *after ) ) || *after == '_' ) ) return false; + p = after; + return true; + } + }; + + // S5-R1 (F07, D-021): parses a complete NPY v1.0/2.0/3.0 file image of `size` bytes into `out`, a matrix< T, A >. + // Validates the magic, version, header length, the header dict, the dtype (exact match to T), the order and + // the shape before any allocation; the element count times sizeof(T) must equal the payload length exactly. + // Builds the result in a temporary with out's allocator and moves it into `out` only on success; on failure + // returns false, leaves `out` unchanged and, when `why` is given, stores a static reason string there. + template < typename T, typename Out > + bool parse_npy( std::uint8_t const* bytes, std::size_t size, Out& out, char const** why = nullptr ) noexcept { - iterator_ -= dt * step_; - return *this; + static_assert( std::is_same_v< typename Out::value_type, T >, "parse_npy: Out must be a matrix of T" ); + auto fail = [why]( char const* reason ) noexcept + { + if ( why ) *why = reason; + return false; + }; + constexpr char kind = npy_dtype< T >::kind; + if constexpr ( kind == '\0' ) + { + (void)bytes; (void)size; (void)out; + return fail( "the element type has no NPY dtype" ); + } + else + { + static_assert( std::is_trivially_copyable_v< T > ); + if ( bytes == nullptr || size < 10 ) return fail( "file too short for the NPY preamble" ); + if ( std::memcmp( bytes, "\x93NUMPY", 6 ) != 0 ) return fail( "bad NPY magic" ); + std::uint8_t const major = bytes[6]; + std::uint8_t const minor = bytes[7]; + if ( ( major != 1 && major != 2 && major != 3 ) || minor != 0 ) return fail( "unsupported NPY version" ); + std::size_t header_begin = 10; + std::size_t header_length = static_cast< std::size_t >( bytes[8] ) | ( static_cast< std::size_t >( bytes[9] ) << 8 ); + if ( major != 1 ) + { + if ( size < 12 ) return fail( "file too short for the NPY header length" ); + header_begin = 12; + header_length |= ( static_cast< std::size_t >( bytes[10] ) << 16 ) | ( static_cast< std::size_t >( bytes[11] ) << 24 ); + } + std::size_t data_offset = 0; + if ( !checked_add( header_begin, header_length, data_offset ) || data_offset > size ) + return fail( "NPY header extends past the end of the file" ); + + // The header is a Python dict literal with exactly the keys descr, fortran_order and shape. + npy_header_cursor cur{ reinterpret_cast< char const* >( bytes ) + header_begin, reinterpret_cast< char const* >( bytes ) + data_offset }; + bool have_descr = false, have_order = false, have_shape = false; + bool fortran = false, foreign = false; + std::size_t dims[2] = { 0, 0 }; + std::size_t rank = 0; + cur.skip_ws(); + if ( !cur.eat( '{' ) ) return fail( "NPY header is not a dict literal" ); + cur.skip_ws(); + while ( !cur.eat( '}' ) ) + { + char const* kb = nullptr; + char const* ke = nullptr; + if ( !cur.string_literal( kb, ke ) ) return fail( "NPY header key is not a string" ); + std::string_view const key{ kb, static_cast< std::size_t >( ke - kb ) }; + cur.skip_ws(); + if ( !cur.eat( ':' ) ) return fail( "NPY header is missing ':'" ); + cur.skip_ws(); + if ( key == "descr" ) + { + if ( have_descr ) return fail( "NPY header repeats 'descr'" ); + have_descr = true; + if ( cur.peek() == '[' ) return fail( "structured NPY dtype is not supported" ); + char const* vb = nullptr; + char const* ve = nullptr; + if ( !cur.string_literal( vb, ve ) ) return fail( "NPY 'descr' is not a plain string" ); + if ( ve - vb < 3 ) return fail( "unsupported NPY dtype" ); + char const order = vb[0]; + if ( order != '<' && order != '>' && order != '|' && order != '=' ) return fail( "bad NPY byte order" ); + if ( vb[1] != kind ) return fail( "NPY dtype kind does not match the element type" ); + if ( vb[2] < '0' || vb[2] > '9' ) return fail( "bad NPY dtype size" ); + if ( vb[2] == '0' ) return fail( "bad NPY dtype size" ); // no zero size, no leading zeros (' 1 ) + { + if ( order == '<' ) foreign = std::endian::native != std::endian::little; + if ( order == '>' ) foreign = std::endian::native != std::endian::big; + } + } + else if ( key == "fortran_order" ) + { + if ( have_order ) return fail( "NPY header repeats 'fortran_order'" ); + have_order = true; + if ( cur.word( "True" ) ) fortran = true; + else if ( cur.word( "False" ) ) fortran = false; + else return fail( "NPY 'fortran_order' is not True or False" ); + } + else if ( key == "shape" ) + { + if ( have_shape ) return fail( "NPY header repeats 'shape'" ); + have_shape = true; + if ( !cur.eat( '(' ) ) return fail( "NPY 'shape' is not a tuple" ); + cur.skip_ws(); + bool comma = false; + while ( !cur.eat( ')' ) ) + { + char const c0 = cur.peek(); + if ( c0 < '0' || c0 > '9' ) return fail( "NPY dimension is not a non-negative integer" ); + std::size_t d = 0; + auto const [ptr, ec] = std::from_chars( cur.p, cur.end, d ); + if ( ec != std::errc{} ) return fail( "NPY dimension is out of range" ); + if ( c0 == '0' && ptr - cur.p > 1 ) return fail( "NPY dimension has a leading zero" ); // '(01, 2)' + cur.p = ptr; + if ( rank == 2 ) return fail( "NPY rank above 2 is not supported" ); + dims[rank++] = d; + cur.skip_ws(); + comma = cur.eat( ',' ); + cur.skip_ws(); + if ( !comma && cur.peek() != ')' ) return fail( "malformed NPY 'shape'" ); + } + if ( rank == 0 ) return fail( "NPY rank 0 is not supported" ); + if ( rank == 1 && !comma ) return fail( "malformed NPY 'shape'" ); + } + else + return fail( "unknown key in the NPY header" ); + cur.skip_ws(); + if ( cur.eat( ',' ) ) + { + cur.skip_ws(); + continue; + } + if ( cur.peek() != '}' ) return fail( "malformed NPY header dict" ); + } + cur.skip_ws(); + if ( !cur.at_end() ) return fail( "trailing characters after the NPY header dict" ); + if ( !have_descr || !have_order || !have_shape ) return fail( "NPY header is missing a key" ); + + std::size_t const r = rank == 1 ? 1 : dims[0]; + std::size_t const c = rank == 1 ? dims[0] : dims[1]; + std::size_t count = 0; + std::size_t payload = 0; + if ( !checked_mul( r, c, count ) || !checked_mul( count, sizeof( T ), payload ) ) + return fail( "NPY shape is too large" ); + if ( payload != size - data_offset ) return fail( "NPY payload length does not match the shape" ); + using size_type = typename Out::size_type; + if ( r > std::numeric_limits< size_type >::max() || c > std::numeric_limits< size_type >::max() ) + return fail( "NPY shape is too large" ); + + Out tmp{ out.get_allocator(), static_cast< size_type >( r ), static_cast< size_type >( c ) }; + std::uint8_t const* const src = bytes + data_offset; + T* const dst = tmp.data(); + constexpr std::size_t unit = npy_dtype< T >::component; + for ( std::size_t k = 0; k != count; ++k ) + { + // Row-major element k is (i, j); a Fortran payload stores it at j*r + i. + std::size_t const from = fortran ? ( k % c ) * r + ( k / c ) : k; + std::uint8_t element[sizeof( T )]; + std::memcpy( element, src + from * sizeof( T ), sizeof( T ) ); + if ( foreign ) + for ( std::size_t u = 0; u != sizeof( T ); u += unit ) + std::reverse( element + u, element + u + unit ); + std::memcpy( dst + k, element, sizeof( T ) ); + } + out = std::move( tmp ); + return true; + } } - friend const self_type operator-( const self_type& lhs, const difference_type rhs ) noexcept + + // S5-R2: the element types the native binary format (save_as_binary / load_binary) supports. + template < typename T > + inline constexpr bool is_binary_element_v = std::is_arithmetic_v< T > || std::is_same_v< T, std::complex< float > > + || std::is_same_v< T, std::complex< double > > || std::is_same_v< T, std::complex< long double > >; + + template < typename T > + struct text_complex : std::false_type {}; + template < typename F > + struct text_complex< std::complex< F > > : std::true_type { using component = F; }; + + // S5-R2: the text token separators; '\n' ends a row. + constexpr bool is_text_separator( char c ) noexcept { - self_type ans( lhs ); - ans -= rhs; - return ans; + return c == ',' || c == ';' || c == ' ' || c == '\t' || c == '\r'; } - reference operator[]( const difference_type dt ) noexcept + + // S5-R2: calls on_token( b, e ) for each token of the line [b, e); a token starting with '(' runs to the + // next ')' so the ',' of a complex pair is not a separator. False on an unclosed '(' or when on_token fails. + template < typename F > + bool for_each_text_token( char const* b, char const* e, F&& on_token ) noexcept { - return iterator_[dt * step_]; + for ( ;; ) + { + while ( b != e && is_text_separator( *b ) ) ++b; + if ( b == e ) return true; + char const* const t = b; + if ( *b == '(' ) + { + while ( b != e && *b != ')' ) ++b; + if ( b == e ) return false; + ++b; + } + while ( b != e && !is_text_separator( *b ) ) ++b; + if ( !on_token( t, b ) ) return false; + } } - const reference operator[]( const difference_type dt ) const noexcept + + enum class text_value_status { ok, syntax, range }; + + // S5-R2: one real number or integer in [b, e), parsed completely by std::from_chars (one leading '+' allowed). + template < typename T > + text_value_status parse_text_real( char const* b, char const* e, T& v ) noexcept { - return iterator_[dt * step_]; + if ( b != e && *b == '+' ) + { + ++b; + if ( b != e && ( *b == '+' || *b == '-' ) ) return text_value_status::syntax; + } + if ( b == e ) return text_value_status::syntax; + std::from_chars_result res{}; + if constexpr ( std::is_integral_v< T > ) + res = std::from_chars( b, e, v, 10 ); + else + res = std::from_chars( b, e, v, std::chars_format::general ); + if ( res.ec == std::errc::result_out_of_range ) return text_value_status::range; + if ( res.ec != std::errc{} || res.ptr != e ) return text_value_status::syntax; + return text_value_status::ok; } - reference operator*() noexcept + + template < typename T > + concept text_real = ( std::is_integral_v< T > && !std::is_same_v< T, bool > ) + || ( std::is_floating_point_v< T > && requires( char const* p, T& x ) { std::from_chars( p, p, x, std::chars_format::general ); } ); + + template < typename T > + concept text_element = text_real< T > || ( text_complex< T >::value && text_real< typename text_complex< T >::component > ); + + // S5-R2: one element token: a real or integer of type T, or for std::complex a bare real or "(re,im)". + template < typename T > + text_value_status parse_text_value( char const* b, char const* e, T& v ) noexcept { - return *iterator_; + if constexpr ( text_complex< T >::value ) + { + using F = typename text_complex< T >::component; + F re{}, im{}; + if ( b == e || *b != '(' ) + { + auto const s = parse_text_real( b, e, re ); + if ( s == text_value_status::ok ) v = T{ re, F{} }; + return s; + } + if ( e - b < 2 || *( e - 1 ) != ')' ) return text_value_status::syntax; + char const* const inner_b = b + 1; + char const* const inner_e = e - 1; + char const* const comma = std::find( inner_b, inner_e, ',' ); + if ( comma == inner_e ) return text_value_status::syntax; + auto trim = []( char const*& x, char const*& y ) noexcept + { + while ( x != y && ( *x == ' ' || *x == '\t' ) ) ++x; + while ( y != x && ( *( y - 1 ) == ' ' || *( y - 1 ) == '\t' ) ) --y; + }; + char const* rb = inner_b; char const* re_ = comma; + char const* ib = comma + 1; char const* ie = inner_e; + trim( rb, re_ ); + trim( ib, ie ); + auto const s1 = parse_text_real( rb, re_, re ); + if ( s1 != text_value_status::ok ) return s1; + auto const s2 = parse_text_real( ib, ie, im ); + if ( s2 != text_value_status::ok ) return s2; + v = T{ re, im }; + return text_value_status::ok; + } + else + return parse_text_real( b, e, v ); } - const reference operator*() const noexcept + + // S5-R2 (F08): parses text of `size` chars into `out`, a matrix< T, A >: each line holding a token is one row, + // all rows hold the same token count >= 1, every token parses completely as T. A first pass validates the + // structure, so the temporary (built with out's allocator) holds at most one element per input char; `out` + // is assigned only on success. On failure returns false, leaves `out` unchanged and sets *why when given. + template < typename T, typename Out > + bool parse_text( char const* chars, std::size_t size, Out& out, char const** why = nullptr ) noexcept + { + static_assert( std::is_same_v< typename Out::value_type, T >, "parse_text: Out must be a matrix of T" ); + auto fail = [why]( char const* reason ) noexcept + { + if ( why ) *why = reason; + return false; + }; + if constexpr ( !text_element< T > ) + { + (void)chars; (void)size; (void)out; + return fail( "the element type cannot be read from text" ); + } + else + { + if ( chars == nullptr || size == 0 ) return fail( "empty text" ); + char const* const end = chars + size; + std::size_t rows = 0, cols = 0; + for ( char const* line = chars; line != end; ) + { + char const* const eol = std::find( line, end, '\n' ); + std::size_t n = 0; + if ( !for_each_text_token( line, eol, [&n]( char const*, char const* ) noexcept { ++n; return true; } ) ) + return fail( "unclosed '(' in a row" ); + if ( n != 0 ) + { + if ( rows == 0 ) cols = n; + else if ( n != cols ) return fail( "rows hold different numbers of values" ); + ++rows; + } + line = eol == end ? end : eol + 1; + } + if ( rows == 0 ) return fail( "no values in the text" ); + using size_type = typename Out::size_type; + std::size_t count = 0; + if ( !checked_mul( rows, cols, count ) || rows > std::numeric_limits< size_type >::max() || cols > std::numeric_limits< size_type >::max() ) + return fail( "text matrix is too large" ); + + Out tmp{ out.get_allocator(), static_cast< size_type >( rows ), static_cast< size_type >( cols ) }; + T* dst = tmp.data(); + text_value_status status = text_value_status::ok; + for ( char const* line = chars; line != end; ) + { + char const* const eol = std::find( line, end, '\n' ); + bool const ok = for_each_text_token( line, eol, [&dst, &status]( char const* b, char const* e ) noexcept + { + status = parse_text_value( b, e, *dst ); + ++dst; + return status == text_value_status::ok; + } ); + if ( !ok ) + return fail( status == text_value_status::range ? "a value is out of range for the element type" + : "a token does not parse completely as the element type" ); + line = eol == end ? end : eol + 1; + } + out = std::move( tmp ); + return true; + } + } + + // S5-R2: writes m as text that parse_text reads back: tab-separated rows, one-byte integers as numbers, + // floating-point values with max_digits10 significant digits. + template < typename M > + void write_text( std::ostream& os, M const& m ) noexcept { - return *iterator_; + using T = typename M::value_type; + if constexpr ( std::is_floating_point_v< T > ) + os.precision( std::numeric_limits< T >::max_digits10 ); + else if constexpr ( text_complex< T >::value ) + os.precision( std::numeric_limits< typename text_complex< T >::component >::max_digits10 ); + else + os.precision( 18 ); + for ( typename M::size_type r = 0; r != m.row(); ++r ) + { + for ( auto it = m.row_begin( r ); it != m.row_end( r ); ++it ) + { + if constexpr ( std::is_integral_v< T > && sizeof( T ) == 1 ) + os << static_cast< int >( *it ) << '\t'; + else + os << *it << '\t'; + } + os << '\n'; + } } - friend bool operator==( const self_type& lhs, const self_type& rhs ) noexcept + + // S5-R2 (F08): parses a native binary image (two size_type counts, then the raw elements) into `out`, a + // matrix< T, A >. The count product and byte size are overflow-checked and the payload length must equal + // the declared size exactly; `out` is assigned only on success, otherwise *why is set when given. + template < typename T, typename Out > + bool parse_binary( std::uint8_t const* bytes, std::size_t size, Out& out, char const** why = nullptr ) noexcept + { + static_assert( std::is_same_v< typename Out::value_type, T >, "parse_binary: Out must be a matrix of T" ); + static_assert( is_binary_element_v< T >, "the native binary format supports arithmetic and std::complex elements only" ); + auto fail = [why]( char const* reason ) noexcept + { + if ( why ) *why = reason; + return false; + }; + using size_type = typename Out::size_type; + constexpr std::size_t head = 2 * sizeof( size_type ); + if ( bytes == nullptr || size < head ) return fail( "file shorter than the two counts" ); + size_type r = 0, c = 0; + std::memcpy( &r, bytes, sizeof( r ) ); + std::memcpy( &c, bytes + sizeof( r ), sizeof( c ) ); + size_type count = 0, payload = 0; + if ( !checked_mul( r, c, count ) || !checked_mul( count, static_cast< size_type >( sizeof( T ) ), payload ) ) + return fail( "the element count or byte size overflows" ); + if ( payload != size - head ) return fail( "payload length does not match the counts" ); + Out tmp{ out.get_allocator(), r, c }; + if ( payload != 0 ) std::memcpy( static_cast< void* >( tmp.data() ), bytes + head, static_cast< std::size_t >( payload ) ); + out = std::move( tmp ); + return true; + } + + // S6-R5 (D-023): a scalar operand is an arithmetic type or a std::complex specialization. + template< typename S > + concept matrix_scalar = std::is_arithmetic_v< S > || matrix_private::is_std_complex_v< S >; + + // The scalar as an operation on R elements sees it: R itself, or R's value type for complex R and a real + // scalar, converted as by static_cast. + template< typename R, typename S > + constexpr auto scalar_as( S const& s ) noexcept + { + if constexpr ( std::same_as< R, S > ) + return s; + else if constexpr ( matrix_private::is_std_complex_v< R > && !matrix_private::is_std_complex_v< S > ) + return static_cast< typename R::value_type >( s ); + else + return static_cast< R >( s ); + } + + // A scalar of another type that a compound operator on T elements converts first (D-023: T is kept). + template< typename S, typename T > + concept compound_scalar = matrix_scalar< S > && matrix_scalar< T > && !std::same_as< S, T >; + + // A scalar or matrix element of type S converts to T: not complex into real. + template< typename S, typename T > + inline constexpr bool converts_to_element_v = !( matrix_private::is_std_complex_v< S > && !matrix_private::is_std_complex_v< T > ); + + }//namespace matrix_details + + // S4-R1 (F05): column, diagonal and anti-diagonal iteration. Holds a base pointer, a stride, a logical index + // and a count; a pointer base_ + i*stride_ is formed only in operator*, operator[] and operator->, for a + // dereferenceable position. Arithmetic, equality, ordering and distance work on the index alone, so one past + // the end is index == count and a stride of 0 is legal. Under FENG_MATRIX_CHECKED_ITERATORS (D-020) the + // iterator also keeps its owner's [origin, origin+extent] and aborts on any position or address outside it. + template < typename P > + struct stride_iterator + { + typedef stride_iterator self_type; + typedef typename std::iterator_traits< P >::value_type value_type; + typedef typename std::iterator_traits< P >::reference reference; + typedef typename std::iterator_traits< P >::difference_type difference_type; + typedef P pointer; + typedef std::size_t size_type; + typedef std::random_access_iterator_tag iterator_category; + typedef std::random_access_iterator_tag iterator_concept; + + private: + template < typename Q > friend struct stride_iterator; + + P base_ = nullptr; + difference_type stride_ = 1; + difference_type index_ = 0; + difference_type count_ = 0; + #ifdef FENG_MATRIX_CHECKED_ITERATORS + P origin_ = nullptr; + difference_type extent_ = 0; + #endif + + void check_position() const noexcept + { + #ifdef FENG_MATRIX_CHECKED_ITERATORS + FENG_MATRIX_EXPECTS( 0 <= index_ && index_ <= count_, "matrix iterator: position ", index_, " outside [0, ", count_, "]" ); + #endif + } + + P address( difference_type i ) const noexcept + { + #ifdef FENG_MATRIX_CHECKED_ITERATORS + FENG_MATRIX_EXPECTS( 0 <= i && i < count_, "matrix iterator: dereference at position ", i, " outside [0, ", count_, ")" ); + #endif + return base_ + i * stride_; + } + + public: + stride_iterator() noexcept = default; + + // base: the element at index 0; count: the number of elements; index: the starting position in [0, count]; + // origin, extent: the owner's data() and size(), checked only under FENG_MATRIX_CHECKED_ITERATORS. + stride_iterator( P base, difference_type stride, difference_type count, difference_type index, [[maybe_unused]] P origin, [[maybe_unused]] size_type extent ) noexcept + : base_( base ), stride_( stride ), index_( index ), count_( count ) + #ifdef FENG_MATRIX_CHECKED_ITERATORS + , origin_( origin ), extent_( static_cast< difference_type >( extent ) ) + #endif + { + #ifdef FENG_MATRIX_CHECKED_ITERATORS + FENG_MATRIX_EXPECTS( 0 <= count_, "matrix iterator: negative count ", count_ ); + FENG_MATRIX_EXPECTS( extent <= static_cast< size_type >( PTRDIFF_MAX ), "matrix iterator: owner extent ", extent, " above PTRDIFF_MAX" ); + // integer offsets from the owner's origin, computed before any other pointer is formed + std::uintptr_t const b = reinterpret_cast< std::uintptr_t >( base ); + std::uintptr_t const o = reinterpret_cast< std::uintptr_t >( origin ); + std::uintptr_t const bytes = b - o; + std::uintptr_t const span = static_cast< std::uintptr_t >( extent_ ) * sizeof( value_type ); + FENG_MATRIX_EXPECTS( b >= o && bytes <= span && bytes % sizeof( value_type ) == 0, "matrix iterator: first address outside the owner's [data(), data()+size()]" ); + difference_type const first = static_cast< difference_type >( bytes / sizeof( value_type ) ); + if ( count_ > 0 ) + { + FENG_MATRIX_EXPECTS( first < extent_, "matrix iterator: first element at offset ", first, " outside an owner of size ", extent_ ); + FENG_MATRIX_EXPECTS( stride_ == 0 || ( count_ - 1 ) <= ( stride_ > 0 ? ( extent_ - 1 - first ) / stride_ : first / -stride_ ), + "matrix iterator: last element of ", count_, " with stride ", stride_, " from offset ", first, " outside an owner of size ", extent_ ); + } + #endif + check_position(); + } + + stride_iterator( const self_type& ) noexcept = default; + stride_iterator( self_type&& ) noexcept = default; + self_type& operator=( const self_type& ) noexcept = default; + self_type& operator=( self_type&& ) noexcept = default; + + // stride_iterator converts to stride_iterator + template < typename Q > + requires ( !std::is_same_v< Q, P > && std::is_convertible_v< Q, P > ) + stride_iterator( stride_iterator< Q > const& other ) noexcept + : base_( other.base_ ), stride_( other.stride_ ), index_( other.index_ ), count_( other.count_ ) + #ifdef FENG_MATRIX_CHECKED_ITERATORS + , origin_( other.origin_ ), extent_( other.extent_ ) + #endif + { } + + self_type& operator++() noexcept + { + ++index_; + check_position(); + return *this; + } + self_type operator++( int ) noexcept + { + self_type ans( *this ); + operator++(); + return ans; + } + self_type& operator--() noexcept + { + --index_; + check_position(); + return *this; + } + self_type operator--( int ) noexcept { - if ( lhs.step_ != rhs.step_ ) return false; - return lhs.iterator_ == rhs.iterator_; + self_type ans( *this ); + operator--(); + return ans; } - friend bool operator!=( const self_type& lhs, const self_type& rhs ) noexcept + self_type& operator+=( const difference_type dt ) noexcept { - if ( lhs.step_ != rhs.step_ ) return true; - return lhs.iterator_ != rhs.iterator_; + index_ += dt; + check_position(); + return *this; } - friend bool operator<( const self_type& lhs, const self_type& rhs ) noexcept + self_type& operator-=( const difference_type dt ) noexcept { - if ( lhs.step_ != rhs.step_ ) return false; - return lhs.iterator_ < rhs.iterator_; + index_ -= dt; + check_position(); + return *this; } - friend bool operator<=( const self_type& lhs, const self_type& rhs ) noexcept + friend self_type operator+( const self_type& lhs, const difference_type rhs ) noexcept { - if ( lhs.step_ != rhs.step_ ) return false; - return lhs.iterator_ <= rhs.iterator_; + self_type ans( lhs ); + ans += rhs; + return ans; } - friend bool operator>( const self_type& lhs, const self_type& rhs ) noexcept + friend self_type operator+( const difference_type lhs, const self_type& rhs ) noexcept { - if ( lhs.step_ != rhs.step_ ) return false; - return lhs.iterator_ > rhs.iterator_; + return rhs + lhs; } - friend bool operator>=( const self_type& lhs, const self_type& rhs ) noexcept + friend self_type operator-( const self_type& lhs, const difference_type rhs ) noexcept { - if ( lhs.step_ != rhs.step_ ) return false; - return lhs.iterator_ >= rhs.iterator_; + self_type ans( lhs ); + ans -= rhs; + return ans; } friend difference_type operator-( const self_type& lhs, const self_type& rhs ) noexcept { - better_assert( lhs.step_ == rhs.step_ && "stride iterators of different steps" ); - return ( lhs.iterator_ - rhs.iterator_ ) / lhs.step_; + return lhs.index_ - rhs.index_; + } + + reference operator*() const noexcept + { + return *address( index_ ); + } + reference operator[]( const difference_type dt ) const noexcept + { + return *address( index_ + dt ); + } + pointer operator->() const noexcept + { + return address( index_ ); + } + + friend bool operator==( const self_type& lhs, const self_type& rhs ) noexcept + { + return lhs.index_ == rhs.index_; + } + friend std::strong_ordering operator<=>( const self_type& lhs, const self_type& rhs ) noexcept + { + return lhs.index_ <=> rhs.index_; + } + }; + + // S4-R2 (F06, F05): row-major iteration over a rank-2 view. Holds the first viewed element, the parent's row + // stride, the view's column count, a logical index and a count; the address + // base_ + (i/cols_)*row_stride_ + i%cols_ is formed only in operator*, operator[] and operator->, for a + // dereferenceable position. Arithmetic, equality, ordering and distance work on the index alone. Under + // FENG_MATRIX_CHECKED_ITERATORS (D-020) it keeps its owner's [origin, origin+extent] and aborts on any position + // or address outside it, as stride_iterator does. + template < typename P > + struct view_iterator + { + typedef view_iterator self_type; + typedef typename std::iterator_traits< P >::value_type value_type; + typedef typename std::iterator_traits< P >::reference reference; + typedef typename std::iterator_traits< P >::difference_type difference_type; + typedef P pointer; + typedef std::size_t size_type; + typedef std::random_access_iterator_tag iterator_category; + typedef std::random_access_iterator_tag iterator_concept; + + private: + template < typename Q > friend struct view_iterator; + + P base_ = nullptr; + difference_type row_stride_ = 0; + difference_type cols_ = 1; + difference_type index_ = 0; + difference_type count_ = 0; + #ifdef FENG_MATRIX_CHECKED_ITERATORS + P origin_ = nullptr; + difference_type extent_ = 0; + #endif + + void check_position() const noexcept + { + #ifdef FENG_MATRIX_CHECKED_ITERATORS + FENG_MATRIX_EXPECTS( 0 <= index_ && index_ <= count_, "matrix iterator: position ", index_, " outside [0, ", count_, "]" ); + #endif + } + + P address( difference_type i ) const noexcept + { + #ifdef FENG_MATRIX_CHECKED_ITERATORS + FENG_MATRIX_EXPECTS( 0 <= i && i < count_, "matrix iterator: dereference at position ", i, " outside [0, ", count_, ")" ); + #endif + return base_ + ( ( i / cols_ ) * row_stride_ + i % cols_ ); + } + + public: + view_iterator() noexcept = default; + + // base: the element at index 0; row_stride: the parent's columns; cols: the view's columns (> 0 when + // count > 0); count: the number of elements; index: the starting position in [0, count]; origin, extent: + // the owner's data() and size(), checked only under FENG_MATRIX_CHECKED_ITERATORS. + view_iterator( P base, difference_type row_stride, difference_type cols, difference_type count, difference_type index, [[maybe_unused]] P origin, [[maybe_unused]] size_type extent ) noexcept + : base_( base ), row_stride_( row_stride ), cols_( cols > 0 ? cols : 1 ), index_( index ), count_( count ) + #ifdef FENG_MATRIX_CHECKED_ITERATORS + , origin_( origin ), extent_( static_cast< difference_type >( extent ) ) + #endif + { + #ifdef FENG_MATRIX_CHECKED_ITERATORS + FENG_MATRIX_EXPECTS( 0 <= count_ && 0 <= row_stride_, "matrix iterator: negative count ", count_, " or row stride ", row_stride_ ); + FENG_MATRIX_EXPECTS( extent <= static_cast< size_type >( PTRDIFF_MAX ), "matrix iterator: owner extent ", extent, " above PTRDIFF_MAX" ); + // integer offsets from the owner's origin, computed before any other pointer is formed + std::uintptr_t const b = reinterpret_cast< std::uintptr_t >( base ); + std::uintptr_t const o = reinterpret_cast< std::uintptr_t >( origin ); + std::uintptr_t const bytes = b - o; + std::uintptr_t const span = static_cast< std::uintptr_t >( extent_ ) * sizeof( value_type ); + FENG_MATRIX_EXPECTS( b >= o && bytes <= span && bytes % sizeof( value_type ) == 0, "matrix iterator: first address outside the owner's [data(), data()+size()]" ); + difference_type const first = static_cast< difference_type >( bytes / sizeof( value_type ) ); + if ( count_ > 0 ) + { + difference_type const last_row = ( count_ - 1 ) / cols_; + difference_type const last_col = ( count_ - 1 ) % cols_; + FENG_MATRIX_EXPECTS( first < extent_ && ( row_stride_ == 0 || last_row <= ( extent_ - 1 - first ) / row_stride_ ) && + last_row * row_stride_ + last_col <= extent_ - 1 - first, + "matrix iterator: last element of ", count_, " in rows of ", cols_, " with row stride ", row_stride_, " from offset ", first, " outside an owner of size ", extent_ ); + } + #endif + check_position(); } + + view_iterator( const self_type& ) noexcept = default; + view_iterator( self_type&& ) noexcept = default; + self_type& operator=( const self_type& ) noexcept = default; + self_type& operator=( self_type&& ) noexcept = default; + + // view_iterator converts to view_iterator + template < typename Q > + requires ( !std::is_same_v< Q, P > && std::is_convertible_v< Q, P > ) + view_iterator( view_iterator< Q > const& other ) noexcept + : base_( other.base_ ), row_stride_( other.row_stride_ ), cols_( other.cols_ ), index_( other.index_ ), count_( other.count_ ) + #ifdef FENG_MATRIX_CHECKED_ITERATORS + , origin_( other.origin_ ), extent_( other.extent_ ) + #endif + { } + + self_type& operator++() noexcept { ++index_; check_position(); return *this; } + self_type operator++( int ) noexcept { self_type ans( *this ); operator++(); return ans; } + self_type& operator--() noexcept { --index_; check_position(); return *this; } + self_type operator--( int ) noexcept { self_type ans( *this ); operator--(); return ans; } + self_type& operator+=( const difference_type dt ) noexcept { index_ += dt; check_position(); return *this; } + self_type& operator-=( const difference_type dt ) noexcept { index_ -= dt; check_position(); return *this; } + friend self_type operator+( const self_type& lhs, const difference_type rhs ) noexcept { self_type ans( lhs ); ans += rhs; return ans; } + friend self_type operator+( const difference_type lhs, const self_type& rhs ) noexcept { return rhs + lhs; } + friend self_type operator-( const self_type& lhs, const difference_type rhs ) noexcept { self_type ans( lhs ); ans -= rhs; return ans; } + friend difference_type operator-( const self_type& lhs, const self_type& rhs ) noexcept { return lhs.index_ - rhs.index_; } + + reference operator*() const noexcept { return *address( index_ ); } + reference operator[]( const difference_type dt ) const noexcept { return *address( index_ + dt ); } + pointer operator->() const noexcept { return address( index_ ); } + + friend bool operator==( const self_type& lhs, const self_type& rhs ) noexcept { return lhs.index_ == rhs.index_; } + friend std::strong_ordering operator<=>( const self_type& lhs, const self_type& rhs ) noexcept { return lhs.index_ <=> rhs.index_; } }; + namespace matrix_private + { + // S4-R3 (F06): a view of an owner with `rows` rows and `cols` columns needs r0 <= r1 <= rows and + // c0 <= c1 <= cols; an empty range is allowed, nothing is normalized. + inline void check_view_range( std::size_t r0, std::size_t r1, std::size_t c0, std::size_t c1, std::size_t rows, std::size_t cols ) noexcept + { + FENG_MATRIX_EXPECTS( r0 <= r1 && r1 <= rows, "matrix view: row range [", r0, ", ", r1, ") outside an owner with ", rows, " rows" ); + FENG_MATRIX_EXPECTS( c0 <= c1 && c1 <= cols, "matrix view: column range [", c0, ", ", c1, ") outside an owner with ", cols, " columns" ); + } + + // S4-R3: a factory's range list must hold exactly two values. + template < typename Integer_Type > + std::pair< std::size_t, std::size_t > view_extent( std::initializer_list< Integer_Type > list, char const* what ) noexcept + { + FENG_MATRIX_EXPECTS( list.size() == 2, "matrix view: the ", what, " range needs exactly two values, got ", list.size() ); + return { static_cast< std::size_t >( *list.begin() ), static_cast< std::size_t >( *( list.begin() + 1 ) ) }; + } + + // S4-R1: a column index must name a column. + inline void check_column( std::size_t c, std::size_t cols ) noexcept + { + FENG_MATRIX_EXPECTS( c < cols, "matrix iterator: column ", c, " outside a matrix with ", cols, " columns" ); + } + + // S4-R1: diagonal (and anti-diagonal) k exists for -rows < k < cols; diagonal 0 always exists (empty on + // an empty matrix). + inline void check_diagonal( std::ptrdiff_t k, std::size_t rows, std::size_t cols ) noexcept + { + bool const valid = k == 0 || ( k > 0 ? static_cast< std::size_t >( k ) < cols : static_cast< std::size_t >( -( k + 1 ) ) + 1 < rows ); + FENG_MATRIX_EXPECTS( valid, "matrix iterator: diagonal ", k, " outside a ", rows, "x", cols, " matrix" ); + } + + // the signed diagonal of an unsigned upper (k >= 0) or lower (k <= 0) index, saturated so that it stays invalid + inline std::ptrdiff_t upper_diagonal( std::size_t index ) noexcept + { + return static_cast< std::ptrdiff_t >( std::min< std::size_t >( index, PTRDIFF_MAX ) ); + } + inline std::ptrdiff_t lower_diagonal( std::size_t index ) noexcept + { + return -static_cast< std::ptrdiff_t >( std::min< std::size_t >( index, PTRDIFF_MAX ) ); + } + + // S4-R1: the [begin, end) pair of a strided range of `length` elements starting at integer offset `start` + // of an owner [data, data+size); an empty range starts at offset 0, so no pointer past data+size is formed. + template < typename It, typename Ptr > + It stride_range( Ptr data, std::size_t size, std::size_t start, std::ptrdiff_t stride, std::size_t length, bool at_end ) noexcept + { + if ( length == 0 ) start = 0; + std::ptrdiff_t const n = static_cast< std::ptrdiff_t >( length ); + return It( data + start, stride, n, at_end ? n : 0, data, size ); + } + } + template < typename Type, Allocator Alloc > struct crtp_typedef { @@ -1356,9 +2402,9 @@ namespace feng typedef Matrix zen_type; typedef Type value_type; - #ifdef OPENCV + #ifdef FENG_MATRIX_OPENCV - cv::Mat const to_opencv( unsigned long channels = 1 ) const + cv::Mat to_opencv( unsigned long channels = 1 ) const noexcept { better_assert( ((channels>=1) && (channels<=4)), "Expecting 1-4 channels." ); @@ -1434,7 +2480,7 @@ namespace feng return ans; } - auto& from_opencv( cv::Mat image ) + auto& from_opencv( cv::Mat image ) noexcept { if ( !image.isContiguous() ) // data stored in Mat is not always continuous in memory image = image.clone(); @@ -1523,7 +2569,7 @@ namespace feng auto get_allocator() const noexcept { auto const& zen = static_cast(*this); - return zen.allocator_; + return matrix_private::storage_access::storage( zen ).get_allocator(); } }; @@ -1540,25 +2586,19 @@ namespace feng typedef typename type_proxy_type::const_reverse_anti_diag_type const_reverse_anti_diag_type; anti_diag_type upper_anti_diag_begin( const size_type index = 0 ) noexcept { - zen_type& zen = static_cast< zen_type& >( *this ); - return anti_diag_type( zen.begin() + zen.col() - index - 1, zen.col() - 1 ); + return upper_anti_diag_range< anti_diag_type >( static_cast< zen_type& >( *this ), index, false ); } anti_diag_type upper_anti_diag_end( const size_type index = 0 ) noexcept { - zen_type& zen = static_cast< zen_type& >( *this ); - size_type const depth = std::min( zen.col()-index, zen.row() ); - return upper_anti_diag_begin( index ) + depth; + return upper_anti_diag_range< anti_diag_type >( static_cast< zen_type& >( *this ), index, true ); } const_anti_diag_type upper_anti_diag_begin( const size_type index = 0 ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - return const_anti_diag_type( zen.begin() + zen.col() - index - 1, zen.col() - 1 ); + return upper_anti_diag_range< const_anti_diag_type >( static_cast< zen_type const& >( *this ), index, false ); } const_anti_diag_type upper_anti_diag_end( const size_type index = 0 ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - size_type const depth = std::min( zen.col()-index, zen.row() ); - return upper_anti_diag_begin( index ) + depth; + return upper_anti_diag_range< const_anti_diag_type >( static_cast< zen_type const& >( *this ), index, true ); } const_anti_diag_type upper_anti_diag_cbegin( const size_type index = 0 ) const noexcept { @@ -1594,27 +2634,19 @@ namespace feng } anti_diag_type lower_anti_diag_begin( const size_type index = 0 ) noexcept { - zen_type& zen = static_cast< zen_type& >( *this ); - return anti_diag_type( zen.begin() + ( zen.col() * ( index + 1 ) ) - 1, zen.col() - 1 ); + return lower_anti_diag_range< anti_diag_type >( static_cast< zen_type& >( *this ), index, false ); } anti_diag_type lower_anti_diag_end( const size_type index = 0 ) noexcept { - zen_type& zen = static_cast< zen_type& >( *this ); - - size_type const depth = std::min( zen.row()-index, zen.col() ); - return lower_anti_diag_begin( index ) + depth; + return lower_anti_diag_range< anti_diag_type >( static_cast< zen_type& >( *this ), index, true ); } const_anti_diag_type lower_anti_diag_begin( const size_type index = 0 ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - return const_anti_diag_type( zen.begin() + ( zen.col() * ( index + 1 ) ) - 1, zen.col() - 1 ); + return lower_anti_diag_range< const_anti_diag_type >( static_cast< zen_type const& >( *this ), index, false ); } const_anti_diag_type lower_anti_diag_end( const size_type index = 0 ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - - size_type const depth = std::min( zen.row()-index, zen.col() ); - return lower_anti_diag_begin( index ) + depth; + return lower_anti_diag_range< const_anti_diag_type >( static_cast< zen_type const& >( *this ), index, true ); } const_anti_diag_type lower_anti_diag_cbegin( const size_type index = 0 ) const noexcept { @@ -1650,111 +2682,108 @@ namespace feng } anti_diag_type anti_diag_begin( const difference_type index = 0 ) noexcept { - if ( index > 0 ) - { - return upper_anti_diag_begin( index ); - } - - return lower_anti_diag_begin( -index ); + zen_type& zen = static_cast< zen_type& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_anti_diag_begin( static_cast< size_type >( index ) ); + return lower_anti_diag_begin( static_cast< size_type >( -index ) ); } anti_diag_type anti_diag_end( const difference_type index = 0 ) noexcept { - if ( index > 0 ) - { - return upper_anti_diag_end( index ); - } - - return lower_anti_diag_end( -index ); + zen_type& zen = static_cast< zen_type& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_anti_diag_end( static_cast< size_type >( index ) ); + return lower_anti_diag_end( static_cast< size_type >( -index ) ); } const_anti_diag_type anti_diag_begin( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) - { - return upper_anti_diag_begin( index ); - } - - return lower_anti_diag_begin( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_anti_diag_begin( static_cast< size_type >( index ) ); + return lower_anti_diag_begin( static_cast< size_type >( -index ) ); } const_anti_diag_type anti_diag_end( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) - { - return upper_anti_diag_end( index ); - } - - return lower_anti_diag_end( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_anti_diag_end( static_cast< size_type >( index ) ); + return lower_anti_diag_end( static_cast< size_type >( -index ) ); } const_anti_diag_type anti_diag_cbegin( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) - { - return upper_anti_diag_cbegin( index ); - } - - return lower_anti_diag_cbegin( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_anti_diag_cbegin( static_cast< size_type >( index ) ); + return lower_anti_diag_cbegin( static_cast< size_type >( -index ) ); } const_anti_diag_type anti_diag_cend( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) - { - return upper_anti_diag_cend( index ); - } - - return lower_anti_diag_cend( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_anti_diag_cend( static_cast< size_type >( index ) ); + return lower_anti_diag_cend( static_cast< size_type >( -index ) ); } reverse_anti_diag_type anti_diag_rbegin( const difference_type index = 0 ) noexcept { - if ( index > 0 ) - { - return upper_anti_diag_rbegin( index ); - } - - return lower_anti_diag_rbegin( -index ); + zen_type& zen = static_cast< zen_type& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_anti_diag_rbegin( static_cast< size_type >( index ) ); + return lower_anti_diag_rbegin( static_cast< size_type >( -index ) ); } reverse_anti_diag_type anti_diag_rend( const difference_type index = 0 ) noexcept { - if ( index > 0 ) - { - return upper_anti_diag_rend( index ); - } - - return lower_anti_diag_rend( -index ); + zen_type& zen = static_cast< zen_type& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_anti_diag_rend( static_cast< size_type >( index ) ); + return lower_anti_diag_rend( static_cast< size_type >( -index ) ); } const_reverse_anti_diag_type anti_diag_rbegin( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) - { - return upper_anti_diag_rbegin( index ); - } - - return lower_anti_diag_rbegin( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_anti_diag_rbegin( static_cast< size_type >( index ) ); + return lower_anti_diag_rbegin( static_cast< size_type >( -index ) ); } const_reverse_anti_diag_type anti_diag_rend( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) - { - return upper_anti_diag_rend( index ); - } - - return lower_anti_diag_rend( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_anti_diag_rend( static_cast< size_type >( index ) ); + return lower_anti_diag_rend( static_cast< size_type >( -index ) ); } const_reverse_anti_diag_type anti_diag_crbegin( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) - { - return upper_anti_diag_crbegin( index ); - } - - return lower_anti_diag_crbegin( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_anti_diag_crbegin( static_cast< size_type >( index ) ); + return lower_anti_diag_crbegin( static_cast< size_type >( -index ) ); } const_reverse_anti_diag_type anti_diag_crend( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) - { - return upper_anti_diag_crend( index ); - } - - return lower_anti_diag_crend( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_anti_diag_crend( static_cast< size_type >( index ) ); + return lower_anti_diag_crend( static_cast< size_type >( -index ) ); + } + private: + // S4-R1: upper anti-diagonal k starts at (0, col()-1-k), lower anti-diagonal k at (k, col()-1); each step + // moves one row down and one column left (stride col()-1, 0 for one column). + template < typename It, typename Z > + static It upper_anti_diag_range( Z& zen, size_type index, bool at_end ) noexcept + { + size_type const rows = zen.row(), cols = zen.col(); + matrix_private::check_diagonal( matrix_private::upper_diagonal( index ), rows, cols ); + size_type const length = index < cols ? std::min( cols - index, rows ) : 0; + size_type const start = length ? cols - 1 - index : 0; + return matrix_private::stride_range< It >( std::to_address( zen.data() ), zen.size(), start, static_cast< difference_type >( cols ) - 1, length, at_end ); + } + template < typename It, typename Z > + static It lower_anti_diag_range( Z& zen, size_type index, bool at_end ) noexcept + { + size_type const rows = zen.row(), cols = zen.col(); + matrix_private::check_diagonal( matrix_private::lower_diagonal( index ), rows, cols ); + size_type const length = index < rows ? std::min( rows - index, cols ) : 0; + size_type const start = length ? index * cols + cols - 1 : 0; + return matrix_private::stride_range< It >( std::to_address( zen.data() ), zen.size(), start, static_cast< difference_type >( cols ) - 1, length, at_end ); } }; template < typename Matrix, typename Type, Allocator Alloc > @@ -1769,7 +2798,7 @@ namespace feng zen_type& zen = static_cast< zen_type& >( *this ); value_type* x = zen.data(); auto && parallel_function = [x, &func]( std::uint_least64_t offset ) { func( x[offset] ); }; - matrix_details::parallel( parallel_function, 0UL, zen.size(), 0UL ); + matrix_details::parallel_work( parallel_function, 0UL, zen.size(), 1, matrix_details::callback_grain ); } template < typename Function > @@ -1796,52 +2825,35 @@ namespace feng row_type operator[]( const size_type index ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); - better_assert( index < zen.row() && "Row index outof boundary!", " with index ", index, " and zen.row() ", zen.row() ); + FENG_MATRIX_EXPECTS( index < zen.row() && "Row index out of boundary!", "matrix index: row ", index, " with row() ", zen.row() ); return zen.row_begin( index ); } const_row_type operator[]( const size_type index ) const noexcept { zen_type const& zen = static_cast< zen_type const& >( *this ); - better_assert( index < zen.row() && "Row index outof boundary!", " with index ", index, " and zen.row() ", zen.row() ); + FENG_MATRIX_EXPECTS( index < zen.row() && "Row index out of boundary!", "matrix index: row ", index, " with row() ", zen.row() ); return zen.row_cbegin( index ); } - value_type operator()( size_type r, size_type c ) const noexcept + // S2-R2: at( r, c ) and m( r, c ) share matrix_private::check_index and abort with `index` out of range. + value_type const& at( size_type r, size_type c ) const noexcept { zen_type const& zen = static_cast< zen_type const& >( *this ); - better_assert( r < zen.row() && "Row index out of boundary!" ); - better_assert( c < zen.col() && "Column index out of boundary!" ); - return zen[r][c]; + matrix_private::check_index( r, c, zen.row(), zen.col() ); + return zen.row_cbegin( r )[c]; } - value_type& operator()( size_type r, size_type c ) noexcept + value_type& at( size_type r, size_type c ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); - better_assert( r < zen.row() && "Row index out of boundary!" ); - better_assert( c < zen.col() && "Column index out of boundary!" ); - return zen[r][c]; + matrix_private::check_index( r, c, zen.row(), zen.col() ); + return zen.row_begin( r )[c]; } - }; - - template < typename Matrix, typename Type, Allocator Alloc > - struct crtp_bracket_operator_view - { - typedef Matrix zen_type; - typedef crtp_typedef< Type, Alloc > type_proxy_type; - typedef typename type_proxy_type::value_type value_type; - typedef typename type_proxy_type::size_type size_type; - typedef typename type_proxy_type::row_type row_type; - typedef typename type_proxy_type::const_row_type const_row_type; - const_row_type operator[]( const size_type index ) const noexcept + value_type operator()( size_type r, size_type c ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - better_assert( index < zen.row() && "Row index outof boundary!" ); - return zen.row_cbegin( index ); + return (*this).at( r, c ); } - value_type operator()( size_type r, size_type c ) const noexcept + value_type& operator()( size_type r, size_type c ) noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - better_assert( r < zen.row() && "Row index out of boundary!" ); - better_assert( c < zen.col() && "Column index out of boundary!" ); - return zen[r][c]; + return (*this).at( r, c ); } }; @@ -1853,13 +2865,11 @@ namespace feng void clear() noexcept { zen_type& zen = static_cast< zen_type& >( *this ); - - if ( zen.data() != nullptr ) - zen.get_allocator().deallocate( zen.data(), zen.size() ); - - zen.dat_ = nullptr; - zen.row_ = 0; - zen.col_ = 0; + auto& storage = matrix_private::storage_access::storage( zen ); + std::remove_reference_t< decltype( storage ) > empty{ storage.get_allocator() }; + storage.swap( empty ); // equal allocators: releases the old buffer + matrix_private::storage_access::rows( zen ) = 0; + matrix_private::storage_access::cols( zen ) = 0; } }; template < typename Matrix, typename Type, Allocator Alloc > @@ -1871,15 +2881,13 @@ namespace feng template < typename Other_Matrix > zen_type& clone( const Other_Matrix& other, std::initializer_list r, std::initializer_list c ) noexcept { - better_assert( r.size() == 2 && "row size should be 2!" ); - better_assert( c.size() == 2 && "col size should be 2!" ); - return clone( other, r[0], r[1], c[0], c[1] ); + matrix_private::check_clone_lists( r.size(), c.size() ); + return clone( other, *r.begin(), *(r.begin()+1), *c.begin(), *(c.begin()+1) ); } template < typename Other_Matrix > zen_type& clone( const Other_Matrix& other, size_type const r0, size_type const r1, size_type const c0, size_type const c1 ) noexcept { - better_assert( r1 > r0 && "row range error!" ); - better_assert( c1 > c0 && "col range error!" ); + matrix_private::check_clone_range( r0, r1, c0, c1, other.row(), other.col() ); zen_type& zen = static_cast< zen_type& >( *this ); zen_type tmp{ zen.get_allocator(), r1-r0, c1-c0 }; @@ -1889,22 +2897,21 @@ namespace feng std::copy_n( other.row_begin(r+r0)+c0, tmp.col(), tmp.row_begin(r) ); }; - matrix_details::parallel( parallel_function, 0UL, tmp.row(), 0UL ); + matrix_details::parallel_work( parallel_function, 0UL, tmp.row(), tmp.col(), matrix_details::elementwise_grain ); zen.swap( tmp ); return zen; } - zen_type const clone( std::initializer_list r, std::initializer_list c ) const noexcept + [[nodiscard]] zen_type clone( std::initializer_list r, std::initializer_list c ) const noexcept { - better_assert( r.size() == 2 && "row size should be 2!" ); - better_assert( c.size() == 2 && "col size should be 2!" ); + matrix_private::check_clone_lists( r.size(), c.size() ); auto const [r0, r1] = std::tuple{ *r.begin(), *(r.begin()+1) }; auto const [c0, c1] = std::tuple{ *c.begin(), *(c.begin()+1) }; return clone( r0, r1, c0, c1 ); } - zen_type const clone( size_type const r0, size_type const r1, size_type const c0, size_type const c1 ) const noexcept + [[nodiscard]] zen_type clone( size_type const r0, size_type const r1, size_type const c0, size_type const c1 ) const noexcept { zen_type const& zen = static_cast< zen_type const& >( *this ); zen_type ans{ zen.get_allocator() }; @@ -1918,39 +2925,34 @@ namespace feng typedef Matrix zen_type; typedef crtp_typedef< Type, Alloc > type_proxy_type; typedef typename type_proxy_type::size_type size_type; + typedef typename type_proxy_type::difference_type difference_type; typedef typename type_proxy_type::col_type col_type; typedef typename type_proxy_type::const_col_type const_col_type; typedef typename type_proxy_type::reverse_col_type reverse_col_type; typedef typename type_proxy_type::const_reverse_col_type const_reverse_col_type; col_type col_begin( const size_type index ) noexcept { - zen_type& zen = static_cast< zen_type& >( *this ); - return col_type( zen.row_begin(0) + index, zen.col() ); + return col_range< col_type >( static_cast< zen_type& >( *this ), index, false ); } col_type col_end( const size_type index ) noexcept { - zen_type& zen = static_cast< zen_type& >( *this ); - return col_begin( index ) + zen.row(); + return col_range< col_type >( static_cast< zen_type& >( *this ), index, true ); } const_col_type col_begin( const size_type index ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - return const_col_type( zen.row_begin(0) + index, zen.col() ); + return col_range< const_col_type >( static_cast< zen_type const& >( *this ), index, false ); } const_col_type col_end( const size_type index ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - return zen.col_begin( index ) + zen.row(); + return col_range< const_col_type >( static_cast< zen_type const& >( *this ), index, true ); } const_col_type col_cbegin( const size_type index ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - return const_col_type( zen.begin() + index, zen.col() ); + return col_begin( index ); } const_col_type col_cend( const size_type index ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - return zen.col_begin( index ) + zen.row(); + return col_end( index ); } reverse_col_type col_rbegin( const size_type index = 0 ) noexcept { @@ -1970,17 +2972,22 @@ namespace feng } const_reverse_col_type col_crbegin( const size_type index = 0 ) const noexcept { - return const_reverse_col_type( col_end( index ) ); + return col_rbegin( index ); } const_reverse_col_type col_crend( const size_type index = 0 ) const noexcept { - return const_reverse_col_type( col_begin( index ) ); + return col_rend( index ); + } + private: + // S4-R1: column c starts at offset c with stride col() and row() elements. + template < typename It, typename Z > + static It col_range( Z& zen, size_type index, bool at_end ) noexcept + { + matrix_private::check_column( index, zen.col() ); + return matrix_private::stride_range< It >( std::to_address( zen.data() ), zen.size(), index, static_cast< difference_type >( zen.col() ), zen.row(), at_end ); } }; - template < typename Matrix, typename Type, Allocator Alloc > - using crtp_col_iterator_view = crtp_col_iterator; - template < typename Matrix, typename Type, Allocator Alloc > struct crtp_copy { @@ -1991,7 +2998,26 @@ namespace feng void copy( const Other_Matrix& rhs ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); - zen.allocator_ = rhs.allocator_; + // S2-R4: copying a matrix onto itself is a no-op that keeps the contents. + if ( static_cast< void const* >( std::addressof( rhs ) ) == static_cast< void const* >( std::addressof( zen ) ) ) return; + // S3 (F04): the destination keeps its own allocator. + // S4-R2 (B-009): a view of the destination is copied through a snapshot taken before resizing. + if constexpr ( requires { rhs.row_stride(); } && std::is_constructible_v< zen_type, Other_Matrix const& > ) + { + if ( rhs.size() != 0 ) + { + void const* const first = static_cast< void const* >( std::addressof( rhs.at( 0, 0 ) ) ); + void const* const lo = static_cast< void const* >( zen.data() ); + void const* const hi = static_cast< void const* >( zen.data() + zen.size() ); + std::less<> const before{}; + if ( !before( first, lo ) && before( first, hi ) ) + { + zen_type const snapshot{ rhs }; + zen.copy( snapshot ); + return; + } + } + } zen.resize( rhs.row(), rhs.col() ); std::copy( rhs.begin(), rhs.end(), zen.begin() );//<- should be overloaded when with cuda_allocator // TODO: copy-and-swap @@ -2018,11 +3044,30 @@ namespace feng better_assert( r1 - r0 == other.row(), " row dim does not match, expected ", other.row(), " rows, but passed parameters are ", r0, " and ", r1 ); better_assert( c1 - c0 == other.col(), " col dim does not match, expected ", other.col(), " cols, but passed parameters are ", c0, " and ", c1 ); + if ( other.row() == 0 || other.col() == 0 ) return; + + // S2-R4: when the source's element range overlaps the destination block, copy a snapshot of the source. + { + auto const* const src_first = static_cast< void const* >( std::addressof( *other.row_begin( 0 ) ) ); + auto const* const src_last = static_cast< void const* >( std::addressof( *( other.row_begin( other.row() - 1 ) + ( other.col() - 1 ) ) ) + 1 ); + auto const* const dst_first = static_cast< void const* >( std::addressof( *( zen.row_begin( r0 ) + c0 ) ) ); + auto const* const dst_last = static_cast< void const* >( std::addressof( *( zen.row_begin( r1 - 1 ) + ( c1 - 1 ) ) ) + 1 ); + std::less<> const before{}; + if ( before( src_first, dst_last ) && before( dst_first, src_last ) ) + { + zen_type snapshot( other.row(), other.col() ); + for ( size_type row_index = 0; row_index != other.row(); ++row_index ) + std::copy( other.row_begin( row_index ), other.row_end( row_index ), snapshot.row_begin( row_index ) ); + zen.copy( snapshot, { r0, r1 }, { c0, c1 } ); + return; + } + } + auto const& copy_function = [&, r0=r0, c0=c0]( size_type const row_index ) { std::copy( other.row_begin(row_index), other.row_end(row_index), zen.row_begin(r0+row_index)+c0 ); }; - matrix_details::parallel( copy_function, 0UL, other.row(), 0UL ); + matrix_details::parallel_work( copy_function, 0UL, other.row(), other.col(), matrix_details::elementwise_grain ); } }; template < typename Matrix, typename Type, Allocator Alloc > @@ -2035,52 +3080,186 @@ namespace feng pointer data() noexcept { zen_type& zen = static_cast< zen_type& >( *this ); - return zen.dat_; + return matrix_private::storage_access::storage( zen ).data(); } const_pointer data() const noexcept { zen_type const& zen = static_cast< zen_type const& >( *this ); - return zen.dat_; + return matrix_private::storage_access::storage( zen ).data(); } }; - template < typename Matrix, typename Type, Allocator Alloc > - struct crtp_det + // ---- linear algebra: status, result and the one LU kernel (S7-R1, S7-R2, D-026, D-027) ---- + // Declared ahead of the crtp members so det() and inverse() can use them. + enum class linalg_status { ok, singular, not_positive_definite, not_converged, nonfinite }; + + template < typename V > + struct linalg_result { - typedef Matrix zen_type; - typedef crtp_typedef< Type, Alloc > type_proxy_type; - typedef typename type_proxy_type::size_type size_type; - typedef typename type_proxy_type::value_type value_type; - typedef typename type_proxy_type::range_type range_type; - value_type det() const noexcept - { - zen_type const& zen = static_cast< zen_type const& >( *this ); - better_assert( zen.row()==zen.col(), " matrix::det(), the row and matrix are supposed to be same, but now row is ", zen.row(), " and col is ", zen.col() ); + V value; + linalg_status status; + constexpr bool ok() const noexcept { return status == linalg_status::ok; } + constexpr explicit operator bool() const noexcept { return ok(); } + }; - if ( 0 == zen.size() ) - { - return value_type{}; - } + // S9-R2 (D-034): the element types linear algebra accepts: floating-point or std::complex. An integral matrix + // fails this constraint; convert it with astype() first. + template < typename T > + concept linalg_element = std::floating_point< T > || matrix_private::is_std_complex_v< T >; - if ( 1 == zen.size() ) - { - return *( zen.begin() ); + namespace matrix_details + { + template < typename T > + struct linalg_real { using type = T; }; + template < typename T > + struct linalg_real< std::complex< T > > { using type = T; }; + template < typename T > + using linalg_real_t = typename linalg_real< T >::type; + + template < typename T > + bool linalg_isfinite( T const& x ) noexcept + { + if constexpr ( matrix_private::is_std_complex_v< T > ) + return std::isfinite( x.real() ) && std::isfinite( x.imag() ); + else + return std::isfinite( x ); + } + + struct lu_info + { + linalg_status status; + std::size_t rank; + int sign; // sign of the row permutation + }; + + // In-place partial-pivoting LU of the row-major n×n array `a`: on return the strict lower triangle holds L + // (unit diagonal implied) and the upper triangle U, with P·A = L·U. `piv` (size n) gets row i of P·A = + // row piv[i] of A. The pivot of column k is the first entry of largest |·| in rows k..n-1; a column that + // is zero there is left as is (no elimination). rank counts |u_kk| > n·ε·max|U| (D-027). + template < typename T > + lu_info lu_in_place( T* a, std::size_t n, std::size_t* piv ) noexcept + { + using R = linalg_real_t< T >; + for ( std::size_t i = 0; i != n; ++i ) piv[i] = i; + for ( std::size_t i = 0; i != n * n; ++i ) + if ( !linalg_isfinite( a[i] ) ) + return lu_info{ linalg_status::nonfinite, 0, 1 }; + int sign = 1; + for ( std::size_t k = 0; k != n; ++k ) + { + std::size_t p = k; + R best = std::abs( a[k * n + k] ); + for ( std::size_t i = k + 1; i != n; ++i ) + if ( R const v = std::abs( a[i * n + k] ); v > best ) { best = v; p = i; } + if ( p != k ) + { + std::swap_ranges( a + k * n, a + k * n + n, a + p * n ); + std::swap( piv[k], piv[p] ); + sign = -sign; + } + if ( best == R{ 0 } ) continue; // zero column: nothing to eliminate + T const pivot = a[k * n + k]; + for ( std::size_t i = k + 1; i != n; ++i ) + { + T* const ri = a + i * n; + T const l = ri[k] / pivot; + ri[k] = l; + if ( l == T{ 0 } ) continue; + T const* const rk = a + k * n; + for ( std::size_t j = k + 1; j != n; ++j ) ri[j] -= l * rk[j]; + } } + for ( std::size_t i = 0; i != n * n; ++i ) // overflow during elimination + if ( !linalg_isfinite( a[i] ) ) return lu_info{ linalg_status::nonfinite, 0, sign }; + R max_u{ 0 }; + for ( std::size_t i = 0; i != n; ++i ) + for ( std::size_t j = i; j != n; ++j ) + max_u = std::max( max_u, R( std::abs( a[i * n + j] ) ) ); + R const tol = static_cast< R >( n ) * std::numeric_limits< R >::epsilon() * max_u; + std::size_t rank = 0; + for ( std::size_t k = 0; k != n; ++k ) + if ( std::abs( a[k * n + k] ) > tol ) ++rank; + return lu_info{ rank < n ? linalg_status::singular : linalg_status::ok, rank, sign }; + } + + // det = sign(P)·∏u_kk; exactly 0 when rank < n, 1 for n = 0, NaN for a nonfinite input or factor + template < typename T > + T lu_det( T const* lu, std::size_t n, lu_info const& info ) noexcept + { + if ( info.status == linalg_status::nonfinite ) return T( std::numeric_limits< linalg_real_t< T > >::quiet_NaN() ); + if ( info.status != linalg_status::ok ) return T{ 0 }; + T d = T( info.sign ); + for ( std::size_t k = 0; k != n; ++k ) d *= lu[k * n + k]; + return d; + } - if ( 4 == zen.size() ) - { - return zen[0][0] * zen[1][1] - zen[1][0] * zen[0][1]; + // Solves A·X = B from the factors of lu_in_place: b is the row-major n×k right-hand side, x (n×k) gets X. + template < typename T > + void lu_solve( T const* lu, std::size_t n, std::size_t const* piv, T const* b, std::size_t k, T* x ) noexcept + { + for ( std::size_t i = 0; i != n; ++i ) + std::copy( b + piv[i] * k, b + piv[i] * k + k, x + i * k ); + for ( std::size_t i = 0; i != n; ++i ) // L·Y = P·B, unit diagonal + for ( std::size_t j = 0; j != i; ++j ) + if ( T const l = lu[i * n + j]; l != T{ 0 } ) + for ( std::size_t c = 0; c != k; ++c ) x[i * k + c] -= l * x[j * k + c]; + for ( std::size_t r = n; r-- != 0; ) // U·X = Y + { + for ( std::size_t j = r + 1; j != n; ++j ) + if ( T const u = lu[r * n + j]; u != T{ 0 } ) + for ( std::size_t c = 0; c != k; ++c ) x[r * k + c] -= u * x[j * k + c]; + T const d = lu[r * n + r]; + for ( std::size_t c = 0; c != k; ++c ) x[r * k + c] /= d; } + } - size_type const n = zen.row(); - size_type const m = n >> 1; - zen_type const P( zen, range_type( 0, m ), range_type( 0, m ) ); - zen_type const Q( zen, range_type( 0, m ), range_type( m, n ) ); - zen_type const R( zen, range_type( m, n ), range_type( 0, m ) ); - zen_type const S( zen, range_type( m, n ), range_type( m, n ) ); - zen_type const& tmp = S - ( R * ( P.inverse() ) * Q ); + template < typename T > + bool all_finite( T const* x, std::size_t n ) noexcept + { + for ( std::size_t i = 0; i != n; ++i ) + if ( !linalg_isfinite( x[i] ) ) return false; + return true; + } - return P.det() * tmp.det(); + // inverse of the square matrix `m` into `out` (resized to n×n only on success); returns the status + template < typename M > + linalg_status lu_inverse( M const& m, M& out ) noexcept + { + using T = typename M::value_type; + std::size_t const n = m.row(); + M lu( m ); + std::vector< std::size_t > piv( n ); + lu_info const info = lu_in_place( lu.data(), n, piv.data() ); + if ( info.status != linalg_status::ok ) return info.status; + M id( n, n ); + std::fill( id.begin(), id.end(), T{ 0 } ); + for ( std::size_t i = 0; i != n; ++i ) id[i][i] = T{ 1 }; + M x( n, n ); + lu_solve( lu.data(), n, piv.data(), id.data(), n, x.data() ); + if ( !all_finite( x.data(), n * n ) ) return linalg_status::nonfinite; + out = std::move( x ); + return linalg_status::ok; + } + }//namespace matrix_details + + template < typename Matrix, typename Type, Allocator Alloc > + struct crtp_det + { + typedef Matrix zen_type; + typedef crtp_typedef< Type, Alloc > type_proxy_type; + typedef typename type_proxy_type::size_type size_type; + typedef typename type_proxy_type::value_type value_type; + typedef typename type_proxy_type::range_type range_type; + // S7-R2: sign(P)·∏u_kk from the one LU kernel; exactly 0 when rank < n, 1 for 0×0 (D-027) + [[nodiscard]] value_type det() const noexcept requires linalg_element< value_type > + { + zen_type const& zen = static_cast< zen_type const& >( *this ); + better_assert( zen.row()==zen.col(), "matrix::det: expecting a square matrix, but got ", zen.row(), "x", zen.col() ); + size_type const n = zen.row(); + zen_type lu( zen ); + std::vector< std::size_t > piv( n ); + auto const info = matrix_details::lu_in_place( lu.data(), n, piv.data() ); + return matrix_details::lu_det( lu.data(), n, info ); } }; template < typename Matrix, typename Type, Allocator Alloc > @@ -2100,36 +3279,27 @@ namespace feng typedef typename type_proxy_type::const_reverse_diag_type const_reverse_diag_type; diag_type upper_diag_begin( const size_type index ) noexcept { - zen_type& zen = static_cast< zen_type& >( *this ); - return diag_type( zen.begin() + index, zen.col() + 1 ); + return upper_diag_range< diag_type >( static_cast< zen_type& >( *this ), index, false ); } diag_type upper_diag_end( const size_type index ) noexcept { - zen_type& zen = static_cast< zen_type& >( *this ); - size_type depth = std::min( zen.col()-index, zen.row() ); - return diag_type( upper_diag_begin( index ) + depth ); + return upper_diag_range< diag_type >( static_cast< zen_type& >( *this ), index, true ); } const_diag_type upper_diag_begin( const size_type index ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - return const_diag_type( zen.begin() + index, zen.col() + 1 ); + return upper_diag_range< const_diag_type >( static_cast< zen_type const& >( *this ), index, false ); } const_diag_type upper_diag_end( const size_type index ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - size_type depth = std::min( zen.col()-index, zen.row() ); - return upper_diag_begin( index ) + depth; + return upper_diag_range< const_diag_type >( static_cast< zen_type const& >( *this ), index, true ); } const_diag_type upper_diag_cbegin( const size_type index ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - return const_diag_type( zen.cbegin() + index, zen.col() + 1 ); + return upper_diag_begin( index ); } const_diag_type upper_diag_cend( const size_type index ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - size_type depth = std::min( zen.col()-index, zen.row() ); - return upper_diag_cbegin( index ) + depth; + return upper_diag_end( index ); } reverse_upper_diag_type upper_diag_rbegin( const size_type index = 0 ) noexcept { @@ -2149,44 +3319,35 @@ namespace feng } const_reverse_upper_diag_type upper_diag_crbegin( const size_type index = 0 ) const noexcept { - return const_reverse_upper_diag_type( upper_diag_end( index ) ); + return upper_diag_rbegin( index ); } const_reverse_upper_diag_type upper_diag_crend( const size_type index = 0 ) const noexcept { - return const_reverse_upper_diag_type( upper_diag_begin( index ) ); + return upper_diag_rend( index ); } diag_type lower_diag_begin( const size_type index ) noexcept { - zen_type& zen = static_cast< zen_type& >( *this ); - return diag_type( zen.begin() + index * zen.col(), zen.col() + 1 ); + return lower_diag_range< diag_type >( static_cast< zen_type& >( *this ), index, false ); } diag_type lower_diag_end( const size_type index ) noexcept { - zen_type& zen = static_cast< zen_type& >( *this ); - size_type depth = std::min( zen.row()-index, zen.col() ); - return lower_diag_begin( index ) + depth; + return lower_diag_range< diag_type >( static_cast< zen_type& >( *this ), index, true ); } const_diag_type lower_diag_begin( const size_type index ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - return const_diag_type( zen.begin() + index * zen.col(), zen.col() + 1 ); + return lower_diag_range< const_diag_type >( static_cast< zen_type const& >( *this ), index, false ); } const_diag_type lower_diag_end( const size_type index ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - size_type depth = std::min( zen.row()-index, zen.col() ); - return lower_diag_begin( index ) + depth; + return lower_diag_range< const_diag_type >( static_cast< zen_type const& >( *this ), index, true ); } const_diag_type lower_diag_cbegin( const size_type index ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - return const_diag_type( zen.begin() + index * zen.col(), zen.col() + 1 ); + return lower_diag_begin( index ); } const_diag_type lower_diag_cend( const size_type index ) const noexcept { - zen_type const& zen = static_cast< zen_type const& >( *this ); - size_type depth = std::min( zen.row()-index, zen.col() ); - return lower_diag_begin( index ) + depth; + return lower_diag_end( index ); } reverse_lower_diag_type lower_diag_rbegin( const size_type index = 0 ) noexcept { @@ -2206,71 +3367,114 @@ namespace feng } const_reverse_lower_diag_type lower_diag_crbegin( const size_type index = 0 ) const noexcept { - return const_reverse_lower_diag_type( lower_diag_end( index ) ); + return lower_diag_rbegin( index ); } const_reverse_lower_diag_type lower_diag_crend( const size_type index = 0 ) const noexcept { - return const_reverse_lower_diag_type( lower_diag_begin( index ) ); + return lower_diag_rend( index ); } diag_type diag_begin( const difference_type index = 0 ) noexcept { - if ( index > 0 ) return upper_diag_begin( index ); - return lower_diag_begin( -index ); + zen_type& zen = static_cast< zen_type& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_diag_begin( static_cast< size_type >( index ) ); + return lower_diag_begin( static_cast< size_type >( -index ) ); } diag_type diag_end( const difference_type index = 0 ) noexcept { - if ( index > 0 ) return upper_diag_end( index ); - return lower_diag_end( -index ); + zen_type& zen = static_cast< zen_type& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_diag_end( static_cast< size_type >( index ) ); + return lower_diag_end( static_cast< size_type >( -index ) ); } const_diag_type diag_begin( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) return upper_diag_begin( index ); - return lower_diag_begin( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_diag_begin( static_cast< size_type >( index ) ); + return lower_diag_begin( static_cast< size_type >( -index ) ); } const_diag_type diag_end( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) return upper_diag_end( index ); - return lower_diag_end( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_diag_end( static_cast< size_type >( index ) ); + return lower_diag_end( static_cast< size_type >( -index ) ); } const_diag_type diag_cbegin( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) return upper_diag_cbegin( index ); - return lower_diag_cbegin( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_diag_cbegin( static_cast< size_type >( index ) ); + return lower_diag_cbegin( static_cast< size_type >( -index ) ); } const_diag_type diag_cend( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) return upper_diag_cend( index ); - return lower_diag_cend( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_diag_cend( static_cast< size_type >( index ) ); + return lower_diag_cend( static_cast< size_type >( -index ) ); } reverse_diag_type diag_rbegin( const difference_type index = 0 ) noexcept { - if ( index > 0 ) return upper_diag_rbegin( index ); - return lower_diag_rbegin( -index ); + zen_type& zen = static_cast< zen_type& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_diag_rbegin( static_cast< size_type >( index ) ); + return lower_diag_rbegin( static_cast< size_type >( -index ) ); } reverse_diag_type diag_rend( const difference_type index = 0 ) noexcept { - if ( index > 0 ) return upper_diag_rend( index ); - return lower_diag_rend( -index ); + zen_type& zen = static_cast< zen_type& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_diag_rend( static_cast< size_type >( index ) ); + return lower_diag_rend( static_cast< size_type >( -index ) ); } const_reverse_diag_type diag_rbegin( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) return upper_diag_rbegin( index ); - return lower_diag_rbegin( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_diag_rbegin( static_cast< size_type >( index ) ); + return lower_diag_rbegin( static_cast< size_type >( -index ) ); } const_reverse_diag_type diag_rend( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) return upper_diag_rend( index ); - return lower_diag_rend( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_diag_rend( static_cast< size_type >( index ) ); + return lower_diag_rend( static_cast< size_type >( -index ) ); } const_reverse_diag_type diag_crbegin( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) return upper_diag_crbegin( index ); - return lower_diag_crbegin( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_diag_crbegin( static_cast< size_type >( index ) ); + return lower_diag_crbegin( static_cast< size_type >( -index ) ); } const_reverse_diag_type diag_crend( const difference_type index = 0 ) const noexcept { - if ( index > 0 ) return upper_diag_crend( index ); - return lower_diag_crend( -index ); + zen_type const& zen = static_cast< zen_type const& >( *this ); + matrix_private::check_diagonal( index, zen.row(), zen.col() ); + if ( index > 0 ) return upper_diag_crend( static_cast< size_type >( index ) ); + return lower_diag_crend( static_cast< size_type >( -index ) ); + } + private: + // S4-R1: upper diagonal k starts at (0, k), lower diagonal k at (k, 0); stride col()+1. + template < typename It, typename Z > + static It upper_diag_range( Z& zen, size_type index, bool at_end ) noexcept + { + size_type const rows = zen.row(), cols = zen.col(); + matrix_private::check_diagonal( matrix_private::upper_diagonal( index ), rows, cols ); + size_type const length = index < cols ? std::min( cols - index, rows ) : 0; + return matrix_private::stride_range< It >( std::to_address( zen.data() ), zen.size(), index, static_cast< difference_type >( cols ) + 1, length, at_end ); + } + template < typename It, typename Z > + static It lower_diag_range( Z& zen, size_type index, bool at_end ) noexcept + { + size_type const rows = zen.row(), cols = zen.col(); + matrix_private::check_diagonal( matrix_private::lower_diagonal( index ), rows, cols ); + size_type const length = index < rows ? std::min( rows - index, cols ) : 0; + size_type const start = length ? index * cols : 0; + return matrix_private::stride_range< It >( std::to_address( zen.data() ), zen.size(), start, static_cast< difference_type >( cols ) + 1, length, at_end ); } }; template < typename Matrix, typename Type, Allocator Alloc > @@ -2344,6 +3548,26 @@ namespace feng typedef Matrix zen_type; typedef crtp_typedef< Type, Alloc > type_proxy_type; typedef typename type_proxy_type::value_type value_type; + // S6-R5 (D-023): a scalar or matrix of another element type keeps T; the operand is converted as by + // static_cast first (to T's value type for complex T and a real scalar); complex into real is rejected. + template< typename S > requires matrix_details::compound_scalar< S, value_type > + zen_type& operator/=( const S& rhs ) noexcept + { + static_assert( matrix_details::converts_to_element_v< S, value_type >, "feng::matrix operator/=: a complex operand does not convert to a real element type" ); + zen_type& zen = static_cast< zen_type& >( *this ); + auto const v = matrix_details::scalar_as< value_type >( rhs ); + zen.elementwise_apply( [&v]( value_type& x ) { x /= v; } ); + return zen; + } + template< typename U, Allocator B > requires ( !std::same_as< matrix< U, B >, zen_type > ) + zen_type& operator/=( const matrix< U, B >& rhs ) noexcept + { + static_assert( matrix_details::converts_to_element_v< U, value_type >, "feng::matrix operator/=: a complex operand does not convert to a real element type" ); + zen_type& zen = static_cast< zen_type& >( *this ); + zen_type converted{ zen.get_allocator(), rhs.row(), rhs.col() }; + std::transform( rhs.begin(), rhs.end(), converted.begin(), []( U const& x ) noexcept { return static_cast< value_type >( x ); } ); + return zen /= converted; + } zen_type& operator/=( const value_type& rhs ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); @@ -2353,6 +3577,7 @@ namespace feng zen_type& operator/=( const zen_type& rhs ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); + FENG_MATRIX_EXPECTS( rhs.row() == rhs.col() && zen.col() == rhs.row(), "operator /=: operand shape mismatch, ", zen.row(), "x", zen.col(), " / ", rhs.row(), "x", rhs.col(), " (the divisor must be square with as many rows as the dividend has columns)" ); zen *= rhs.inverse(); return zen; } @@ -2365,45 +3590,14 @@ namespace feng typedef typename type_proxy_type::size_type size_type; typedef typename type_proxy_type::value_type value_type; typedef typename type_proxy_type::range_type range_type; - // TODO: block-wise inverse here - const zen_type inverse() const noexcept + // S7-R2: the inverse from the LU factors; an empty 0×0 matrix for a singular or nonfinite input (D-026) + [[nodiscard]] zen_type inverse() const noexcept requires linalg_element< value_type > { zen_type const& zen = static_cast< zen_type const& >( *this ); - better_assert( zen.row() == zen.col(), " Expecting square matrix, but now with row = ", zen.row(), " and col = ", zen.col() ); - - // case of empty matrix - if ( zen.size() == 0 ) + better_assert( zen.row() == zen.col(), "matrix::inverse: expecting a square matrix, but got ", zen.row(), "x", zen.col() ); + zen_type ans; + if ( matrix_details::lu_inverse( zen, ans ) != linalg_status::ok ) return zen_type{}; - - if ( zen.size() == 1 ) - return zen_type{ 1, 1, {value_type{1} / zen[0][0]} }; - - if ( zen.size() == 4 ) - { - auto const [a, b, c, d] = std::make_tuple( zen[0][0], zen[0][1], zen[1][0], zen[1][1] ); - value_type const factor = a*d - b*c; - return zen_type{ 2, 2, { d / factor, -b / factor, -c / factor, a / factor } }; - } - - size_type const N = zen.row(); - size_type const n = N >> 1; - zen_type const& A = zen.clone({0, n}, {0, n}); zen_type const& B = zen.clone({0, n}, {n, N}); - zen_type const& C = zen.clone({n, N}, {0, n}); zen_type const& D = zen.clone({n, N}, {n, N}); - - auto const& D_ = D.inverse(); - auto const& D_C = D_ * C; - auto const& BD_C = B * D_C; - auto const& A__BD_C = A - BD_C; - auto const& A__BD_C_ = A__BD_C.inverse(); - auto const& BD_ = B * D_; - auto const& A__BD_C_BD_ = A__BD_C_ * BD_; - - auto const& E = A__BD_C_; auto const& F = -A__BD_C_BD_; - auto const& G = -D_C * A__BD_C_; auto const& H = D_ + D_C * A__BD_C_BD_; - - zen_type ans{ N, N }; - ans.copy( E, {0, n}, {0, n} ); ans.copy( F, {0, n}, {n, N} ); - ans.copy( G, {n, N}, {0, n} ); ans.copy( H, {n, N}, {n, N} ); return ans; } }; @@ -2413,46 +3607,27 @@ namespace feng typedef Matrix zen_type; typedef typename crtp_typedef< Type, Alloc >::value_type value_type; typedef typename crtp_typedef< Type, Alloc >::size_type size_type; - bool load_txt( std::string const& file_name ) noexcept + [[nodiscard]] bool load_txt( std::string const& file_name ) noexcept { return load_txt(file_name.c_str()); } - bool load_txt( const char* const file_name ) noexcept + // S5-R2/S5-R4 (F08, D-011): reads the whole file, parses it into a temporary with this matrix's allocator + // and commits only on success; on failure returns false, leaves the matrix unchanged and prints one stderr + // line naming load_txt and the reason. Never aborts. + [[nodiscard]] bool load_txt( const char* const file_name ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); - std::ifstream ifs( file_name ); - better_assert( ifs && "matrix::load_txt -- failed to open file" ); - if (!ifs.good()) - { - std::cerr << "matrix::load_txt -- Failed to open file " << file_name << "\n"; + std::vector< std::uint8_t > bytes; + if ( !matrix_details::read_file( file_name, bytes, "matrix::load_txt" ) ) return false; - } - std::stringstream iss; - std::copy( std::istreambuf_iterator< char >( ifs ), std::istreambuf_iterator< char >(), std::ostreambuf_iterator< char >( iss ) ); //TODO: parallel here? - std::string cache = iss.str(); - //for_each( cache.begin(), cache.end(), []( auto & ch ) { if ( ch == ',' || ch == ';' ) ch = ' '; } ); - auto && replace_delimiter_func = [&cache]( size_type idx ){ auto& ch = cache[idx]; ch = (ch==','||ch==';') ? ' ' : ch; }; - matrix_details::parallel( replace_delimiter_func, 0UL, cache.size(), 0UL ); - - iss.str( cache ); - std::vector< value_type > buff; - std::copy( std::istream_iterator< value_type >( iss ), std::istream_iterator< value_type >(), std::back_inserter( buff ) ); // TODO: parallel here? - size_type const total_elements = buff.size(); - const std::string& stream_buff = iss.str(); - size_type const r_ = std::count( stream_buff.begin(), stream_buff.end(), '\n' ); - size_type const r = *( stream_buff.rbegin() ) == '\n' ? r_ : r_ + 1; - size_type const c = total_elements / r; - - if ( r * c != total_elements ) - { - std::cerr << "Error: Failed to match matrix size.\n \tthe size of matrix stored in file \"" << file_name << "\" is " << buff.size() << ".\n"; - std::cerr << " \tthe size of the destination matrix is " << r << " by " << c << " elements.\n"; + zen_type tmp{ zen.get_allocator() }; + char const* why = "invalid text"; + if ( !matrix_details::parse_text< value_type >( reinterpret_cast< char const* >( bytes.data() ), bytes.size(), tmp, &why ) ) + { + std::cerr << "matrix::load_txt -- " << why << ": " << file_name << "\n"; return false; } - - zen.resize( r, c ); - std::copy( buff.begin(), buff.end(), zen.begin() ); - ifs.close(); + zen = std::move( tmp ); return true; } }; @@ -2462,36 +3637,29 @@ namespace feng typedef Matrix zen_type; typedef typename crtp_typedef< Type, Alloc >::value_type value_type; typedef typename crtp_typedef< Type, Alloc >::size_type size_type; - bool load_binary( std::string const& file_name ) noexcept + [[nodiscard]] bool load_binary( std::string const& file_name ) noexcept { return load_binary( file_name.c_str() ); } - bool load_binary( char const* const file_name ) noexcept + // S5-R2/S5-R4 (F08, D-011): reads the whole file, validates the counts against the payload length with + // checked arithmetic and commits only on success; on failure returns false, leaves the matrix unchanged and + // prints one stderr line naming load_binary and the reason. Never aborts. + [[nodiscard]] bool load_binary( char const* const file_name ) noexcept { + static_assert( matrix_details::is_binary_element_v< value_type >, + "load_binary: the native binary format supports arithmetic and std::complex elements only" ); zen_type& zen = static_cast< zen_type& >( *this ); - std::ifstream ifs( file_name, std::ios::binary ); - better_assert( ifs && "matrix::load_binary -- failed to open file" ); - - if ( !ifs ) - return false; - - std::vector< char > buffer{ ( std::istreambuf_iterator< char >( ifs ) ), ( std::istreambuf_iterator< char >() ) }; - better_assert( buffer.size() >= sizeof( size_type ) + sizeof( size_type ) && "matrix::load_library -- file too small, maybe be damaged" ); - - if ( buffer.size() <= sizeof( size_type ) + sizeof( size_type ) ) + std::vector< std::uint8_t > bytes; + if ( !matrix_details::read_file( file_name, bytes, "matrix::load_binary" ) ) return false; - - size_type r; - std::copy( buffer.begin(), buffer.begin() + sizeof( r ), reinterpret_cast< std::int8_t* >( std::addressof( r ) ) ); - size_type c; - std::copy( buffer.begin() + sizeof( r ), buffer.begin() + sizeof( r ) + sizeof( c ), reinterpret_cast< std::int8_t* >( std::addressof( c ) ) ); - zen.resize( r, c ); - better_assert( buffer.size() == sizeof( r ) + sizeof( c ) + sizeof( Type ) * zen.size() && "matrix::load_binary -- data does not match, maybe damaged" ); - - if ( buffer.size() != sizeof( r ) + sizeof( c ) + sizeof( Type ) * zen.size() ) + zen_type tmp{ zen.get_allocator() }; + char const* why = "invalid binary file"; + if ( !matrix_details::parse_binary< value_type >( bytes.data(), bytes.size(), tmp, &why ) ) + { + std::cerr << "matrix::load_binary -- " << why << ": " << file_name << "\n"; return false; - - std::copy( buffer.begin() + sizeof( r ) + sizeof( c ), buffer.end(), reinterpret_cast< std::int8_t* >( zen.data() ) ); + } + zen = std::move( tmp ); return true; } }; @@ -2501,67 +3669,27 @@ namespace feng typedef Matrix zen_type; typedef typename crtp_typedef< Type, Alloc >::value_type value_type; typedef typename crtp_typedef< Type, Alloc >::size_type size_type; - bool load_npy( std::string const& file_name ) noexcept + [[nodiscard]] bool load_npy( std::string const& file_name ) noexcept { return load_npy( file_name.c_str() ); } - bool load_npy( char const* const file_name ) noexcept + // S5-R1/S5-R4 (F07, D-011): reads the whole file, validates and parses it into a temporary with this matrix's + // allocator and commits by move assignment only on success. On failure returns false, leaves the matrix + // unchanged and prints one stderr line naming load_npy and the reason; never aborts. + [[nodiscard]] bool load_npy( char const* const file_name ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); - std::ifstream ifs( file_name, std::ios::binary ); - better_assert( ifs, "matrix::load_npy -- failed to open file ", file_name ); - - if ( !ifs ) + std::vector< std::uint8_t > bytes; + if ( !matrix_details::read_file( file_name, bytes, "matrix::load_npy" ) ) return false; - - std::vector< char > buffer{ ( std::istreambuf_iterator< char >( ifs ) ), ( std::istreambuf_iterator< char >() ) }; - - //get version - std::size_t const version = *(reinterpret_cast< std::uint8_t* >( buffer.data()+6 )); - - //get header length and header - std::uint32_t header_length; - std::string header; - if ( version == 1 ) //version 1 using 2 bytes - { - std::uint32_t const _l0 = *(reinterpret_cast( buffer.data() + 8 )); - std::uint32_t const _l1 = *(reinterpret_cast( buffer.data() + 9 )); - header_length = (_l1 << 8) + _l0; - header = std::string{ buffer.data() + 10, buffer.data() + 10 + header_length }; - } - else //version 2/3 using 4 bytes + zen_type tmp{ zen.get_allocator() }; + char const* why = "invalid NPY file"; + if ( !matrix_details::parse_npy< value_type >( bytes.data(), bytes.size(), tmp, &why ) ) { - std::uint32_t const _l0 = *(reinterpret_cast( buffer.data() + 8 )); - std::uint32_t const _l1 = *(reinterpret_cast( buffer.data() + 9 )); - std::uint32_t const _l2 = *(reinterpret_cast( buffer.data() + 10 )); - std::uint32_t const _l3 = *(reinterpret_cast( buffer.data() + 11 )); - header_length = (_l3 << 24) + (_l2 << 16) + (_l1 << 8) + _l0; - header = std::string{ buffer.data() + 12, buffer.data() + 12 + header_length }; + std::cerr << "matrix::load_npy -- " << why << ": " << file_name << "\n"; + return false; } - - // fortran format or not - bool const row_major = ( header.find("T") != std::string::npos ) ? false : true; - - //extract row and column - std::size_t const shape_pos = header.find("'shape': ("); - std::size_t const row_pos = shape_pos + 10; //start of row - std::size_t const row_pos_end = header.find( ",", row_pos ); //end of row - std::string const row_string = header.substr( row_pos, row_pos_end - row_pos ); - std::size_t const row = std::stoul( row_string ); - std::size_t const col_pos = row_pos_end + 1; //start of col - std::size_t const col_pos_end = header.find( ")", col_pos ); //end of col - std::string const col_string = header.substr( col_pos, col_pos_end - col_pos ); - std::size_t const col = std::stoul( col_string ); - - //resize matrix - zen.resize( row, col ); - if (!row_major) - zen.reshape( col, row ); - - //copy binary value - std::size_t const data_offset = (version==1) ? (10 + header_length) : (12 +header_length); - std::copy_n( reinterpret_cast(buffer.data()+data_offset), row*col, zen.data() ); - + zen = std::move( tmp ); return true; } };//struct crtp_load_npy @@ -2573,6 +3701,26 @@ namespace feng typedef typename type_proxy_type::value_type value_type; typedef typename type_proxy_type::size_type size_type; + // S6-R5 (D-023): a scalar or matrix of another element type keeps T; the operand is converted as by + // static_cast first (to T's value type for complex T and a real scalar); complex into real is rejected. + template< typename S > requires matrix_details::compound_scalar< S, value_type > + zen_type& operator-=( const S& rhs ) noexcept + { + static_assert( matrix_details::converts_to_element_v< S, value_type >, "feng::matrix operator-=: a complex operand does not convert to a real element type" ); + zen_type& zen = static_cast< zen_type& >( *this ); + auto const v = matrix_details::scalar_as< value_type >( rhs ); + zen.elementwise_apply( [&v]( value_type& x ) { x -= v; } ); + return zen; + } + template< typename U, Allocator B > requires ( !std::same_as< matrix< U, B >, zen_type > ) + zen_type& operator-=( const matrix< U, B >& rhs ) noexcept + { + static_assert( matrix_details::converts_to_element_v< U, value_type >, "feng::matrix operator-=: a complex operand does not convert to a real element type" ); + zen_type& zen = static_cast< zen_type& >( *this ); + zen_type converted{ zen.get_allocator(), rhs.row(), rhs.col() }; + std::transform( rhs.begin(), rhs.end(), converted.begin(), []( U const& x ) noexcept { return static_cast< value_type >( x ); } ); + return zen -= converted; + } zen_type& operator-=( const value_type& rhs ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); @@ -2583,16 +3731,152 @@ namespace feng zen_type& operator-=( const zen_type& rhs ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); + FENG_MATRIX_EXPECTS( zen.row() == rhs.row() && zen.col() == rhs.col(), "operator -=: operand shape mismatch, ", zen.row(), "x", zen.col(), " -= ", rhs.row(), "x", rhs.col() ); auto v = zen.data(); auto x = rhs.data(); auto const& elementwise_minus = [v, x]( size_type offset ) { v[offset] -= x[offset]; }; - matrix_details::parallel( elementwise_minus, 0UL, zen.size(), 0UL ); + matrix_details::parallel_work( elementwise_minus, 0UL, zen.size(), 1, matrix_details::elementwise_grain ); return zen; } }; + + namespace matrix_details + { + // S10-R3 (D-008): the GEMM kernel behind operator*= and direct_multiply. C = A * B on row-major contiguous + // arrays (A is m x k, B is k x n, C is m x n). Every C entry is computed as ( ( 0 + a0*b0 ) + a1*b1 ) + ... + // with k ascending, the order std::inner_product used before S10, so results are bit-identical to + // gemm_reference for every element type (under the portable flags, which do not contract a*b+c into FMA). + inline constexpr std::size_t gemm_k_block = 128; + inline constexpr std::size_t gemm_n_block = 256; + + // Rows [i0, i1) of C. B rows are read contiguously (i-k-j order), four C rows share each B load, and the + // k and j loops are blocked so a B panel stays in cache; narrow B (n < 4) uses a row-dot path instead. + template < typename T > + void gemm_rows( T const* __restrict a, T const* __restrict b, T* __restrict c, std::size_t i0, std::size_t i1, std::size_t K, std::size_t N ) noexcept + { + if ( !( i0 < i1 ) || N == 0 ) return; + std::fill( c + i0 * N, c + i1 * N, T( 0 ) ); + if ( K == 0 ) return; + if ( N < 4 ) + { + std::size_t i = i0; + for ( ; i + 4 <= i1; i += 4 ) + { + T const* __restrict a0 = a + i * K; + T const* __restrict a1 = a0 + K; + T const* __restrict a2 = a1 + K; + T const* __restrict a3 = a2 + K; + for ( std::size_t j = 0; j != N; ++j ) + { + T s0 = c[i * N + j], s1 = c[( i + 1 ) * N + j], s2 = c[( i + 2 ) * N + j], s3 = c[( i + 3 ) * N + j]; + for ( std::size_t k = 0; k != K; ++k ) + { + T const bk = b[k * N + j]; + s0 = s0 + a0[k] * bk; + s1 = s1 + a1[k] * bk; + s2 = s2 + a2[k] * bk; + s3 = s3 + a3[k] * bk; + } + c[i * N + j] = s0; + c[( i + 1 ) * N + j] = s1; + c[( i + 2 ) * N + j] = s2; + c[( i + 3 ) * N + j] = s3; + } + } + for ( ; i != i1; ++i ) + { + T const* __restrict a0 = a + i * K; + for ( std::size_t j = 0; j != N; ++j ) + { + T s0 = c[i * N + j]; + for ( std::size_t k = 0; k != K; ++k ) + s0 = s0 + a0[k] * b[k * N + j]; + c[i * N + j] = s0; + } + } + return; + } + for ( std::size_t jb = 0; jb < N; jb += gemm_n_block ) + { + std::size_t const je = std::min( N, jb + gemm_n_block ); + for ( std::size_t kb = 0; kb < K; kb += gemm_k_block ) + { + std::size_t const ke = std::min( K, kb + gemm_k_block ); + std::size_t i = i0; + for ( ; i + 4 <= i1; i += 4 ) + { + T* __restrict c0 = c + i * N; + T* __restrict c1 = c0 + N; + T* __restrict c2 = c1 + N; + T* __restrict c3 = c2 + N; + T const* __restrict ar = a + i * K; + for ( std::size_t k = kb; k != ke; ++k ) + { + T const x0 = ar[k], x1 = ar[K + k], x2 = ar[2 * K + k], x3 = ar[3 * K + k]; + T const* __restrict bk = b + k * N; + for ( std::size_t j = jb; j != je; ++j ) + { + T const y = bk[j]; + c0[j] = c0[j] + x0 * y; + c1[j] = c1[j] + x1 * y; + c2[j] = c2[j] + x2 * y; + c3[j] = c3[j] + x3 * y; + } + } + } + for ( ; i != i1; ++i ) + { + T* __restrict c0 = c + i * N; + T const* __restrict ar = a + i * K; + for ( std::size_t k = kb; k != ke; ++k ) + { + T const x0 = ar[k]; + T const* __restrict bk = b + k * N; + for ( std::size_t j = jb; j != je; ++j ) + c0[j] = c0[j] + x0 * bk[j]; + } + } + } + } + } + + // C = A * B with the rows of C split into contiguous chunks over `workers` threads (parallel_workers; 0 or + // 1 runs serially on the caller). Each entry is computed the same way whatever the split. + template < typename T > + void gemm_blocked( T const* a, T const* b, T* c, std::size_t M, std::size_t K, std::size_t N, std::size_t workers ) noexcept + { + std::size_t const w = effective_workers( std::size_t{ 0 }, M, workers ); + if ( w <= 1 ) + { + gemm_rows( a, b, c, 0, M, K, N ); + return; + } + auto const chunk = [&]( std::size_t k ) noexcept + { + auto const [first, last] = chunk_bounds( std::size_t{ 0 }, M, w, k ); + gemm_rows( a, b, c, first, last, K, N ); + }; + parallel_workers( chunk, std::size_t{ 0 }, w, w ); + } + + // The pre-S10 kernel, kept as the reference and fallback: each C entry is a strided std::inner_product of + // an A row and a B column from value_type( 0 ). The result keeps a's allocator. + template < typename Matrix > + Matrix gemm_reference( Matrix const& a, Matrix const& b ) noexcept + { + better_assert( a.col() == b.row() && "gemm_reference: dimesion not match!", "gemm_reference: operand shape mismatch, ", a.row(), "x", a.col(), " * ", b.row(), "x", b.col() ); + using value_type = typename Matrix::value_type; + Matrix tmp( a.get_allocator(), a.row(), b.col() ); + for ( std::size_t i = 0; i != tmp.row(); ++i ) + for ( std::size_t j = 0; j != tmp.col(); ++j ) + tmp[i][j] = std::inner_product( a.row_begin( i ), a.row_end( i ), b.col_begin( j ), value_type( 0 ) ); + return tmp; + } + }//namespace matrix_details + template < typename Matrix, typename Type, Allocator Alloc > struct crtp_multiply_equal_operator { @@ -2601,6 +3885,26 @@ namespace feng typedef typename type_proxy_type::value_type value_type; typedef typename type_proxy_type::size_type size_type; typedef typename type_proxy_type::range_type range_type; + // S6-R5 (D-023): a scalar or matrix of another element type keeps T; the operand is converted as by + // static_cast first (to T's value type for complex T and a real scalar); complex into real is rejected. + template< typename S > requires matrix_details::compound_scalar< S, value_type > + zen_type& operator*=( const S& rhs ) noexcept + { + static_assert( matrix_details::converts_to_element_v< S, value_type >, "feng::matrix operator*=: a complex operand does not convert to a real element type" ); + zen_type& zen = static_cast< zen_type& >( *this ); + auto const v = matrix_details::scalar_as< value_type >( rhs ); + zen.elementwise_apply( [&v]( value_type& x ) { x *= v; } ); + return zen; + } + template< typename U, Allocator B > requires ( !std::same_as< matrix< U, B >, zen_type > ) + zen_type& operator*=( const matrix< U, B >& rhs ) noexcept + { + static_assert( matrix_details::converts_to_element_v< U, value_type >, "feng::matrix operator*=: a complex operand does not convert to a real element type" ); + zen_type& zen = static_cast< zen_type& >( *this ); + zen_type converted{ zen.get_allocator(), rhs.row(), rhs.col() }; + std::transform( rhs.begin(), rhs.end(), converted.begin(), []( U const& x ) noexcept { return static_cast< value_type >( x ); } ); + return zen *= converted; + } zen_type& operator*=( const value_type& rhs ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); @@ -2612,15 +3916,12 @@ namespace feng zen_type& direct_multiply( const zen_type& other ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); - better_assert( zen.col() == other.row() && "direct_multiply: dimesion not match!" ); - zen_type tmp( zen.row(), other.col() ); - - auto const& func = [&]( size_type i ) - { - for ( size_type j = 0; j != tmp.col(); ++j ) - tmp[i][j] = std::inner_product( zen.row_begin( i ), zen.row_end( i ), other.col_begin( j ), value_type( 0 ) ); - }; - matrix_details::parallel( func, 0UL, tmp.row(), 0UL ); + better_assert( zen.col() == other.row() && "direct_multiply: dimesion not match!", "direct_multiply: operand shape mismatch, ", zen.row(), "x", zen.col(), " * ", other.row(), "x", other.col() ); + zen_type tmp( zen.get_allocator(), zen.row(), other.col() ); // S3-R1: the product stays on this matrix's resource + // S10-R3: the blocked kernel, rows split over work_workers( M*K*N, gemm_grain ) workers (gemm_blocked caps + // them at M); the strided inner_product kernel it replaced is matrix_details::gemm_reference. + std::size_t const work = matrix_details::saturating_work( matrix_details::saturating_work( zen.row(), zen.col() ), other.col() ); + matrix_details::gemm_blocked( zen.data(), other.data(), tmp.data(), zen.row(), zen.col(), other.col(), matrix_details::work_workers( work, matrix_details::gemm_grain ) ); zen.swap( tmp ); return zen; } @@ -2717,55 +4018,10 @@ namespace feng zen_type& operator*=( const zen_type& other ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); - better_assert( zen.col() == other.row() && "operator *= :: the matrix dims not match!" ); - - if constexpr( parallel_mode ) - return direct_multiply( other ); - - static const size_type threshold = 17; - const size_type max_dims = std::max( std::max( zen.row(), zen.col() ), other.col() ); - const size_type min_dims = std::min( std::min( zen.row(), zen.col() ), other.col() ); - - if ( ( max_dims < threshold ) || ( min_dims == 1 ) ) - { - return direct_multiply( other ); - } - - const size_type R = zen.row(); - const size_type C = zen.col(); - const size_type OC = other.col(); - - if ( R & 1 ) - { - if ( R & 2 ) - { - return rr1( other ); - } - - return rr2( other ); - } - - if ( C & 1 ) - { - if ( C & 2 ) - { - return cc1( other ); - } - - return cc2( other ); - } - - if ( OC & 1 ) - { - if ( OC & 2 ) - { - return oc1( other ); - } - - return oc2( other ); - } - - return strassen_multiply( other ); + better_assert( zen.col() == other.row() && "operator *= :: the matrix dims not match!", "operator *=: operand shape mismatch, ", zen.row(), "x", zen.col(), " * ", other.row(), "x", other.col() ); + // S10-R3: every shape goes through the blocked kernel in serial and parallel builds; Strassen + // (strassen_multiply, rr1 ... oc2) is no longer dispatched but stays callable. + return direct_multiply( other ); } }; template < typename Matrix, typename Type, Allocator Alloc > @@ -2775,6 +4031,26 @@ namespace feng typedef crtp_typedef< Type, Alloc > type_proxy_type; typedef typename type_proxy_type::value_type value_type; typedef typename type_proxy_type::size_type size_type; + // S6-R5 (D-023): a scalar or matrix of another element type keeps T; the operand is converted as by + // static_cast first (to T's value type for complex T and a real scalar); complex into real is rejected. + template< typename S > requires matrix_details::compound_scalar< S, value_type > + zen_type& operator+=( const S& rhs ) noexcept + { + static_assert( matrix_details::converts_to_element_v< S, value_type >, "feng::matrix operator+=: a complex operand does not convert to a real element type" ); + zen_type& zen = static_cast< zen_type& >( *this ); + auto const v = matrix_details::scalar_as< value_type >( rhs ); + zen.elementwise_apply( [&v]( value_type& x ) { x += v; } ); + return zen; + } + template< typename U, Allocator B > requires ( !std::same_as< matrix< U, B >, zen_type > ) + zen_type& operator+=( const matrix< U, B >& rhs ) noexcept + { + static_assert( matrix_details::converts_to_element_v< U, value_type >, "feng::matrix operator+=: a complex operand does not convert to a real element type" ); + zen_type& zen = static_cast< zen_type& >( *this ); + zen_type converted{ zen.get_allocator(), rhs.row(), rhs.col() }; + std::transform( rhs.begin(), rhs.end(), converted.begin(), []( U const& x ) noexcept { return static_cast< value_type >( x ); } ); + return zen += converted; + } zen_type& operator+=( const value_type& rhs ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); @@ -2786,6 +4062,7 @@ namespace feng zen_type& operator+=( const zen_type& rhs ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); + FENG_MATRIX_EXPECTS( zen.row() == rhs.row() && zen.col() == rhs.col(), "operator +=: operand shape mismatch, ", zen.row(), "x", zen.col(), " += ", rhs.row(), "x", rhs.col() ); auto x = zen.data(); auto y = rhs.data(); @@ -2793,7 +4070,7 @@ namespace feng { x[offset] += y[offset]; }; - matrix_details::parallel( elementwise_add, 0UL, zen.size(), 0UL ); + matrix_details::parallel_work( elementwise_add, 0UL, zen.size(), 1, matrix_details::elementwise_grain ); return zen; } }; @@ -2804,7 +4081,7 @@ namespace feng typedef crtp_typedef< Type, Alloc > type_proxy_type; typedef typename type_proxy_type::size_type size_type; typedef typename type_proxy_type::value_type value_type; - const zen_type operator-() const noexcept + zen_type operator-() const noexcept { zen_type const& zen = static_cast< zen_type const& >( *this ); zen_type ans{ zen }; @@ -2814,7 +4091,7 @@ namespace feng { x[offset] = -x[offset]; }; - matrix_details::parallel( minus_function, 0UL, ans.size(), 0UL ); + matrix_details::parallel_work( minus_function, 0UL, ans.size(), 1, matrix_details::elementwise_grain ); return ans; } }; @@ -2825,7 +4102,7 @@ namespace feng typedef crtp_typedef< Type, Alloc > type_proxy_type; typedef typename type_proxy_type::value_type value_type; typedef typename type_proxy_type::size_type size_type; - const zen_type operator+() const noexcept + zen_type operator+() const noexcept { return static_cast< zen_type const& >( *this ); } @@ -2839,9 +4116,11 @@ namespace feng zen_type& reshape( const size_type new_row, const size_type new_col ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); - better_assert( new_row * new_col == zen.row() * zen.col() && "error: size before and after reshape does not agree, use resize() instead!" ); - zen.row_ = new_row; - zen.col_ = new_col; + constexpr size_type size_max = std::numeric_limits< size_type >::max(); + FENG_MATRIX_EXPECTS( new_row == 0 || new_col <= size_max / new_row, "matrix reshape: rows*cols overflows, rows = ", new_row, ", cols = ", new_col ); + FENG_MATRIX_EXPECTS( new_row * new_col == zen.size(), "matrix reshape: size before and after reshape does not agree, use resize() instead; size() = ", zen.size(), ", rows = ", new_row, ", cols = ", new_col ); + matrix_private::storage_access::rows( zen ) = new_row; + matrix_private::storage_access::cols( zen ) = new_col; return zen; } }; @@ -2856,7 +4135,7 @@ namespace feng { zen_type& zen = static_cast< zen_type& >( *this ); - if ( zen.size() == new_row * new_col ) + if ( zen.size() == matrix_private::checked_count( zen.get_allocator(), new_row, new_col ) ) { zen.reshape( new_row, new_col ); return zen; @@ -2877,39 +4156,17 @@ namespace feng size_type row() const noexcept { zen_type const& zen = static_cast< zen_type const& >( *this ); - return zen.row_; - } - size_type col() const noexcept - { - zen_type const& zen = static_cast< zen_type const& >( *this ); - return zen.col_; - } - size_type size() const noexcept - { - return row()*col(); - } - }; - - template < typename Matrix, typename Type, Allocator Alloc > - struct crtp_row_col_size_view - { - typedef Matrix zen_type; - typedef crtp_typedef< Type, Alloc > type_proxy_type; - typedef typename type_proxy_type::size_type size_type; - size_type row() const noexcept - { - zen_type const& zen = static_cast< zen_type const& >( *this ); - return zen.row_dim_.second - zen.row_dim_.first; + return matrix_private::storage_access::rows( zen ); } size_type col() const noexcept { zen_type const& zen = static_cast< zen_type const& >( *this ); - return zen.col_dim_.second - zen.col_dim_.first; + return matrix_private::storage_access::cols( zen ); } size_type size() const noexcept { zen_type const& zen = static_cast< zen_type const& >( *this ); - return zen.row()*zen.col(); + return matrix_private::storage_access::storage( zen ).size(); } }; @@ -2981,101 +4238,94 @@ namespace feng }; template < typename Matrix, typename Type, Allocator Alloc > - struct crtp_row_iterator_view : crtp_row_iterator + struct crtp_save_as_txt { typedef Matrix zen_type; - typedef crtp_typedef< Type, Alloc > type_proxy_type; - typedef typename type_proxy_type::size_type size_type; - typedef typename type_proxy_type::row_type row_type; - typedef typename type_proxy_type::const_row_type const_row_type; - typedef typename type_proxy_type::reverse_row_type reverse_row_type; - typedef typename type_proxy_type::const_reverse_row_type const_reverse_row_type; -/* - auto row_begin( const size_type index = 0 ) noexcept + typedef Type value_type; + [[nodiscard]] bool save_as_txt( std::string const& file_name ) const noexcept { - zen_type& zen = static_cast< zen_type& >( *this ); - size_type const row_offset = zen.row_dim_.first; - size_type const col_offset = zen.col_dim_.first; - return zen.matrix_.row_begin(row_offset+index) + col_offset; + return save_as_txt( file_name.c_str() ); } - */ - auto row_begin( const size_type index = 0 ) const noexcept + // S5-R4 (D-011): false with one stderr line on a directory, open, write or close failure; never aborts. + [[nodiscard]] bool save_as_txt( char const * const file_name ) const noexcept { zen_type const& zen = static_cast< zen_type const& >( *this ); - size_type const row_offset = zen.row_dim_.first; - size_type const col_offset = zen.col_dim_.first; - return zen.matrix_.row_cbegin(row_offset+index) + col_offset; + if ( file_name == nullptr ) return matrix_details::write_failed( "matrix::save_as_txt", "no file name given", "" ); + return matrix_details::write_stream( "matrix::save_as_txt", std::string{ file_name }, std::ios_base::out, + [&zen]( std::ofstream& ofs ) { matrix_details::write_text( ofs, zen ); } ); // S5-R2: loads back through load_txt } - - auto row_cbegin( const size_type index = 0 ) const noexcept + }; + template < typename Matrix, typename Type, Allocator Alloc > + struct crtp_save_as_binary + { + typedef Matrix zen_type; + typedef Type value_type; + [[nodiscard]] bool save_as_binary( std::string const& file_name ) const noexcept + { + return save_as_binary( file_name.c_str() ); + } + [[nodiscard]] bool save_as_binary( char const* const file_name ) const noexcept { + static_assert( matrix_details::is_binary_element_v< Type >, + "save_as_binary: the native binary format supports arithmetic and std::complex elements only" ); zen_type const& zen = static_cast< zen_type const& >( *this ); - size_type const row_offset = zen.row_dim_.first; - size_type const col_offset = zen.col_dim_.first; - return zen.matrix_.row_cbegin(row_offset+index) + col_offset; + if ( file_name == nullptr ) return matrix_details::write_failed( "matrix::save_as_binary", "no file name given", "" ); + // S5-R4 (D-011): false with one stderr line on a directory, open, write or close failure; never aborts. + return matrix_details::write_stream( "matrix::save_as_binary", std::string{ file_name }, std::ios_base::out | std::ios_base::binary, + [&zen]( std::ofstream& ofs ) + { + auto const r = zen.row(); + ofs.write( reinterpret_cast< char const* >( std::addressof( r ) ), sizeof( r ) ); + auto const c = zen.col(); + ofs.write( reinterpret_cast< char const* >( std::addressof( c ) ), sizeof( c ) ); + ofs.write( reinterpret_cast< char const* >( zen.data() ), static_cast< std::streamsize >( sizeof( Type ) * zen.size() ) ); + } ); } - }; template < typename Matrix, typename Type, Allocator Alloc > - struct crtp_save_as_txt + struct crtp_save_as_npy { typedef Matrix zen_type; typedef Type value_type; - bool save_as_txt( std::string const& file_name ) const noexcept + [[nodiscard]] bool save_as_npy( std::string const& file_name ) const noexcept { - return save_as_txt( file_name.c_str() ); + return save_as_npy( file_name.c_str() ); } - bool save_as_txt( char const * const file_name ) const noexcept + // D-022: a v1.0 C-order NPY file with the D-021 dtype of Type and the native-order payload; S5-R4 (D-011): + // false with one stderr line on a directory, open, write or close failure; never aborts. + [[nodiscard]] bool save_as_npy( char const* const file_name ) const noexcept { + constexpr char kind = matrix_details::npy_dtype< Type >::kind; + static_assert( kind != '\0', "save_as_npy: the element type has no NPY dtype (D-021)" ); zen_type const& zen = static_cast< zen_type const& >( *this ); - - if ( !matrix_details::create_directory_if_not_present( file_name ) ) - better_assert( !"save_as_txt", " failed to create parent directory: ", " with the target file name ", file_name ); - - std::ofstream ofs( file_name ); - better_assert( ofs, " with the target file name is ", file_name ); - - if ( !ofs ) return false; - - ofs.precision( 16 ); - ofs << zen; - ofs.close(); - return true; - } - }; - template < typename Matrix, typename Type, Allocator Alloc > - struct crtp_save_as_binary - { - typedef Matrix zen_type; - typedef Type value_type; - bool save_as_binary( std::string const& file_name ) const - { - return save_as_binary( file_name.c_str() ); - } - bool save_as_binary( char const* const file_name ) const - { - zen_type const& zen = static_cast< zen_type const& >( *this ); - - if ( !matrix_details::create_directory_if_not_present( file_name ) ) - better_assert( !"save_as_binary", " failed to create directory: ", " with the target file name ", file_name ); - - std::ofstream ofs( file_name, std::ios::out | std::ios::binary ); - better_assert( ofs, " save_as_binary failed to create file: ", " with the target file name ", file_name ); - - if ( !ofs ) return false; - - auto const r = zen.row(); - ofs.write( reinterpret_cast< char const* >( std::addressof( r ) ), sizeof( r ) ); - auto const c = zen.col(); - ofs.write( reinterpret_cast< char const* >( std::addressof( c ) ), sizeof( c ) ); - ofs.write( reinterpret_cast< char const* >( zen.data() ), sizeof( Type ) * zen.size() ); - - better_assert( ofs.good(), " save_as_binary failed to write: ", " with the target file name ", file_name ); - if ( !ofs.good() ) return false; - - ofs.close(); - return true; + if ( file_name == nullptr ) return matrix_details::write_failed( "matrix::save_as_npy", "no file name given", "" ); + + char const order = sizeof( Type ) == 1 ? '|' : ( std::endian::native == std::endian::little ? '<' : '>' ); + std::string header{ "{'descr': '" }; + header += order; + header += kind; + header += std::to_string( sizeof( Type ) ); + header += "', 'fortran_order': False, 'shape': ("; + header += std::to_string( zen.row() ); + header += ", "; + header += std::to_string( zen.col() ); + header += "), }"; + std::size_t const unpadded = 10 + header.size() + 1; // magic, version, length, header, '\n' + header.append( ( 64 - unpadded % 64 ) % 64, ' ' ); + header += '\n'; + std::size_t const header_len = header.size(); // at most a few hundred bytes, fits the v1.0 u16 length + + return matrix_details::write_stream( "matrix::save_as_npy", std::string{ file_name }, std::ios_base::out | std::ios_base::binary, + [&]( std::ofstream& ofs ) + { + char const preamble[10] = { '\x93', 'N', 'U', 'M', 'P', 'Y', '\x01', '\x00', + static_cast< char >( header_len & 0xff ), + static_cast< char >( ( header_len >> 8 ) & 0xff ) }; + ofs.write( preamble, 10 ); + ofs.write( header.data(), static_cast< std::streamsize >( header_len ) ); + ofs.write( reinterpret_cast< char const* >( zen.data() ), static_cast< std::streamsize >( sizeof( Type ) * zen.size() ) ); + } ); } }; @@ -3083,7 +4333,7 @@ namespace feng struct crtp_plot { typedef Matrix zen_type; - bool plot( std::string const& file_name, std::string const& color_map=std::string{"parula"} ) const noexcept + [[nodiscard]] bool plot( std::string const& file_name, std::string const& color_map=std::string{"parula"} ) const noexcept { zen_type const& zen = static_cast< zen_type const& >( *this ); return zen.save_as_bmp( file_name, color_map ); @@ -3093,11 +4343,26 @@ namespace feng namespace { // adapted from https://github.com/miloyip/svpng/blob/master/svpng.inc - inline static void save_png( std::uint8_t* img, unsigned w, unsigned h, int alpha, char const* const file_name ) noexcept - { + // S5-R4 (D-011): returns false, with the reason in `*why`, when the size does not fit the format (width or + // height above 2^31-1, a row or the IDAT chunk length overflowing 32 bits; checked before opening), or when + // fopen, a write or fclose fails. Never aborts. + inline static bool save_png( std::uint8_t const* img, std::uint64_t width, std::uint64_t height, int alpha, char const* const file_name, char const** why = nullptr ) noexcept + { + auto const fail = [why]( char const* reason ) noexcept { if ( why ) *why = reason; return false; }; + constexpr std::uint64_t png_max = 0x7fffffffULL; + if ( width > png_max || height > png_max ) return fail( "image too large for PNG" ); + std::uint64_t row_bytes = 0, idat = 0; + if ( !matrix_details::checked_mul( width, std::uint64_t{ alpha ? 4U : 3U }, row_bytes ) || ++row_bytes > std::numeric_limits< unsigned >::max() ) + return fail( "PNG row size overflows" ); + if ( !matrix_details::checked_mul( height, row_bytes + 5, idat ) || !matrix_details::checked_add( idat, std::uint64_t{ 6 }, idat ) || idat > png_max ) + return fail( "PNG image data size overflows" ); + if ( file_name == nullptr ) return fail( "no file name given" ); + constexpr unsigned t[] = { 0, 0x1db71064, 0x3b6e20c8, 0x26d930ac, 0x76dc4190, 0x6b6b51f4, 0x4db26158, 0x5005713c, 0xedb88320, 0xf00f9344, 0xd6d6a3e8, 0xcb61b38c, 0x9b64c2b0, 0x86d3d2d4, 0xa00ae278, 0xbdbdf21c }; - unsigned a = 1, b = 0, c, p = w * ( alpha ? 4 : 3 ) + 1, x, y, i; + unsigned const w = static_cast< unsigned >( width ), h = static_cast< unsigned >( height ); + unsigned a = 1, b = 0, c, p = static_cast< unsigned >( row_bytes ), x, y, i; FILE* fp = fopen( file_name, "wb" ); + if ( fp == nullptr ) return fail( "failed to open file" ); for ( i = 0; i < 8; i++ ) fputc( ( "\x89PNG\r\n\32\n" )[i], fp );; @@ -3344,7 +4609,10 @@ namespace feng fputc( ( ~c ) & 255, fp ); } - fclose( fp ); + bool const written = ferror( fp ) == 0; + bool const closed = fclose( fp ) == 0; + if ( !written || !closed ) return fail( "failed to write or close file" ); + return true; }//save_png } @@ -3352,7 +4620,7 @@ namespace feng struct crtp_save_as_png { typedef Matrix zen_type; - bool save_as_png( const std::string& file_name, std::string const& color_map = std::string{ "parula" } ) const + [[nodiscard]] bool save_as_png( const std::string& file_name, std::string const& color_map = std::string{ "parula" } ) const noexcept { zen_type const& zen = static_cast< zen_type const& >( *this ); better_assert( zen.row() && "save_as_png: matrix row cannot be zero" ); @@ -3363,9 +4631,11 @@ namespace feng auto&& selected_map = ( *( color_maps.find( map_name ) ) ).second; auto const [the_row, the_col] = zen.shape(); - matrix> channel_r{ the_row, the_col }; - matrix> channel_g{ the_row, the_col }; - matrix> channel_b{ the_row, the_col }; + // dependent on Type so the lookup waits until matrix is complete (F10); same type as before + using byte_matrix = matrix, std::allocator, void>>; + byte_matrix channel_r{ the_row, the_col }; + byte_matrix channel_g{ the_row, the_col }; + byte_matrix channel_b{ the_row, the_col }; auto const& [mn, mx] = zen.minmax(); std::vector cache; @@ -3381,8 +4651,10 @@ namespace feng cache.push_back( b_ ); } - save_png( cache.data(), the_col, the_row, 0, file_name.c_str() ); - + // S5-R4 (D-011): an oversized image, an open, write or close failure returns false with one stderr line + char const* why = ""; + if ( !save_png( cache.data(), the_col, the_row, 0, file_name.c_str(), &why ) ) + return matrix_details::write_failed( "matrix::save_as_png", why, file_name ); return true; } };//struct crtp_save_as_png @@ -3392,7 +4664,7 @@ namespace feng struct crtp_save_as_bmp { typedef Matrix zen_type; - bool save_as_bmp( const std::string& file_name, std::string const& color_map = std::string{ "parula" } ) const + [[nodiscard]] bool save_as_bmp( const std::string& file_name, std::string const& color_map = std::string{ "parula" } ) const noexcept { zen_type const& zen = static_cast< zen_type const& >( *this ); better_assert( zen.row() && "save_as_bmp: matrix row cannot be zero" ); @@ -3403,9 +4675,11 @@ namespace feng auto&& selected_map = ( *( color_maps.find( map_name ) ) ).second; auto const [the_row, the_col] = zen.shape(); - matrix> channel_r{ the_row, the_col }; - matrix> channel_g{ the_row, the_col }; - matrix> channel_b{ the_row, the_col }; + // dependent on Type so the lookup waits until matrix is complete (F10); same type as before + using byte_matrix = matrix, std::allocator, void>>; + byte_matrix channel_r{ the_row, the_col }; + byte_matrix channel_g{ the_row, the_col }; + byte_matrix channel_b{ the_row, the_col }; //auto const& [mx, mn] = std::make_tuple( zen.max(), zen.min() ); auto const& [mn, mx] = zen.minmax(); @@ -3420,32 +4694,26 @@ namespace feng channel_b[row_index][c] = b_; } }; - matrix_details::parallel( make_colormap, 0UL, the_row, 0UL ); + matrix_details::parallel_work( make_colormap, 0UL, the_row, the_col, matrix_details::callback_grain ); auto const& encoding = matrix_details::encode_bmp_stream( channel_r, channel_g, channel_b ); - if ( encoding ) - { - std::string new_file_name{ file_name }; - std::string const extension{ ".bmp" }; - if ( ( new_file_name.size() < 4 ) || ( std::string{ new_file_name.begin() + new_file_name.size() - 4, new_file_name.end() } != extension ) ) - new_file_name += extension; - - if ( !matrix_details::create_directory_if_not_present( new_file_name ) ) - better_assert( !"save_as_bmp", " failed to create directory with the target file name is ", new_file_name ); + if ( !encoding ) return matrix_details::write_failed( "matrix::save_as_bmp", "failed to encode the BMP stream", file_name ); - std::ofstream stream( new_file_name.c_str(), std::ios_base::out | std::ios_base::binary ); - better_assert( stream, " failed to open file with the target file name is ", new_file_name ); - if ( !stream ) return false; + std::string new_file_name{ file_name }; + std::string const extension{ ".bmp" }; + if ( ( new_file_name.size() < 4 ) || ( std::string{ new_file_name.begin() + new_file_name.size() - 4, new_file_name.end() } != extension ) ) + new_file_name += extension; - stream.write( reinterpret_cast((*encoding).data()), (*encoding).size() ); - return true; - } - return false; + // S5-R4 (D-011): false with one stderr line on a directory, open, write or close failure; never aborts. + return matrix_details::write_stream( "matrix::save_as_bmp", new_file_name, std::ios_base::out | std::ios_base::binary, + [&encoding]( std::ofstream& stream ) + { stream.write( reinterpret_cast((*encoding).data()), static_cast( (*encoding).size() ) ); } ); } - bool save_as_bmp( char const* const file_name ) const + [[nodiscard]] bool save_as_bmp( char const* const file_name ) const noexcept { + if ( file_name == nullptr ) return matrix_details::write_failed( "matrix::save_as_bmp", "no file name given", "" ); return save_as_bmp( std::string{ file_name } ); } }; //crtp_save_as_bmp @@ -3457,7 +4725,7 @@ namespace feng struct crtp_save_as_pgm { typedef Matrix zen_type; - bool save_as_pgm( const std::string& file_name ) const noexcept + [[nodiscard]] bool save_as_pgm( const std::string& file_name ) const noexcept { zen_type const& zen = static_cast< zen_type const& >( *this ); std::string new_file_name{ file_name }; @@ -3466,41 +4734,32 @@ namespace feng if ( ( new_file_name.size() < 4 ) || ( std::string{ new_file_name.begin() + new_file_name.size() - 4, new_file_name.end() } != extension ) ) new_file_name += extension; - if ( !matrix_details::create_directory_if_not_present( new_file_name ) ) - better_assert( !"save_as_pgm", " failed to create directory with the target file name is ", new_file_name ); - - std::ofstream stream( new_file_name.c_str() ); - better_assert( stream, " save_as_pgm failed to open file with the target file name is ", new_file_name ); - if ( !stream ) return false; - + // S5-R4 (D-011): false with one stderr line on a directory, open, write or close failure; never aborts. + return matrix_details::write_stream( "matrix::save_as_pgm", new_file_name, std::ios_base::out, [&]( std::ofstream& stream ) { stream << "P2\n"; stream << zen.col() << " " << zen.row() << "\n"; stream << "255\n"; stream << "# Generated Portable GrayMap image for path [" << file_name << "]\n"; - } - double const max_val = static_cast< double >( *std::max_element( zen.begin(), zen.end() ) ); - double const min_val = static_cast< double >( *std::min_element( zen.begin(), zen.end() ) ); - double const divider = max_val - min_val + 1.0e-10; + double const max_val = static_cast< double >( *std::max_element( zen.begin(), zen.end() ) ); + double const min_val = static_cast< double >( *std::min_element( zen.begin(), zen.end() ) ); + double const divider = max_val - min_val + 1.0e-10; - //for ( std::uint_least64_t r = 0; r < zen.row(); r++ ) - for ( auto const r : matrix_details::range( zen.row() ) ) - { - //for ( std::uint_least64_t c = 0; c < zen.col(); c++ ) - for ( auto const c : matrix_details::range( zen.col() ) ) + for ( auto const r : matrix_details::range( zen.row() ) ) { - unsigned long const rgb = static_cast( 256.0 * ( zen[r][c] - min_val ) / divider ); - stream << std::min( rgb, 255UL ) << " "; + for ( auto const c : matrix_details::range( zen.col() ) ) + { + unsigned long const rgb = static_cast( 256.0 * ( zen[r][c] - min_val ) / divider ); + stream << std::min( rgb, 255UL ) << " "; + } + stream << "\n"; } - stream << "\n"; - } - - stream.close(); - return true; + } ); } - bool save_as_pgm( char const* const file_name ) const + [[nodiscard]] bool save_as_pgm( char const* const file_name ) const noexcept { + if ( file_name == nullptr ) return matrix_details::write_failed( "matrix::save_as_pgm", "no file name given", "" ); return save_as_pgm( std::string{ file_name } ); } }; @@ -3517,7 +4776,7 @@ namespace feng // otherwise, drop these elements zen_type& shrink_to_size( const size_type new_row, const size_type new_col ) noexcept { - better_assert( new_row && new_col ); + FENG_MATRIX_EXPECTS( new_row && new_col, "matrix shrink_to_size: zero extent, new_row = ", new_row, ", new_col = ", new_col ); zen_type& zen = static_cast< zen_type& >( *this ); if ( new_row == zen.row() && new_col == zen.col() ) @@ -3529,7 +4788,7 @@ namespace feng size_type const the_cols_to_copy = std::min( zen.col(), new_col ); for ( size_type r = 0; r != the_rows_to_copy; ++r ) - std::copy( zen.row_begin( r ), zen.row_begin( r ) + the_rows_to_copy, other.row_begin( r ) ); + std::copy( zen.row_begin( r ), zen.row_begin( r ) + the_cols_to_copy, other.row_begin( r ) ); zen.swap( other ); return zen; @@ -3542,7 +4801,7 @@ namespace feng typedef crtp_typedef< Type, Alloc > type_proxy_type; typedef typename type_proxy_type::size_type size_type; typedef typename type_proxy_type::value_type value_type; - friend std::ostream& operator<<( std::ostream& lhs, zen_type const& rhs ) + friend std::ostream& operator<<( std::ostream& lhs, zen_type const& rhs ) noexcept { lhs.precision( 18 ); @@ -3554,31 +4813,36 @@ namespace feng return lhs; } - friend std::istream& operator>>( std::istream& is, zen_type& rhs ) + // S5-R2/S5-R4: reads the rest of the stream as load_txt text; on failure sets failbit and leaves rhs unchanged. + friend std::istream& operator>>( std::istream& is, zen_type& rhs ) noexcept { - std::vector< std::string > row_element; - std::string string_line; - - while ( std::getline( is, string_line, '\n' ) ) - row_element.push_back( string_line ); - - size_type const row = row_element.size(); - size_type const col = std::count_if( row_element[0].begin(), row_element[0].end(), []( char ch ) { return '\t' == ch; } ); - - if ( row == 0 || col == 0 ) + std::istream::sentry const ok( is, true ); + if ( !ok ) return is; + std::string text; +#if defined( __cpp_exceptions ) + // libstdc++'s filebuf throws std::ios_base::failure from underflow on a read error (e.g. EISDIR) whatever the + // exception mask, so that one is turned into badbit|failbit. Only it is caught: std::bad_alloc and anything + // else escapes this noexcept function and terminates, as a real allocation failure must (D-012). + try { - is.setstate( std::ios::failbit ); + text.assign( std::istreambuf_iterator< char >( is ), std::istreambuf_iterator< char >() ); + } + catch ( std::ios_base::failure const& ) + { + is.setstate( std::ios::badbit | std::ios::failbit ); return is; } - - rhs.resize( row, col ); - - for ( size_type r = 0; r != row; ++r ) +#else + text.assign( std::istreambuf_iterator< char >( is ), std::istreambuf_iterator< char >() ); +#endif + is.setstate( std::ios::eofbit ); + zen_type tmp{ rhs.get_allocator() }; + if ( !matrix_details::parse_text< value_type >( text.data(), text.size(), tmp ) ) { - std::istringstream the_row( row_element[r] ); - std::copy( std::istream_iterator< value_type >( the_row ), std::istream_iterator< value_type >(), rhs.row_begin( r ) ); + is.setstate( std::ios::failbit ); + return is; } - + rhs = std::move( tmp ); return is; } }; @@ -3589,10 +4853,24 @@ namespace feng void swap( zen_type& other ) noexcept { zen_type& zen = static_cast< zen_type& >( *this ); - std::swap( zen.row_, other.row_ ); - std::swap( zen.col_, other.col_ ); - std::swap( zen.dat_, other.dat_ ); - std::swap( zen.allocator_, other.allocator_ ); + if ( std::addressof( zen ) == std::addressof( other ) ) return; + typedef typename zen_type::allocator_type allocator_type; + typedef matrix_private::storage_access access; + if ( std::allocator_traits< allocator_type >::propagate_on_container_swap::value || zen.get_allocator() == other.get_allocator() ) + { + access::storage( zen ).swap( access::storage( other ) ); + std::swap( access::rows( zen ), access::rows( other ) ); + std::swap( access::cols( zen ), access::cols( other ) ); + return; + } + // S3: unequal allocators and no propagation on swap: exchange the contents element-wise, each side + // keeping its own allocator; the moves below are between equal allocators, so they steal. + zen_type to_zen{ zen.get_allocator(), other.row(), other.col() }; + std::copy( other.begin(), other.end(), to_zen.begin() ); + zen_type to_other{ other.get_allocator(), zen.row(), zen.col() }; + std::copy( zen.begin(), zen.end(), to_other.begin() ); + zen = std::move( to_zen ); + other = std::move( to_other ); } }; template < typename Matrix, typename Type, Allocator Alloc > @@ -3601,7 +4879,7 @@ namespace feng typedef Matrix zen_type; typedef crtp_typedef< Type, Alloc > type_proxy_type; typedef typename type_proxy_type::value_type value_type; - const value_type tr() const noexcept + [[nodiscard]] value_type tr() const noexcept { zen_type const& zen = static_cast< zen_type const& >( *this ); return std::accumulate( zen.diag_begin(), zen.diag_end(), value_type() ); @@ -3613,7 +4891,7 @@ namespace feng typedef Matrix zen_type; typedef crtp_typedef< Type, Alloc > type_proxy_type; typedef typename type_proxy_type::size_type size_type; - const zen_type transpose() const noexcept + [[nodiscard]] zen_type transpose() const noexcept { const zen_type& zen = static_cast< zen_type const& >( *this ); zen_type ans( zen.get_allocator(), zen.col(), zen.row() ); @@ -3632,12 +4910,14 @@ namespace feng typedef Matrix zen_type; template< typename LessThanCompare > - auto minmax( LessThanCompare comp ) const noexcept + [[nodiscard]] auto minmax( LessThanCompare comp ) const noexcept { + // S6-R4 (F15): start from element 0, so all-negative and non-arithmetic inputs give their extremes auto const& zen = static_cast( *this ); - value_type min_val = std::numeric_limits::max(); - value_type max_val = std::numeric_limits::min(); auto const [the_row, the_col] = zen.shape(); + FENG_MATRIX_EXPECTS( the_row != 0 && the_col != 0, "matrix minmax: empty matrix, shape ", the_row, "x", the_col ); + value_type min_val = zen[0][0]; + value_type max_val = zen[0][0]; for ( auto r : matrix_details::range(the_row) ) for ( auto c : matrix_details::range(the_col) ) { @@ -3648,7 +4928,7 @@ namespace feng return std::make_tuple( min_val, max_val ); } - auto minmax() const noexcept + [[nodiscard]] auto minmax() const noexcept { auto const& zen = static_cast( *this ); return zen.minmax( []( value_type const& x, value_type const& y ) noexcept { return x < y; } ); @@ -3662,26 +4942,28 @@ namespace feng typedef Matrix zen_type; template< typename LessThanCompare > - value_type min( LessThanCompare comp ) const noexcept + [[nodiscard]] value_type min( LessThanCompare comp ) const noexcept { auto const& zen = static_cast( *this ); - return matrix_details::reduce( zen.begin(), zen.end(), std::numeric_limits::max(), [&comp]( value_type const& a, value_type const& b ){ return comp(a, b) ? a : b; } ); + FENG_MATRIX_EXPECTS( zen.size() != 0, "matrix min: empty matrix, shape ", zen.row(), "x", zen.col() ); + return matrix_details::reduce( zen.begin() + 1, zen.end(), *zen.begin(), [&comp]( value_type const& a, value_type const& b ){ return comp(a, b) ? a : b; } ); } template< typename LessThanCompare > - value_type max( LessThanCompare comp ) const noexcept + [[nodiscard]] value_type max( LessThanCompare comp ) const noexcept { auto const& zen = static_cast( *this ); - return matrix_details::reduce( zen.begin(), zen.end(), std::numeric_limits::min(), [&comp]( value_type const& a, value_type const& b ){ return comp(a, b) ? b : a; } ); + FENG_MATRIX_EXPECTS( zen.size() != 0, "matrix max: empty matrix, shape ", zen.row(), "x", zen.col() ); + return matrix_details::reduce( zen.begin() + 1, zen.end(), *zen.begin(), [&comp]( value_type const& a, value_type const& b ){ return comp(a, b) ? b : a; } ); } - value_type min() const noexcept + [[nodiscard]] value_type min() const noexcept { auto const& zen = static_cast( *this ); return zen.min( []( value_type const& x, value_type const& y ) noexcept { return x < y; } ); } - value_type max() const noexcept + [[nodiscard]] value_type max() const noexcept { auto const& zen = static_cast( *this ); return zen.max( []( value_type const& x, value_type const& y ) noexcept { return x < y; } ); @@ -3692,34 +4974,196 @@ namespace feng template < typename Matrix, typename Type, Allocator Alloc > using crtp_minmax_view = crtp_minmax; - template < typename Type, Allocator Alloc > + // S4-R2 (F06, F05; D-019): rank-2 views. Each keeps its parent (allocator, checked-mode origin and extent), a + // pointer to the first viewed element (the parent's data() for an empty view), its extents and the parent's + // physical row stride. A view does not own or extend the parent's lifetime; it is invalidated by the + // parent's destruction, reallocation, assignment or shape change. Shared members of both view types. + template < typename View, typename P, typename Matrix > + struct view_members + { + typedef std::size_t size_type; + typedef std::ptrdiff_t difference_type; + typedef std::remove_const_t< std::remove_pointer_t< P > > value_type; + typedef value_type const* const_pointer; + typedef P pointer; + typedef pointer row_type; + typedef const_pointer const_row_type; + typedef stride_iterator< pointer > col_type; + typedef stride_iterator< const_pointer > const_col_type; + typedef std::reverse_iterator< col_type > reverse_col_type; + typedef std::reverse_iterator< const_col_type > const_reverse_col_type; + typedef view_iterator< pointer > iterator; + typedef view_iterator< const_pointer > const_iterator; + typedef std::reverse_iterator< iterator > reverse_iterator; + typedef std::reverse_iterator< const_iterator > const_reverse_iterator; + typedef std::iter_reference_t< pointer > reference; + typedef value_type const& const_reference; + + protected: + Matrix const* parent_ = nullptr; + pointer data_ = nullptr; + size_type rows_ = 0; + size_type cols_ = 0; + size_type row_stride_ = 0; + + view_members() noexcept = default; + view_members( Matrix const* parent, pointer data, size_type rows, size_type cols, size_type row_stride ) noexcept + : parent_( parent ), data_( data ), rows_( rows ), cols_( cols ), row_stride_( row_stride ) { } + + // the first viewed element of rows [r0, r1), columns [c0, c1) of `mat`, after check_view_range + template < typename Q > + static Q first_element( Q origin, std::size_t r0, std::size_t r1, std::size_t c0, std::size_t c1, std::size_t parent_cols ) noexcept + { + return ( r0 == r1 || c0 == c1 ) ? origin : origin + ( r0 * parent_cols + c0 ); + } + + void check_row( size_type r ) const noexcept + { + FENG_MATRIX_EXPECTS( r < rows_ && "Row index out of boundary!", "matrix index: row ", r, " with row() ", rows_ ); + } + + // the parent's data() as a Q; a mutable view was made from a non-const parent, so dropping const is sound + template < typename Q > + Q origin() const noexcept + { + return Q( const_cast< value_type* >( parent_->data() ) ); + } + + template < typename Q > + stride_iterator< Q > column( size_type c, difference_type index ) const noexcept + { + matrix_private::check_column( c, cols_ ); + Q const base = rows_ ? Q( data_ + c ) : Q( data_ ); + return stride_iterator< Q >( base, static_cast< difference_type >( row_stride_ ), static_cast< difference_type >( rows_ ), index, origin< Q >(), parent_->size() ); + } + + template < typename Q > + view_iterator< Q > element( difference_type index ) const noexcept + { + return view_iterator< Q >( Q( data_ ), static_cast< difference_type >( row_stride_ ), static_cast< difference_type >( cols_ ), + static_cast< difference_type >( rows_ * cols_ ), index, origin< Q >(), parent_->size() ); + } + + public: + size_type row() const noexcept { return rows_; } + size_type col() const noexcept { return cols_; } + size_type size() const noexcept { return rows_ * cols_; } + bool empty() const noexcept { return size() == 0; } + size_type row_stride() const noexcept { return row_stride_; } + auto get_allocator() const noexcept { return parent_->get_allocator(); } + + row_type operator[]( size_type r ) const noexcept { check_row( r ); return data_ + r * row_stride_; } + reference at( size_type r, size_type c ) const noexcept + { + matrix_private::check_index( r, c, rows_, cols_ ); + return data_[ r * row_stride_ + c ]; + } + + row_type row_begin( size_type r ) const noexcept { check_row( r ); return data_ + r * row_stride_; } + row_type row_end( size_type r ) const noexcept { return row_begin( r ) + cols_; } + const_row_type row_cbegin( size_type r ) const noexcept { return row_begin( r ); } + const_row_type row_cend( size_type r ) const noexcept { return row_end( r ); } + + col_type col_begin( size_type c ) const noexcept { return column< pointer >( c, 0 ); } + col_type col_end( size_type c ) const noexcept { return column< pointer >( c, static_cast< difference_type >( rows_ ) ); } + const_col_type col_cbegin( size_type c ) const noexcept { return column< const_pointer >( c, 0 ); } + const_col_type col_cend( size_type c ) const noexcept { return column< const_pointer >( c, static_cast< difference_type >( rows_ ) ); } + reverse_col_type col_rbegin( size_type c ) const noexcept { return reverse_col_type( col_end( c ) ); } + reverse_col_type col_rend( size_type c ) const noexcept { return reverse_col_type( col_begin( c ) ); } + const_reverse_col_type col_crbegin( size_type c ) const noexcept { return const_reverse_col_type( col_cend( c ) ); } + const_reverse_col_type col_crend( size_type c ) const noexcept { return const_reverse_col_type( col_cbegin( c ) ); } + + iterator begin() const noexcept { return element< pointer >( 0 ); } + iterator end() const noexcept { return element< pointer >( static_cast< difference_type >( size() ) ); } + const_iterator cbegin() const noexcept { return element< const_pointer >( 0 ); } + const_iterator cend() const noexcept { return element< const_pointer >( static_cast< difference_type >( size() ) ); } + reverse_iterator rbegin() const noexcept { return reverse_iterator( end() ); } + reverse_iterator rend() const noexcept { return reverse_iterator( begin() ); } + const_reverse_iterator crbegin() const noexcept { return const_reverse_iterator( cend() ); } + const_reverse_iterator crend() const noexcept { return const_reverse_iterator( cbegin() ); } + }; + + // the const view; make_view( m, {r0, r1}, {c0, c1} ) of an lvalue owner + template < matrix_element Type, Allocator Alloc > struct matrix_view : + view_members< matrix_view, Type const*, matrix >, crtp_save_as_bmp_view, Type, Alloc>, crtp_shape_view, Type, Alloc>, - crtp_bracket_operator_view, Type, Alloc>, - crtp_minmax_view, Type, Alloc>, - crtp_row_col_size_view, Type, Alloc>, - crtp_row_iterator_view, Type, Alloc>, - crtp_col_iterator_view, Type, Alloc> - { - typedef crtp_typedef< Type, Alloc > type_proxy_type; - typedef typename type_proxy_type::size_type size_type; - //matrix_view( matrix const& mat, std::pair const& row_dim, std::pair const& col_dim ) noexcept; - //template < typename Type, class Alloc > - matrix_view( matrix const& mat, std::pair const& row_dim, std::pair const& col_dim ) noexcept: matrix_{ mat } - { - row_dim_.first = (row_dim.first >= matrix_.row() || row_dim.first >= row_dim.second) ? 0 : row_dim.first; - col_dim_.first = (col_dim.first >= matrix_.col() || col_dim.first >= col_dim.second) ? 0 : col_dim.first; - row_dim_.second = (row_dim.second <= matrix_.row()) ? row_dim.second : matrix_.row(); - col_dim_.second = (col_dim.second <= matrix_.col()) ? col_dim.second : matrix_.col(); - } - - matrix const& matrix_; - std::pair row_dim_; - std::pair col_dim_; + crtp_minmax_view, Type, Alloc> + { + private: + typedef view_members< matrix_view, Type const*, matrix > base_type; + public: + typedef typename base_type::size_type size_type; + typedef typename base_type::value_type value_type; + typedef Alloc allocator_type; + typedef matrix matrix_type; + typedef std::pair range_type; + + matrix_view( matrix_type const& mat, range_type const& row_dim, range_type const& col_dim ) noexcept + { + matrix_private::check_view_range( row_dim.first, row_dim.second, col_dim.first, col_dim.second, mat.row(), mat.col() ); + (*this).parent_ = std::addressof( mat ); + (*this).data_ = base_type::first_element( mat.data(), row_dim.first, row_dim.second, col_dim.first, col_dim.second, mat.col() ); + (*this).rows_ = row_dim.second - row_dim.first; + (*this).cols_ = col_dim.second - col_dim.first; + (*this).row_stride_ = mat.col(); + } + // S4-R3: a view of a temporary owner would dangle + matrix_view( matrix_type&&, range_type const&, range_type const& ) = delete; + matrix_view( matrix_type const&&, range_type const&, range_type const& ) = delete; + + // a mutable view converts to the const view of the same elements + matrix_view( mutable_matrix_view const& other ) noexcept + : base_type( other.parent_, other.data_, other.rows_, other.cols_, other.row_stride_ ) { } + + matrix_view( matrix_view const& ) noexcept = default; + matrix_view& operator=( matrix_view const& ) noexcept = default; + + // read-only access, as for a const matrix + value_type operator()( size_type r, size_type c ) const noexcept { return (*this).at( r, c ); } };//struct matrix_view - template < typename Type, Allocator Alloc = std::allocator > + // the mutable view; make_mutable_view( m, {r0, r1}, {c0, c1} ) of a non-const lvalue owner. Constness is + // shallow, as for std::span: a const mutable view still writes the parent's elements. + template < matrix_element Type, Allocator Alloc > + struct mutable_matrix_view : + view_members< mutable_matrix_view, Type*, matrix >, + crtp_save_as_bmp_view, Type, Alloc>, + crtp_shape_view, Type, Alloc>, + crtp_minmax_view, Type, Alloc> + { + private: + typedef view_members< mutable_matrix_view, Type*, matrix > base_type; + friend struct matrix_view; + public: + typedef typename base_type::size_type size_type; + typedef typename base_type::value_type value_type; + typedef Alloc allocator_type; + typedef matrix matrix_type; + typedef std::pair range_type; + + mutable_matrix_view( matrix_type& mat, range_type const& row_dim, range_type const& col_dim ) noexcept + { + matrix_private::check_view_range( row_dim.first, row_dim.second, col_dim.first, col_dim.second, mat.row(), mat.col() ); + (*this).parent_ = std::addressof( mat ); + (*this).data_ = base_type::first_element( mat.data(), row_dim.first, row_dim.second, col_dim.first, col_dim.second, mat.col() ); + (*this).rows_ = row_dim.second - row_dim.first; + (*this).cols_ = col_dim.second - col_dim.first; + (*this).row_stride_ = mat.col(); + } + // S4-R3: a view of a temporary owner would dangle; a mutable view of a const owner would write through it + mutable_matrix_view( matrix_type const&, range_type const&, range_type const& ) = delete; + mutable_matrix_view( matrix_type&&, range_type const&, range_type const& ) = delete; + mutable_matrix_view( matrix_type const&&, range_type const&, range_type const& ) = delete; + + mutable_matrix_view( mutable_matrix_view const& ) noexcept = default; + mutable_matrix_view& operator=( mutable_matrix_view const& ) noexcept = default; + + typename base_type::reference operator()( size_type r, size_type c ) const noexcept { return (*this).at( r, c ); } + };//struct mutable_matrix_view + + template < matrix_element Type, Allocator Alloc = std::allocator > struct matrix : crtp_anti_diag_iterator< matrix< Type, Alloc >, Type, Alloc > , crtp_apply< matrix< Type, Alloc >, Type, Alloc > , crtp_bracket_operator< matrix< Type, Alloc >, Type, Alloc > @@ -3752,6 +5196,7 @@ namespace feng , crtp_row_col_size< matrix< Type, Alloc >, Type, Alloc > , crtp_row_iterator< matrix< Type, Alloc >, Type, Alloc > , crtp_save_as_binary< matrix< Type, Alloc >, Type, Alloc > + , crtp_save_as_npy< matrix< Type, Alloc >, Type, Alloc > , crtp_save_as_bmp< matrix< Type, Alloc >, Type, Alloc > , crtp_save_as_pgm< matrix< Type, Alloc >, Type, Alloc > , crtp_save_as_png< matrix< Type, Alloc >, Type, Alloc > @@ -3760,6 +5205,7 @@ namespace feng , crtp_shrink_to_size< matrix< Type, Alloc >, Type, Alloc > , crtp_stream_operator< matrix< Type, Alloc >, Type, Alloc > , crtp_swap< matrix< Type, Alloc >, Type, Alloc > + , crtp_tr< matrix< Type, Alloc >, Type, Alloc > , crtp_transpose< matrix< Type, Alloc >, Type, Alloc > { typedef matrix self_type; @@ -3771,86 +5217,92 @@ namespace feng typedef typename type_proxy_type::allocator_type allocator_type; typedef typename type_proxy_type::range_type range_type; - size_type row_; - size_type col_; - pointer dat_; - allocator_type allocator_; + static_assert( std::is_same_v< typename std::allocator_traits< Alloc >::value_type, Type >, + "feng::matrix: the allocator's value_type must be Type; rebind the allocator to the element type" ); - ~matrix() - { - (*this).clear(); - } + private: + friend struct matrix_private::storage_access; + typedef std::allocator_traits< allocator_type > allocator_traits_type; + // S3-R1: private RAII storage; storage_.size() == row_ * col_ after every operation. + size_type row_ = 0; + size_type col_ = 0; + std::vector< value_type, allocator_type > storage_; + + public: + ~matrix() noexcept = default; + + // S3-R2: the free swap follows the member swap (POCS, or element-wise for unequal allocators), not std::swap's + // move-based fallback, which would follow POCMA instead. + friend void swap( self_type& lhs, self_type& rhs ) noexcept { lhs.swap( rhs ); } matrix( std::integral auto row, std::integral auto col, std::initializer_list const& value_list ) noexcept : matrix{} { - (*this).resize( static_cast(row), static_cast(col) ); + FENG_MATRIX_EXPECTS( matrix_private::dimension_fits( row ), "matrix size: negative or unrepresentable rows = ", row ); + FENG_MATRIX_EXPECTS( matrix_private::dimension_fits( col ), "matrix size: negative or unrepresentable cols = ", col ); + matrix_private::checked_count( (*this).get_allocator(), static_cast(row), static_cast(col) ); + (*this).resize( static_cast(row), static_cast(col) ); size_type elements_to_copy = std::min( (*this).size(), value_list.size() ); std::copy( value_list.begin(), value_list.begin()+elements_to_copy, (*this).begin() ); } - matrix( self_type&& other ) noexcept : matrix{} + // S3-R1: steals the storage; the source is left 0x0 with size 0. + matrix( self_type&& other ) noexcept : row_{ other.row_ }, col_{ other.col_ }, storage_( std::move( other.storage_ ) ) { - (*this).swap( other ); + other.storage_.clear(); + other.row_ = 0; + other.col_ = 0; } - self_type& operator = ( self_type&& other ) + // S3-R1: self-move keeps the contents; otherwise the vector honours POCMA and the source is left 0x0. + self_type& operator = ( self_type&& other ) noexcept { - (*this).clear(); - (*this).swap( other ); + if ( this == std::addressof( other ) ) return *this; + // S3-R2: unequal allocators that do not propagate make the vector allocate on this side; check first. + if constexpr ( !allocator_traits_type::propagate_on_container_move_assignment::value && !allocator_traits_type::is_always_equal::value ) + if ( (*this).get_allocator() != other.get_allocator() ) + matrix_private::checked_count( (*this).get_allocator(), other.row_, other.col_ ); + storage_ = std::move( other.storage_ ); + row_ = other.row_; + col_ = other.col_; + other.storage_.clear(); + other.row_ = 0; + other.col_ = 0; return *this; } - matrix( const self_type& other ) noexcept : row_{ other.row_ }, col_{ other.col_ }, dat_{ nullptr }, allocator_{ other.allocator_ } + // S3-R5: the selected allocator may differ from other's (and so may its max_size); check before copying. + matrix( const self_type& other ) noexcept : row_{ other.row_ }, col_{ other.col_ }, + storage_( other.storage_, matrix_private::checked_allocator( allocator_traits_type::select_on_container_copy_construction( other.storage_.get_allocator() ), other.row_, other.col_ ) ) { - if ( row_*col_ != 0 ) - { - dat_ = allocator_.allocate( row_*col_ ); - std::copy_n( other.dat_, row_*col_, dat_ ); - } } template < typename T, Allocator A > - matrix( matrix< T, A> const& other ) noexcept : row_{ other.row_ }, col_{ other.col_ }, dat_{ nullptr }, allocator_{ other.allocator_ } + matrix( matrix< T, A> const& other ) noexcept : matrix{ allocator_type( other.get_allocator() ), other.row(), other.col() } { - if ( row_*col_ != 0 ) - { - dat_ = allocator_.allocate( row_*col_ ); - std::copy_n( other.dat_, row_*col_, dat_ ); - } + std::copy( other.begin(), other.end(), (*this).begin() ); } - explicit matrix( const size_type r = 0, const size_type c = 0, value_type const& v = value_type{} ) noexcept : row_{ r } , col_{ c }, dat_{ nullptr } + explicit matrix( const size_type r = 0, const size_type c = 0, value_type const& v = value_type{} ) noexcept : matrix{ allocator_type{}, r, c, v } { - if ( r*c != 0 ) - { - dat_ = allocator_.allocate( r*c ); - std::fill_n( dat_, r*c, v ); - } } - explicit matrix( allocator_type a, const size_type r = 0, const size_type c = 0, value_type const& v = value_type{} ) noexcept : row_{ r } , col_{ c }, dat_{ nullptr }, allocator_{ a } + explicit matrix( allocator_type a, const size_type r = 0, const size_type c = 0, value_type const& v = value_type{} ) noexcept : row_{ r }, col_{ c }, + storage_( matrix_private::checked_count( a, r, c ), v, a ) { - if ( r*c != 0 ) - { - dat_ = allocator_.allocate( r*c ); - std::fill_n( dat_, r*c, v ); - } } template< typename T, Allocator A > - matrix( matrix const& other, std::initializer_list rr, std::initializer_list rc ) noexcept : row_{0}, col_{0}, dat_{nullptr}, allocator_{ other.get_allocator() } + matrix( matrix const& other, std::initializer_list rr, std::initializer_list rc ) noexcept : storage_( allocator_type( other.get_allocator() ) ) { - better_assert( rr.size() == 2 && "row dims not match!" ); - better_assert( rc.size() == 2 && "col dims not match!" ); + matrix_private::check_clone_lists( rr.size(), rc.size() ); auto [rr0, rr1] = std::make_pair( *(rr.begin()), *(rr.begin()+1) ); auto [rc0, rc1] = std::make_pair( *(rc.begin()), *(rc.begin()+1) ); (*this).clone( other, rr0, rr1, rc0, rc1 ); } - matrix( matrix const& other, std::initializer_list rr, std::initializer_list rc ) noexcept : row_{0}, col_{0}, dat_{nullptr}, allocator_{ other.get_allocator() } + matrix( matrix const& other, std::initializer_list rr, std::initializer_list rc ) noexcept : storage_( other.get_allocator() ) { - better_assert( rr.size() == 2 && "row dims not match!" ); - better_assert( rc.size() == 2 && "col dims not match!" ); + matrix_private::check_clone_lists( rr.size(), rc.size() ); auto [rr0, rr1] = std::make_pair( *(rr.begin()), *(rr.begin()+1) ); auto [rc0, rc1] = std::make_pair( *(rc.begin()), *(rc.begin()+1) ); (*this).clone( other, rr0, rr1, rc0, rc1 ); @@ -3868,13 +5320,22 @@ namespace feng self_type& operator = ( const self_type& rhs ) noexcept { - ( *this ).copy( rhs ); + if ( this == std::addressof( rhs ) ) return *this; // S2-R4: self-assignment keeps the contents + // S3-R1: check the count against the allocator the destination will hold, then let the vector honour POCCA. + if constexpr ( allocator_traits_type::propagate_on_container_copy_assignment::value ) + matrix_private::checked_count( rhs.get_allocator(), rhs.row_, rhs.col_ ); + else + matrix_private::checked_count( (*this).get_allocator(), rhs.row_, rhs.col_ ); + storage_ = rhs.storage_; + row_ = rhs.row_; + col_ = rhs.col_; return *this; } template < typename T, Allocator A > self_type& operator = ( const matrix< T, A >& rhs ) noexcept { + if ( static_cast< void const* >( this ) == static_cast< void const* >( std::addressof( rhs ) ) ) return *this; // S2-R4 ( *this ).copy( rhs ); return *this; } @@ -3885,22 +5346,25 @@ namespace feng return *this; } - //matrix( matrix_view const& view ) noexcept; - template< typename T > - auto const astype() const noexcept + [[nodiscard]] auto astype() const noexcept { matrix:: template rebind_alloc > ans{ (*this).get_allocator(), (*this).row(), (*this).col() }; std::copy( (*this).begin(), (*this).end(), ans.begin() ); //TODO: should move to crtp_xxx return ans; } - //template < typename Type, class Alloc > - matrix( matrix_view const& view ) noexcept + // S3-R4 (F03), S4-R2: copies exactly the viewed rectangle, row by row, using the parent's allocator. + matrix( matrix_view const& v ) noexcept : matrix{ v.get_allocator(), v.row(), v.col() } + { + for ( size_type r = 0; r != v.row(); ++r ) + std::copy( v.row_begin( r ), v.row_end( r ), (*this).row_begin( r ) ); + } + + matrix( mutable_matrix_view const& v ) noexcept : matrix{ v.get_allocator(), v.row(), v.col() } { - (*this).resize( view.row_dim_.second-view.row_dim_.first, view.col_dim_.second-view.col_dim_.first ); - for ( auto const row : matrix_details::range((*this).row() ) )//copy row by row - std::copy( view.matrix_.row_begin(row+view.row_dim_.first), view.matrix_.row_end(row+view.row_dim_.second), (*this).row_begin(row) ); + for ( size_type r = 0; r != v.row(); ++r ) + std::copy( v.row_begin( r ), v.row_end( r ), (*this).row_begin( r ) ); } };//struct matrix @@ -3932,7 +5396,7 @@ namespace feng inline constexpr bool is_complex_matrix_v = is_complex_matrix::value; template< typename M > - concept ComplexMatrix = is_complex_matrix_v; + concept ComplexMatrix = Matrix && is_complex_matrix_v; // refines Matrix so the complex overloads win // // - end of Matrix and ComplexMatrix concepts @@ -3940,6 +5404,92 @@ namespace feng namespace matrix_details { + // integral 0 < floating 1 < complex 2 + template< matrix_scalar T > + struct element_kind : std::integral_constant< int, std::is_integral_v< T > ? 0 : ( std::is_floating_point_v< T > ? 1 : 2 ) > {}; + + template< typename T > + struct element_real { using type = T; }; + template< typename X > + struct element_real< std::complex< X > > { using type = X; }; + template< typename T > + using element_real_t = typename element_real< T >::type; + + // matrix (+) matrix: std::common_type_t for two real types; complex> when either is complex. + template< typename T, typename U > + struct common_element { using type = T; }; // same non-scalar element type: unchanged + template< matrix_scalar T, matrix_scalar U > + struct common_element< T, U > + { + using type = std::conditional_t< element_kind< T >::value == 2 || element_kind< U >::value == 2, + std::complex< std::common_type_t< element_real_t< T >, element_real_t< U > > >, + std::common_type_t< T, U > >; + }; + template< typename T, typename U > + using common_element_t = typename common_element< T, U >::type; + + // matrix (+) scalar: T when kind(S) <= kind(T), else common_element_t. + template< typename T, typename S > + struct scalar_result { using type = T; }; // same non-scalar element type: unchanged + template< matrix_scalar T, matrix_scalar S > + struct scalar_result< T, S > + { + using type = std::conditional_t< ( element_kind< S >::value <= element_kind< T >::value ), T, common_element_t< T, S > >; + }; + template< typename T, typename S > + using scalar_result_t = typename scalar_result< T, S >::type; + + // The operand pairs the operators accept: two scalars, or a non-scalar element type with itself (as before S6). + template< typename S, typename T > + concept scalar_operand_for = ( matrix_scalar< S > && matrix_scalar< T > ) || ( !matrix_scalar< T > && std::same_as< S, T > ); + template< typename T, typename U > + concept element_pair = ( matrix_scalar< T > && matrix_scalar< U > ) || std::same_as< T, U >; + + template< typename R, typename T, Allocator A > + using promoted_matrix_t = matrix< R, typename std::allocator_traits< A >::template rebind_alloc< R > >; + + // A copy of m with elements static_cast to R; the allocator is m's (as a copy selects it) rebound to R. + template< typename R, typename T, Allocator A > + promoted_matrix_t< R, T, A > promote( matrix< T, A > const& m ) noexcept + { + if constexpr ( std::same_as< promoted_matrix_t< R, T, A >, matrix< T, A > > ) + return m; + else + { + typename std::allocator_traits< A >::template rebind_alloc< R > alloc( std::allocator_traits< A >::select_on_container_copy_construction( m.get_allocator() ) ); + promoted_matrix_t< R, T, A > ans{ alloc, m.row(), m.col() }; + std::transform( m.begin(), m.end(), ans.begin(), []( T const& x ) noexcept { return static_cast< R >( x ); } ); + return ans; + } + } + + // S9-R6: the one shape check for the matrix operands of an elementwise function (map, the binary family, fma); + // a mismatch aborts with ": operand shape mismatch, RxC and RxC" (or "RxC, RxC and RxC"). + template< Matrix M, Matrix N > + void expect_same_shape( char const* name, M const& m, N const& n ) noexcept + { + FENG_MATRIX_EXPECTS( m.row() == n.row() && m.col() == n.col(), name, ": operand shape mismatch, ", m.row(), "x", m.col(), " and ", n.row(), "x", n.col() ); + } + + template< Matrix M, Matrix N, Matrix L > + void expect_same_shape( char const* name, M const& m, N const& n, L const& l ) noexcept + { + FENG_MATRIX_EXPECTS( m.row() == n.row() && m.col() == n.col() && m.row() == l.row() && m.col() == l.col(), name, ": operand shape mismatch, ", m.row(), "x", m.col(), ", ", n.row(), "x", n.col(), " and ", l.row(), "x", l.col() ); + } + + // rhs as an operand of a compound operator on a matrix: itself when the types match, else converted. + template< typename R, Allocator B, typename T, Allocator A > + decltype( auto ) operand_as( matrix< R, B > const& target, matrix< T, A > const& rhs ) noexcept + { + if constexpr ( std::same_as< matrix< R, B >, matrix< T, A > > ) + return ( rhs ); + else + { + matrix< R, B > ans{ target.get_allocator(), rhs.row(), rhs.col() }; + std::transform( rhs.begin(), rhs.end(), ans.begin(), []( T const& x ) noexcept { return static_cast< R >( x ); } ); + return ans; + } + } namespace map_impl_private { @@ -3959,6 +5509,7 @@ namespace feng template< typename T, Allocator A, typename T2, Allocator A2 > auto map_impl( matrix const& mat, matrix const& nat ) noexcept { + matrix_details::expect_same_shape( "map", mat, nat ); return [&]( auto const& func ) noexcept { typedef typename std::invoke_result_t value_type; @@ -3972,6 +5523,7 @@ namespace feng template< typename T, Allocator A, typename T2, Allocator A2, typename T3, Allocator A3 > auto map_impl( matrix const& mat, matrix const& nat, matrix const& lat ) noexcept { + matrix_details::expect_same_shape( "map", mat, nat, lat ); return [&]( auto const& func ) noexcept { typedef typename std::invoke_result_t value_type; @@ -4009,10 +5561,11 @@ namespace feng } } + // S4-R4 (F12): the returned lambda holds a copy of func, so it may outlive the argument. template< typename Func > auto map( Func const& func ) noexcept { - return [&]( auto const& ... mat ) noexcept + return [func]( auto const& ... mat ) noexcept { return map_impl_private::map_impl( mat ... )( func ); }; @@ -4020,43 +5573,29 @@ namespace feng namespace reduce_impl_private { + // S4-R4, S6-R1 (F11, F12): index reads only, so no pointer past mat.end() is formed; the fold is + // matrix_details::reduce_range. An optional trailing worker count defaults to work_workers( n, reduce_grain ) (S10-R3). template< typename T, Allocator A > auto reduce_impl( matrix const& mat ) noexcept { - return [&]( auto const& func, auto const& init ) noexcept + return [&mat]( auto const& func, auto const& init, auto const... workers ) noexcept { - auto const& stride_func = [&]( auto start_itor, auto stride, auto end_itor ) noexcept - { - T init_ = init; - for ( auto itor = start_itor; itor < end_itor; itor += stride ) - init_ = func( init_, *itor ); - return init_; - }; - - std::uint_least64_t parallel_size = std::thread::hardware_concurrency(); - - //direct reduce - if ( parallel_size<= 1 || mat.size() < 32 ) - return stride_func( mat.begin(), 1, mat.end() ); - - //parallel reduce to cache - std::vector result_cache( parallel_size ); - auto const& paralle_func = [&]( std::uint_least64_t idx ) noexcept - { - result_cache[idx] = stride_func( mat.begin()+idx, parallel_size, mat.end() ); - }; - matrix_details::parallel( paralle_func, 0UL, parallel_size, 0UL ); - - //final reduce - return stride_func( result_cache.begin(), 1, result_cache.end() ); + static_assert( sizeof...( workers ) <= 1, "reduce_impl( mat )( func, init [, workers] )" ); + std::size_t const n = mat.size(); + T const* const data = mat.data(); + std::size_t w = work_workers( n, reduce_grain ); + ( ( w = static_cast( workers ) ), ... ); + auto const at = [data]( std::size_t i ) noexcept -> T const& { return data[i]; }; + return matrix_details::reduce_range( at, n, static_cast( init ), func, w ); }; } }//reduce_impl_private + // S4-R4 (F12): the returned lambda holds copies of func and init, so it may outlive both. template< typename Func, typename Type > auto reduce( Func const& func, Type init ) noexcept { - return [&]( auto const& mat ) noexcept + return [func, init]( auto const& mat ) noexcept { return reduce_impl_private::reduce_impl( mat )( func, init ); }; @@ -4064,113 +5603,161 @@ namespace feng } - /* - template < typename Type, class Alloc > - matrix_view::matrix_view( matrix const& mat, std::pair const& row_dim, std::pair const& col_dim ) noexcept: matrix_{ mat } + // S4-R2, S4-R3 (D-019): the const view of rows [r0, r1) and columns [c0, c1) of an lvalue owner; each list + // holds exactly two values and the ranges must lie in the owner (checked in every build). + template< typename Type, Allocator Alloc, typename Integer_Type > + [[nodiscard]] matrix_view + make_view( matrix const& owner, std::initializer_list row_dim, std::initializer_list col_dim ) noexcept { - row_dim_.first = (row_dim.first >= matrix_.row() || row_dim.first >= row_dim.second) ? 0 : row_dim.first; - col_dim_.first = (col_dim.first >= matrix_.col() || col_dim.first >= col_dim.second) ? 0 : col_dim.first; - row_dim_.second = (row_dim.second <= matrix_.row()) ? row_dim.second : matrix_.row(); - col_dim_.second = (col_dim.second <= matrix_.col()) ? col_dim.second : matrix_.col(); + auto const rows = matrix_private::view_extent( row_dim, "row" ); + auto const cols = matrix_private::view_extent( col_dim, "column" ); + return matrix_view{ owner, rows, cols }; } - */ - //matrix const& matrix_; - //std::pair row_dim_; - //std::pair col_dim_; + // S4-R3: a view of a temporary owner would dangle. + template< typename Type, Allocator Alloc, typename Integer_Type > + matrix_view + make_view( matrix&& owner, std::initializer_list row_dim, std::initializer_list col_dim ) = delete; - /* - template < typename Type, class Alloc > - matrix::matrix( matrix_view const& view ) noexcept + template< typename Type, Allocator Alloc, typename Integer_Type > + matrix_view + make_view( matrix const&& owner, std::initializer_list row_dim, std::initializer_list col_dim ) = delete; + + // S4-R2, S4-R3 (D-019): the mutable view of rows [r0, r1) and columns [c0, c1) of a non-const lvalue owner. + template< typename Type, Allocator Alloc, typename Integer_Type > + [[nodiscard]] mutable_matrix_view + make_mutable_view( matrix& owner, std::initializer_list row_dim, std::initializer_list col_dim ) noexcept { - (*this).resize( view.row_dim_.second-view.row_dim_.first, view.col_dim_.second-view.col_dim_.first ); - for ( auto const row : matrix_details::range((*this).row() ) )//copy row by row - std::copy( view.matrix_.row_begin(row+view.row_dim_.first), view.matrix_.row_end(row+view.row_dim_.second), (*this).row_begin(row) ); + auto const rows = matrix_private::view_extent( row_dim, "row" ); + auto const cols = matrix_private::view_extent( col_dim, "column" ); + return mutable_matrix_view{ owner, rows, cols }; } - */ + // S4-R3: a mutable view of a const owner would write through it; one of a temporary owner would dangle. template< typename Type, Allocator Alloc, typename Integer_Type > - matrix_view const - make_view( matrix const& matrix_, std::initializer_list row_dim, std::initializer_list col_dim ) noexcept - { - typedef crtp_typedef< Type, Alloc > type_proxy_type; - typedef typename type_proxy_type::size_type size_type; + mutable_matrix_view + make_mutable_view( matrix const& owner, std::initializer_list row_dim, std::initializer_list col_dim ) = delete; - better_assert( row_dim.size() == 2 && "Error: row parameters for a matrix view should be 2!" ); - better_assert( col_dim.size() == 2 && "Error: col parameters for a matrix view should be 2!" ); - return matrix_view - { - matrix_, - std::make_pair( static_cast(*(row_dim.begin())), static_cast(*(row_dim.begin()+1)) ), - std::make_pair( static_cast(*(col_dim.begin())), static_cast(*(col_dim.begin()+1)) ) - }; + template< typename Type, Allocator Alloc, typename Integer_Type > + mutable_matrix_view + make_mutable_view( matrix&& owner, std::initializer_list row_dim, std::initializer_list col_dim ) = delete; + + template< typename Type, Allocator Alloc, typename Integer_Type > + mutable_matrix_view + make_mutable_view( matrix const&& owner, std::initializer_list row_dim, std::initializer_list col_dim ) = delete; + + // S9-R4 (D-032): spans over an owner's m.size() contiguous elements and over row r's m.col() elements + // (std::span< T const > for a const owner); r must name a row. Rvalue owners are rejected, the span would dangle. + template< typename T, Allocator A > + [[nodiscard]] std::span< T > as_span( matrix< T, A >& m ) noexcept { return { std::to_address( m.data() ), m.size() }; } + template< typename T, Allocator A > + [[nodiscard]] std::span< T const > as_span( matrix< T, A > const& m ) noexcept { return { std::to_address( m.data() ), m.size() }; } + template< typename T, Allocator A > + std::span< T const > as_span( matrix< T, A > const&& m ) = delete; + + template< typename M > requires is_matrix_v< std::remove_const_t< M > > + [[nodiscard]] auto row_span( M& m, std::size_t r ) noexcept + { + FENG_MATRIX_EXPECTS( r < m.row(), "row_span: row ", r, " outside a matrix with ", m.row(), " rows" ); + return as_span( m ).subspan( r * m.col(), m.col() ); } + template< typename T, Allocator A > + std::span< T const > row_span( matrix< T, A > const&& m, std::size_t r ) = delete; - template < typename T1, Allocator A1, typename T2, Allocator A2 > - const matrix< T1, A1 > - operator+( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) +#if defined( __cpp_lib_mdspan ) + // S9-R4: the layout_right mdspan of extents (row, col) aliasing m.data(). + template< typename T, Allocator A > + [[nodiscard]] auto to_mdspan( matrix< T, A >& m ) noexcept { return std::mdspan< T, std::dextents< std::size_t, 2 > >( std::to_address( m.data() ), m.row(), m.col() ); } + template< typename T, Allocator A > + [[nodiscard]] auto to_mdspan( matrix< T, A > const& m ) noexcept { return std::mdspan< T const, std::dextents< std::size_t, 2 > >( std::to_address( m.data() ), m.row(), m.col() ); } + template< typename T, Allocator A > + std::mdspan< T const, std::dextents< std::size_t, 2 > > to_mdspan( matrix< T, A > const&& m ) = delete; +#endif +#if defined( __cpp_lib_submdspan ) + // S9-R4: rows [r0, r1) and columns [c0, c1) of to_mdspan( m ), validated as make_view's ranges (S4-R3). + template< typename M, typename Integer_Type > requires is_matrix_v< std::remove_const_t< M > > + [[nodiscard]] auto submdspan( M& m, std::initializer_list< Integer_Type > row_dim, std::initializer_list< Integer_Type > col_dim ) noexcept + { + auto const [r0, r1] = matrix_private::view_extent( row_dim, "row" ); + auto const [c0, c1] = matrix_private::view_extent( col_dim, "column" ); + matrix_private::check_view_range( r0, r1, c0, c1, m.row(), m.col() ); + return std::submdspan( to_mdspan( m ), std::pair{ r0, r1 }, std::pair{ c0, c1 } ); + } + template< typename T, Allocator A, typename Integer_Type > + void submdspan( matrix< T, A > const&& m, std::initializer_list< Integer_Type > row_dim, std::initializer_list< Integer_Type > col_dim ) = delete; +#endif + + // S6-R5 (D-023): matrix (+) matrix gives matrix< common_element_t< T1, T2 >, A1 rebound >; when that is + // matrix< T1, A1 > the result is a copy of lhs as before, otherwise lhs promoted; rhs is converted to the result. + template < typename T1, Allocator A1, typename T2, Allocator A2 > requires matrix_details::element_pair< T1, T2 > + matrix_details::promoted_matrix_t< matrix_details::common_element_t< T1, T2 >, T1, A1 > + operator+( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) noexcept { - matrix< T1, A1 > ans{ lhs }; - ans += rhs; + FENG_MATRIX_EXPECTS( lhs.row() == rhs.row() && lhs.col() == rhs.col(), "operator +: operand shape mismatch, ", lhs.row(), "x", lhs.col(), " + ", rhs.row(), "x", rhs.col() ); + auto ans = matrix_details::promote< matrix_details::common_element_t< T1, T2 > >( lhs ); + ans += matrix_details::operand_as( ans, rhs ); return ans; } - template < typename T1, Allocator A1, typename T2, Allocator A2 > - const matrix< T1, A1 > - operator-( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) + template < typename T1, Allocator A1, typename T2, Allocator A2 > requires matrix_details::element_pair< T1, T2 > + matrix_details::promoted_matrix_t< matrix_details::common_element_t< T1, T2 >, T1, A1 > + operator-( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) noexcept { - matrix< T1, A1 > ans{ lhs }; - ans -= rhs; + FENG_MATRIX_EXPECTS( lhs.row() == rhs.row() && lhs.col() == rhs.col(), "operator -: operand shape mismatch, ", lhs.row(), "x", lhs.col(), " - ", rhs.row(), "x", rhs.col() ); + auto ans = matrix_details::promote< matrix_details::common_element_t< T1, T2 > >( lhs ); + ans -= matrix_details::operand_as( ans, rhs ); return ans; } - template < typename T1, Allocator A1, typename T2, Allocator A2 > - const matrix< T1, A1 > - operator*( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) + template < typename T1, Allocator A1, typename T2, Allocator A2 > requires matrix_details::element_pair< T1, T2 > + matrix_details::promoted_matrix_t< matrix_details::common_element_t< T1, T2 >, T1, A1 > + operator*( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) noexcept { - matrix< T1, A1 > ans( lhs ); - ans *= rhs; + FENG_MATRIX_EXPECTS( lhs.col() == rhs.row(), "operator *: operand shape mismatch, ", lhs.row(), "x", lhs.col(), " * ", rhs.row(), "x", rhs.col() ); + auto ans = matrix_details::promote< matrix_details::common_element_t< T1, T2 > >( lhs ); + ans *= matrix_details::operand_as( ans, rhs ); return ans; } - template < typename T1, Allocator A1, typename T2, Allocator A2 > - const matrix< T1, A1 > - operator/( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) + template < typename T1, Allocator A1, typename T2, Allocator A2 > requires matrix_details::element_pair< T1, T2 > + matrix_details::promoted_matrix_t< matrix_details::common_element_t< T1, T2 >, T1, A1 > + operator/( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) noexcept { - matrix< T1, A1 > ans( lhs ); - ans /= rhs; + FENG_MATRIX_EXPECTS( rhs.row() == rhs.col() && lhs.col() == rhs.row(), "operator /: operand shape mismatch, ", lhs.row(), "x", lhs.col(), " / ", rhs.row(), "x", rhs.col() ); + auto ans = matrix_details::promote< matrix_details::common_element_t< T1, T2 > >( lhs ); + ans /= matrix_details::operand_as( ans, rhs ); return ans; } template < typename T1, Allocator A1, typename T2, Allocator A2 > - bool operator<( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) + bool operator<( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) noexcept { better_assert( lhs.row() == rhs.row() ); better_assert( lhs.col() == rhs.col() ); return std::lexicographical_compare( lhs.begin(), lhs.end(), rhs.begin(), rhs.end() ); } template < typename T1, Allocator A1, typename T2, Allocator A2 > - bool operator==( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) + bool operator==( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) noexcept { better_assert( lhs.row() == rhs.row() ); better_assert( lhs.col() == rhs.col() ); return std::equal( lhs.begin(), lhs.end(), rhs.begin() ); } template < typename T1, Allocator A1, typename T2, Allocator A2 > - bool operator>( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) + bool operator>( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) noexcept { return !( ( lhs < rhs ) || ( lhs == rhs ) ); } template < typename T1, Allocator A1, typename T2, Allocator A2 > - bool operator>=( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) + bool operator>=( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) noexcept { return !( lhs < rhs ); } template < typename T1, Allocator A1, typename T2, Allocator A2 > - bool operator<=( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) + bool operator<=( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) noexcept { return !( lhs > rhs ); } template < typename T1, Allocator A1, typename T2, Allocator A2 > - const matrix< T1, A1 > - operator||( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) + matrix< T1, A1 > + operator||( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) noexcept { if ( lhs.row() == 0 ) return rhs; @@ -4194,8 +5781,8 @@ namespace feng return ans; } template < typename T1, Allocator A1, typename T2, Allocator A2 > - const matrix< T1, A1 > - operator&&( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) + matrix< T1, A1 > + operator&&( const matrix< T1, A1 >& lhs, const matrix< T2, A2 >& rhs ) noexcept { if ( lhs.col() == 0 ) return rhs; @@ -4219,8 +5806,8 @@ namespace feng return ans; } template < typename T, Allocator A, typename T_ > - const matrix< T, A > - operator*( const matrix< T, A >& lhs, const T_* const rhs ) + matrix< T, A > + operator*( const matrix< T, A >& lhs, const T_* const rhs ) noexcept { matrix< T, A > ans( lhs.row(), 1 ); @@ -4230,8 +5817,8 @@ namespace feng return ans; } template < typename T, Allocator A, typename T_ > - const matrix< T, A > - operator*( const T_* lhs, const matrix< T, A >& rhs ) + matrix< T, A > + operator*( const T_* lhs, const matrix< T, A >& rhs ) noexcept { matrix< T, A > ans( 1, rhs.col() ); @@ -4241,10 +5828,10 @@ namespace feng return ans; } template < typename T, Allocator A, typename T_ > - const matrix< T, A > - operator*( const matrix< T, A >& lhs, const std::valarray< T_ >& rhs ) + matrix< T, A > + operator*( const matrix< T, A >& lhs, const std::valarray< T_ >& rhs ) noexcept { - better_assert( lhs.col() == rhs.size() ); + better_assert( lhs.col() == rhs.size(), "matrix * valarray: shape mismatch, matrix columns ", lhs.col(), " but valarray length ", rhs.size() ); matrix< T, A > ans( lhs.row(), 1 ); for ( std::uint_least64_t i = 0; i < lhs.row(); ++i ) @@ -4253,11 +5840,11 @@ namespace feng return ans; } template < typename T, Allocator A, typename T_ > - const matrix< T, A > - operator*( const std::valarray< T_ >& lhs, const matrix< T, A >& rhs ) + matrix< T, A > + operator*( const std::valarray< T_ >& lhs, const matrix< T, A >& rhs ) noexcept { - better_assert( rhs.row() == lhs.size() ); - matrix< T, A > ans( 1, lhs.row() ); + better_assert( rhs.row() == lhs.size(), "valarray * matrix: shape mismatch, valarray length ", lhs.size(), " but matrix rows ", rhs.row() ); + matrix< T, A > ans( 1, rhs.col() ); for ( std::uint_least64_t i = 0; i < rhs.col(); ++i ) ans[0][i] = std::inner_product( std::begin( lhs ), std::begin( lhs ) + rhs.row(), rhs.col_begin( i ), T() ); @@ -4265,10 +5852,10 @@ namespace feng return ans; } template < typename T, Allocator A, typename T_ > - const matrix< T, A > - operator*( const matrix< T, A >& lhs, const std::vector< T_ >& rhs ) + matrix< T, A > + operator*( const matrix< T, A >& lhs, const std::vector< T_ >& rhs ) noexcept { - better_assert( lhs.col() == rhs.size() ); + better_assert( lhs.col() == rhs.size(), "matrix * vector: shape mismatch, matrix columns ", lhs.col(), " but vector length ", rhs.size() ); matrix< T, A > ans( lhs.row(), 1 ); for ( std::uint_least64_t i = 0; i < lhs.row(); ++i ) @@ -4277,91 +5864,94 @@ namespace feng return ans; } template < typename T, Allocator A, typename T_ > - const matrix< T, A > - operator*( const std::vector< T_ >& lhs, const matrix< T, A >& rhs ) + matrix< T, A > + operator*( const std::vector< T_ >& lhs, const matrix< T, A >& rhs ) noexcept { - better_assert( rhs.col() == lhs.size() ); + // S2-R4 (B-003): the vector length must equal the matrix rows; only rhs.row() elements are read. + better_assert( rhs.row() == lhs.size(), "vector * matrix: shape mismatch, vector length ", lhs.size(), " but matrix rows ", rhs.row() ); matrix< T, A > ans( 1, rhs.col() ); for ( std::uint_least64_t i = 0; i < rhs.col(); ++i ) - ans[0][i] = std::inner_product( lhs.begin(), lhs.end(), rhs.col_begin( i ), T() ); + ans[0][i] = std::inner_product( lhs.begin(), lhs.begin() + rhs.row(), rhs.col_begin( i ), T() ); return ans; } template < typename T1, Allocator A1, typename T2, Allocator A2 > - matrix< T1, A1 > const blkdiag( const matrix< T1, A1 >& m1, const matrix< T2, A2 >& m2 ) + [[nodiscard]] matrix< T1, A1 > blkdiag( const matrix< T1, A1 >& m1, const matrix< T2, A2 >& m2 ) noexcept { - return ( m1 || zeros( m1, m1.row(), m2.col() ) ) && ( zeros( m1, m2.row(), m1.col() ) || m2 ); + return ( m1 || matrix< T1, A1 >{ m1.get_allocator(), m1.row(), m2.col() } ) && ( matrix< T1, A1 >{ m1.get_allocator(), m2.row(), m1.col() } || m2 ); } template < typename T1, Allocator A1, typename T2, Allocator A2, typename... Matrices > - matrix< T1, A1 > const blkdiag( const matrix< T1, A1 >& m1, const matrix< T2, A2 >& m2, const Matrices& ... matrices ) + [[nodiscard]] matrix< T1, A1 > blkdiag( const matrix< T1, A1 >& m1, const matrix< T2, A2 >& m2, const Matrices& ... matrices ) noexcept { return blkdiag( blkdiag( m1, m2 ), matrices... ); } template < typename T, Allocator A, typename... Matrices > - matrix< T, A > const blk_diag( const matrix< T, A >& m, const Matrices& ... matrices ) + [[nodiscard]] matrix< T, A > blk_diag( const matrix< T, A >& m, const Matrices& ... matrices ) noexcept { return blkdiag( m, matrices... ); } template < typename T, Allocator A, typename... Matrices > - matrix< T, A > const block_diag( const matrix< T, A >& m, const Matrices& ... matrices ) + [[nodiscard]] matrix< T, A > block_diag( const matrix< T, A >& m, const Matrices& ... matrices ) noexcept { return blkdiag( m, matrices... ); } - template < typename T, Allocator A> - matrix< std::complex< T >, A> const ctranspose( const matrix< std::complex< T >, A>& m ) + template < ComplexMatrix CMat > + [[nodiscard]] CMat ctranspose( CMat const& m ) noexcept { return conj( m.transpose() ); } - template < typename T, Allocator A> - T const det( const matrix< T, A >& m ) + template < typename T, Allocator A> requires linalg_element< T > + [[nodiscard]] T det( const matrix< T, A >& m ) noexcept { return m.det(); } template < typename T, Allocator A> - matrix< T, A > const diag( const matrix< T, A >& m, const std::ptrdiff_t offset = 0 ) + [[nodiscard]] matrix< T, A > diag( const matrix< T, A >& m, const std::ptrdiff_t offset = 0 ) noexcept { const std::uint_least64_t dim = std::min( m.row(), m.col() ) + ( offset > 0 ? offset : -offset ); matrix< T, A > ans{ dim, dim }; - std::copy( m.diag_begin(), m.diag_end(), ans.diag_begin( offset ) ); + if ( m.diag_begin() != m.diag_end() ) // an empty source leaves diagonal `offset` absent from ans + std::copy( m.diag_begin(), m.diag_end(), ans.diag_begin( offset ) ); return ans; } namespace diag_private { template < typename Itor > - matrix< typename std::iterator_traits< Itor >::value_type > const - impl_diag( Itor first, Itor last, const std::ptrdiff_t offset = 0 ) + matrix< typename std::iterator_traits< Itor >::value_type > + impl_diag( Itor first, Itor last, const std::ptrdiff_t offset = 0 ) noexcept { std::uint_least64_t dim = std::distance( first, last ) + ( offset > 0 ? offset : -offset ); matrix< typename std::iterator_traits< Itor >::value_type > ans{ dim, dim }; - std::copy( first, last, ans.diag_begin( offset ) ); + if ( first != last ) // an empty source leaves diagonal `offset` absent from ans + std::copy( first, last, ans.diag_begin( offset ) ); return ans; } } template < typename T, Allocator A> - matrix< T > const diag( const std::vector< T, A >& v, const std::ptrdiff_t offset = 0 ) + [[nodiscard]] matrix< T > diag( const std::vector< T, A >& v, const std::ptrdiff_t offset = 0 ) noexcept { return diag_private::impl_diag( v.begin(), v.end(), offset ); } template < typename T, Allocator A> - matrix< T > const diag( const std::deque< T, A >& v, const std::ptrdiff_t offset = 0 ) + [[nodiscard]] matrix< T > diag( const std::deque< T, A >& v, const std::ptrdiff_t offset = 0 ) noexcept { return diag_private::impl_diag( v.begin(), v.end(), offset ); } template < typename T, typename C, Allocator A> - matrix< T > const diag( const std::set< T, C, A >& v, const std::ptrdiff_t offset = 0 ) + [[nodiscard]] matrix< T > diag( const std::set< T, C, A >& v, const std::ptrdiff_t offset = 0 ) noexcept { return diag_private::impl_diag( v.begin(), v.end(), offset ); } template < typename T, typename C, Allocator A> - matrix< T > const diag( const std::multiset< T, C, A >& v, const std::ptrdiff_t offset = 0 ) + [[nodiscard]] matrix< T > diag( const std::multiset< T, C, A >& v, const std::ptrdiff_t offset = 0 ) noexcept { return diag_private::impl_diag( v.begin(), v.end(), offset ); } template < typename T > - matrix< T > const diag( const std::valarray< T >& v, const std::ptrdiff_t offset = 0 ) + [[nodiscard]] matrix< T > diag( const std::valarray< T >& v, const std::ptrdiff_t offset = 0 ) noexcept { return diag_private::impl_diag( std::begin( v ), std::end( v ), offset ); } @@ -4370,25 +5960,25 @@ namespace feng struct impl_make_diag { std::uint_least64_t pos; - impl_make_diag( const std::uint_least64_t pos_ = 0 ) + impl_make_diag( const std::uint_least64_t pos_ = 0 ) noexcept : pos( pos_ ) { } template < typename T, Allocator A, typename Arg, typename... Args > - void operator()( matrix< T, A >& m, const Arg& arg, const Args& ... args ) const + void operator()( matrix< T, A >& m, const Arg& arg, const Args& ... args ) const noexcept { m[pos][pos] = arg; impl_make_diag( pos + 1 )( m, args... ); } template < typename T, Allocator A, typename Arg > - void operator()( matrix< T, A >& m, const Arg& arg ) const + void operator()( matrix< T, A >& m, const Arg& arg ) const noexcept { *( m.diag_rbegin() ) = arg; } }; } template < typename T, typename... Tn > - matrix< T > const make_diag( const T& v1, const Tn& ... vn ) + [[nodiscard]] matrix< T > make_diag( const T& v1, const Tn& ... vn ) noexcept { const std::uint_least64_t n = 1 + sizeof...( vn ); matrix< T > ans{ n, n }; @@ -4396,18 +5986,18 @@ namespace feng return ans; } template < typename T, Allocator A> - void display( const matrix< T, A >& m ) + void display( const matrix< T, A >& m ) noexcept { std::cout << m << std::endl; } template < typename T, Allocator A> - void disp( const matrix< T, A >& m ) + void disp( const matrix< T, A >& m ) noexcept { display( m ); } template < typename Matrix1, typename Matrix2 > - typename Matrix1::value_type - dot( const Matrix1& m1, const Matrix2& m2 ) + [[nodiscard]] typename Matrix1::value_type + dot( const Matrix1& m1, const Matrix2& m2 ) noexcept { better_assert( m1.row() == m2.row() ); better_assert( m1.col() == m2.col() ); @@ -4418,7 +6008,7 @@ namespace feng template < typename T > struct one_maker { - T operator()() const + T operator()() const noexcept { return T( 1 ); } @@ -4426,80 +6016,66 @@ namespace feng template < typename T > struct one_maker< std::complex< T >> { - std::complex< T > const operator()() const + std::complex< T > operator()() const noexcept { return std::complex< T >( T( 1 ), T( 0 ) ); } }; }; template < typename T, typename A = std::allocator< T > > - matrix< T, A > const eye( const std::uint_least64_t r, const std::uint_least64_t c ) + [[nodiscard]] matrix< T, A > eye( const std::uint_least64_t r, const std::uint_least64_t c ) noexcept { matrix< T > ans{ r, c }; std::fill( ans.diag_begin(), ans.diag_end(), eye_private::one_maker< T >()() ); return ans; } template < typename T, typename A = std::allocator< T > > - matrix< T, A > const eye( const std::uint_least64_t n ) + [[nodiscard]] matrix< T, A > eye( const std::uint_least64_t n ) noexcept { return eye< T, A >( n, n ); } template < typename T, typename A = std::allocator< T > > - matrix< T, A > const eye( const matrix< T, A >& m ) + [[nodiscard]] matrix< T, A > eye( const matrix< T, A >& m ) noexcept { return eye< T, A >( m.row(), m.col() ); } + // S2-R3 (D-004): dim 1 maps row i to row rows-1-i, dim 2 maps column j to column cols-1-j; fewer than two rows + // or columns returns the copy; any other dim aborts with `flipdim`. template < typename T, Allocator A> - matrix< T, A > const flipdim( const matrix< T, A >& m, const std::uint_least64_t dim ) + [[nodiscard]] matrix< T, A > flipdim( const matrix< T, A >& m, const std::uint_least64_t dim ) noexcept { + FENG_MATRIX_EXPECTS( 1 == dim || 2 == dim, "matrix flipdim: dim should be 1 or 2, got ", dim ); matrix< T, A > ans{ m }; if ( 1 == dim ) { - std::uint_least64_t index_upper = 0; - std::uint_least64_t index_lower = m.row() - 1; - - while ( index_lower > index_upper ) - { - std::swap_ranges( ans.row_begin( index_lower ), ans.row_end( index_lower ), ans.row_begin( index_upper ) ); - --index_lower; - ++index_upper; - } - + if ( ans.row() < 2 ) + return ans; + for ( std::uint_least64_t upper = 0, lower = ans.row() - 1; upper < lower; ++upper, --lower ) + std::swap_ranges( ans.row_begin( upper ), ans.row_end( upper ), ans.row_begin( lower ) ); return ans; } - if ( 2 == dim ) - { - std::uint_least64_t index_left = 0; - std::uint_least64_t index_right = m.col() - 1; - - while ( index_right > index_left ) - { - std::swap_ranges( ans.col_begin( index_left ), ans.col_end( index_left ), ans.row_begin( index_right ) ); - --index_right; - ++index_left; - } - + if ( ans.col() < 2 ) return ans; - } - - better_assert( !"the second argument of flipdim should be '1' or '2'" ); + for ( std::uint_least64_t r = 0; r != ans.row(); ++r ) + std::reverse( ans.row_begin( r ), ans.row_end( r ) ); return ans; } + // D-004: MATLAB/NumPy convention, fliplr flips columns and flipud flips rows. template < typename T, Allocator A> - matrix< T, A > const fliplr( const matrix< T, A >& m ) + [[nodiscard]] matrix< T, A > fliplr( const matrix< T, A >& m ) noexcept { - return flipdim( m, 1 ); + return flipdim( m, 2 ); } template < typename T, Allocator A> - matrix< T, A > const flipud( const matrix< T, A >& m ) + [[nodiscard]] matrix< T, A > flipud( const matrix< T, A >& m ) noexcept { - return flipdim( m, 2 ); + return flipdim( m, 1 ); } template < typename T, - typename A = std::allocator< typename std::remove_const< typename std::remove_reference< T >::result_type >::result_type >> - matrix< T, A > const hilb( const std::uint_least64_t n ) + typename A = std::allocator< std::remove_cvref_t< T > >> + [[nodiscard]] matrix< T, A > hilb( const std::uint_least64_t n ) noexcept { matrix< T, A > ans( n, n ); @@ -4513,13 +6089,13 @@ namespace feng return ans; } template < typename T, - typename A = std::allocator< typename std::remove_const< typename std::remove_reference< T >::result_type >::result_type >> - matrix< T, A > const hilbert( const std::uint_least64_t n ) + typename A = std::allocator< std::remove_cvref_t< T > >> + [[nodiscard]] matrix< T, A > hilbert( const std::uint_least64_t n ) noexcept { return hilb< T, A >( n ); } template < typename Matrix > - Matrix const hilb( const std::uint_least64_t n, const Matrix& ) + [[nodiscard]] Matrix hilb( const std::uint_least64_t n, const Matrix& ) noexcept { typedef typename Matrix::value_type value_type; Matrix ans( n, n ); @@ -4534,102 +6110,102 @@ namespace feng return ans; } template < typename Matrix > - Matrix const hilbert( const std::uint_least64_t n, const Matrix& m ) + [[nodiscard]] Matrix hilbert( const std::uint_least64_t n, const Matrix& m ) noexcept { return hilb( n, m ); } - template < typename Matrix > - Matrix const inverse( const Matrix& m ) + template < typename Matrix > requires linalg_element< typename Matrix::value_type > + [[nodiscard]] Matrix inverse( const Matrix& m ) noexcept { return m.inverse(); } - template < typename Matrix > - Matrix const inv( const Matrix& m ) + template < typename Matrix > requires linalg_element< typename Matrix::value_type > + [[nodiscard]] Matrix inv( const Matrix& m ) noexcept { return m.inverse(); } template < typename T, Allocator A> - bool is_column( const matrix< T, A >& m ) + [[nodiscard]] bool is_column( const matrix< T, A >& m ) noexcept { return m.col() == 1; } template < typename T, Allocator A> - bool is_column_matrix( const matrix< T, A >& m ) + [[nodiscard]] bool is_column_matrix( const matrix< T, A >& m ) noexcept { return is_column( m ); } template < typename T, Allocator A> - bool iscolumn( const matrix< T, A >& m ) + [[nodiscard]] bool iscolumn( const matrix< T, A >& m ) noexcept { return is_column( m ); } template < typename T, Allocator A> - bool is_empty( const matrix< T, A >& m ) + [[nodiscard]] bool is_empty( const matrix< T, A >& m ) noexcept { return m.size() == 0; } template < typename T, Allocator A> - bool is_empty_matrix( const matrix< T, A >& m ) + [[nodiscard]] bool is_empty_matrix( const matrix< T, A >& m ) noexcept { return is_empty( m ); } template < typename T, Allocator A> - bool isempty( const matrix< T, A >& m ) + [[nodiscard]] bool isempty( const matrix< T, A >& m ) noexcept { return is_empty( m ); } template < typename T, Allocator A> - bool is_equal( const matrix< T, A >& m1, const matrix< T, A > m2 ) + [[nodiscard]] bool is_equal( const matrix< T, A >& m1, const matrix< T, A > m2 ) noexcept { return m1 == m2; } template < typename M1, typename M2, typename... Mn > - bool is_equal( const M1& m1, const M2& m2, const Mn& ... mn ) + [[nodiscard]] bool is_equal( const M1& m1, const M2& m2, const Mn& ... mn ) noexcept { return is_equal( m1, m2 ) && is_equal( m2, mn... ); } template < typename T, Allocator A> - bool isequal( const matrix< T, A >& m1, const matrix< T, A > m2 ) + [[nodiscard]] bool isequal( const matrix< T, A >& m1, const matrix< T, A > m2 ) noexcept { return is_equal( m1, m2 ); } template < typename M1, typename M2, typename... Mn > - bool isequal( const M1& m1, const M2& m2, const Mn& ... mn ) + [[nodiscard]] bool isequal( const M1& m1, const M2& m2, const Mn& ... mn ) noexcept { return is_equal( m1, m2 ) && is_equal( m2, mn... ); } template < typename T, Allocator A> - matrix< bool > is_inf( const matrix< T, A >& m ) + [[nodiscard]] matrix< std::uint8_t > is_inf( const matrix< T, A >& m ) noexcept // D-017: 0/1 byte mask { - matrix< bool > ans( m.row(), m.col() ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( const T & v, bool & a ) + matrix< std::uint8_t > ans( m.row(), m.col() ); + matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( const T & v, std::uint8_t & a ) { - a = std::isinf( v ) ? true : false; + a = std::isinf( v ) ? 1 : 0; } ); return ans; } template < typename T, Allocator A> - matrix< bool > isinf( const matrix< T, A >& m ) + [[nodiscard]] matrix< std::uint8_t > isinf( const matrix< T, A >& m ) noexcept { return is_inf( m ); } template < typename T, Allocator A> - matrix< bool > is_nan( const matrix< T, A >& m ) + [[nodiscard]] matrix< std::uint8_t > is_nan( const matrix< T, A >& m ) noexcept // D-017: 0/1 byte mask { - matrix< bool > ans( m.row(), m.col() ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( const T & v, bool & a ) + matrix< std::uint8_t > ans( m.row(), m.col() ); + matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( const T & v, std::uint8_t & a ) { - a = std::isnan( v ) ? true : false; + a = std::isnan( v ) ? 1 : 0; } ); return ans; } template < typename T, Allocator A> - matrix< bool > isnan( const matrix< T, A >& m ) + [[nodiscard]] matrix< std::uint8_t > isnan( const matrix< T, A >& m ) noexcept { return is_nan( m ); } template < typename T, Allocator A, typename F > - bool is_orthogonal( const matrix< T, A >& m, F f ) + [[nodiscard]] bool is_orthogonal( const matrix< T, A >& m, F f ) noexcept { if ( m.row() != m.col() ) return false; @@ -4642,7 +6218,7 @@ namespace feng return std::all_of( mm.begin(), mm.end(), f ); } template < typename T, Allocator A> - bool is_orthogonal( const matrix< T, A >& m ) + [[nodiscard]] bool is_orthogonal( const matrix< T, A >& m ) noexcept { return is_orthogonal( m, []( const T v ) { @@ -4650,7 +6226,7 @@ namespace feng } ); } template < typename T, Allocator A> - bool is_positive_definite( const matrix< T, A >& m ) + [[nodiscard]] bool is_positive_definite( const matrix< T, A >& m ) noexcept { typedef matrix< T, A > matrix_type; typedef typename matrix_type::range_type range_type; @@ -4669,22 +6245,22 @@ namespace feng return true; } template < typename T, Allocator A> - bool is_row( const matrix< T, A >& m ) + [[nodiscard]] bool is_row( const matrix< T, A >& m ) noexcept { return m.row() == 1; } template < typename T, Allocator A> - bool is_row_matrix( const matrix< T, A >& m ) + [[nodiscard]] bool is_row_matrix( const matrix< T, A >& m ) noexcept { return is_row( m ); } template < typename T, Allocator A> - bool isrow( const matrix< T, A >& m ) + [[nodiscard]] bool isrow( const matrix< T, A >& m ) noexcept { return is_row( m ); } template < typename T, Allocator A, typename F > - bool is_symmetric( const matrix< T, A >& m, F f ) + [[nodiscard]] bool is_symmetric( const matrix< T, A >& m, F f ) noexcept { if ( m.row() != m.col() ) return false; @@ -4697,7 +6273,7 @@ namespace feng } template < typename T, Allocator A> - bool is_symmetric( const matrix< T, A >& m ) + [[nodiscard]] bool is_symmetric( const matrix< T, A >& m ) noexcept { return is_symmetric( m, []( const T v1, const T v2 ) { @@ -4705,17 +6281,20 @@ namespace feng } ); } - inline matrix< std::uint_least64_t > const magic( const std::uint_least64_t n ) noexcept + // S9-R1: the element type and allocator are parameters; the construction runs in std::uint_least64_t + // arithmetic and stores each value as T, so magic( n ) is unchanged. + template < typename T = std::uint_least64_t, typename A = std::allocator< T > > + [[nodiscard]] matrix< T, A > magic( const std::uint_least64_t n ) noexcept { - matrix< std::uint_least64_t > ans{ n, n }; - - if ( 2 == n ) return ans; // no magic for n = 2 - if ( 3 == n ) - return matrix< std::uint_least64_t >{ 3, 3, { 8, 1, 6, 3, 5, 7, 4, 9, 2 } }; + return matrix< T, A >{ 3, 3, { 8, 1, 6, 3, 5, 7, 4, 9, 2 } }; if ( 4 == n ) - return matrix< std::uint_least64_t >{ 4, 4, { 16, 3, 2, 13, 5, 10, 11, 8, 9, 6, 7, 12, 4, 15, 14, 1} }; + return matrix< T, A >{ 4, 4, { 16, 3, 2, 13, 5, 10, 11, 8, 9, 6, 7, 12, 4, 15, 14, 1} }; + + matrix< T, A > ans{ n, n }; // after the literal cases: one allocation per result + + if ( 2 == n ) return ans; // no magic for n = 2 // odd case if ( n & 1 ) @@ -4723,7 +6302,7 @@ namespace feng for ( std::uint_least64_t i = 0; i < n; ++i ) for ( std::uint_least64_t j = 0; j < n; ++j ) //ans[( ( n - 1 ) / 2 + i - j + n ) % n][( 3 * n - 1 + j - 2 * i ) % n] = i * n + j + 1; - ans[n - (( 3 * n - 1 + j - 2 * i ) % n) - 1][n - (( ( n - 1 ) / 2 + i - j + n ) % n) - 1] = i * n + j + 1; + ans[n - (( 3 * n - 1 + j - 2 * i ) % n) - 1][n - (( ( n - 1 ) / 2 + i - j + n ) % n) - 1] = static_cast< T >( i * n + j + 1 ); return ans; } @@ -4732,14 +6311,14 @@ namespace feng if ( n & 2 ) { auto const half = n >> 1; - auto const& m_half = magic( half ); + auto const m_half = magic< std::uint_least64_t >( half ); // L for ( auto r : matrix_details::range( (half+1) >> 1 ) ) for ( auto c : matrix_details::range( half ) ) { auto const val = m_half[r][c]; - ans[r<<1][c<<1] = val << 2; ans[r<<1][(c<<1)+1] = (val << 2) - 3; - ans[(r<<1)+1][c<<1] = (val << 2) - 2; ans[(r<<1)+1][(c<<1)+1] = (val << 2) - 1; + ans[r<<1][c<<1] = static_cast< T >( val << 2 ); ans[r<<1][(c<<1)+1] = static_cast< T >( (val << 2) - 3 ); + ans[(r<<1)+1][c<<1] = static_cast< T >( (val << 2) - 2 ); ans[(r<<1)+1][(c<<1)+1] = static_cast< T >( (val << 2) - 1 ); } // U @@ -4747,8 +6326,8 @@ namespace feng { auto const r = (half+1) >> 1; auto const val = m_half[r][c]; - ans[r<<1][c<<1] = (val << 2) - 3; ans[r<<1][(c<<1)+1] = (val << 2); - ans[(r<<1)+1][c<<1] = (val << 2) - 2; ans[(r<<1)+1][(c<<1)+1] = (val << 2) - 1; + ans[r<<1][c<<1] = static_cast< T >( (val << 2) - 3 ); ans[r<<1][(c<<1)+1] = static_cast< T >( (val << 2) ); + ans[(r<<1)+1][c<<1] = static_cast< T >( (val << 2) - 2 ); ans[(r<<1)+1][(c<<1)+1] = static_cast< T >( (val << 2) - 1 ); } // swap central block @@ -4757,14 +6336,14 @@ namespace feng { auto const [r,c] = std::make_tuple( (half-1)>>1, (half-1)>>1 ); auto const val = m_half[r][c]; - ans[r<<1][c<<1] = (val << 2) - 3; ans[r<<1][(c<<1)+1] = (val << 2); - ans[(r<<1)+1][c<<1] = (val << 2) - 2; ans[(r<<1)+1][(c<<1)+1] = (val << 2) - 1; + ans[r<<1][c<<1] = static_cast< T >( (val << 2) - 3 ); ans[r<<1][(c<<1)+1] = static_cast< T >( (val << 2) ); + ans[(r<<1)+1][c<<1] = static_cast< T >( (val << 2) - 2 ); ans[(r<<1)+1][(c<<1)+1] = static_cast< T >( (val << 2) - 1 ); } { auto const [r,c] = std::make_tuple( (half+1)>>1, (half+1)>>1 ); auto const val = m_half[r][c]; - ans[r<<1][c<<1] = (val << 2); ans[r<<1][(c<<1)+1] = (val << 2) - 3; - ans[(r<<1)+1][c<<1] = (val << 2) - 2; ans[(r<<1)+1][(c<<1)+1] = (val << 2) - 1; + ans[r<<1][c<<1] = static_cast< T >( (val << 2) ); ans[r<<1][(c<<1)+1] = static_cast< T >( (val << 2) - 3 ); + ans[(r<<1)+1][c<<1] = static_cast< T >( (val << 2) - 2 ); ans[(r<<1)+1][(c<<1)+1] = static_cast< T >( (val << 2) - 1 ); } } // X @@ -4772,15 +6351,15 @@ namespace feng for ( auto c : matrix_details::range( half ) ) { auto const& val = m_half[r][c]; - ans[r<<1][c<<1] = (val << 2) - 3; ans[r<<1][(c<<1)+1] = (val << 2); - ans[(r<<1)+1][c<<1] = (val << 2) - 1; ans[(r<<1)+1][(c<<1)+1] = (val << 2) - 2; + ans[r<<1][c<<1] = static_cast< T >( (val << 2) - 3 ); ans[r<<1][(c<<1)+1] = static_cast< T >( (val << 2) ); + ans[(r<<1)+1][c<<1] = static_cast< T >( (val << 2) - 1 ); ans[(r<<1)+1][(c<<1)+1] = static_cast< T >( (val << 2) - 2 ); } return ans; } // doubly even - std::iota( ans.begin(), ans.end(), 1 ); + std::iota( ans.begin(), ans.end(), static_cast< T >( 1 ) ); std::reverse( ans.diag_begin(), ans.diag_end() ); std::reverse( ans.anti_diag_begin(), ans.anti_diag_end() ); std::swap_ranges( ans.lower_diag_begin( n >> 1 ), ans.lower_diag_end( n >> 1 ), ans.upper_diag_rbegin( n >> 1 ) ); @@ -4789,21 +6368,21 @@ namespace feng } template< Matrix Mat > - auto const max( Mat const& m ) + [[nodiscard]] auto max( Mat const& m ) noexcept { - better_assert( m.size() ); + FENG_MATRIX_EXPECTS( m.size() != 0, "feng::max: empty matrix, shape ", m.row(), "x", m.col() ); return *std::max_element( m.begin(), m.end() ); } template< Matrix Mat > - auto const min( Mat const& m ) + [[nodiscard]] auto min( Mat const& m ) noexcept { - better_assert( m.size() ); + FENG_MATRIX_EXPECTS( m.size() != 0, "feng::min: empty matrix, shape ", m.row(), "x", m.col() ); return *std::min_element( m.begin(), m.end() ); } template < typename T > - matrix arange( std::uint_least64_t start, const std::uint_least64_t stop, const std::uint_least64_t step = 1ULL ) + [[nodiscard]] matrix arange( std::uint_least64_t start, const std::uint_least64_t stop, const std::uint_least64_t step = 1ULL ) noexcept { matrix ans{ 1, stop-start/step }; for ( auto& v : ans ) @@ -4815,7 +6394,7 @@ namespace feng } template < typename T > - matrix arange( const std::uint_least64_t length ) + [[nodiscard]] matrix arange( const std::uint_least64_t length ) noexcept { matrix ans{ 1, length }; std::iota( ans.begin(), ans.end(), T{0} ); @@ -4823,7 +6402,7 @@ namespace feng } template < typename T > - matrix linspace( T start, T stop, const std::uint_least64_t num = 50ULL, bool end_point=true ) + [[nodiscard]] matrix linspace( T start, T stop, const std::uint_least64_t num = 50ULL, bool end_point=true ) noexcept { if ( 0 == num ) return matrix{}; @@ -4841,452 +6420,520 @@ namespace feng } template < Matrix Mat > - Mat const ones_like( Mat const& mat ) noexcept + [[nodiscard]] Mat ones_like( Mat const& mat ) noexcept { return Mat{ mat.get_allocator(), mat.row(), mat.col(), typename Mat::value_type{1} }; } template < typename T > - matrix> const ones( std::integral auto r, std::integral auto c ) noexcept + [[nodiscard]] matrix> ones( std::integral auto r, std::integral auto c ) noexcept { matrix< T > ans{ static_cast(r), static_cast(c), T{ 1 } }; return ans; } template < typename T > - matrix> const ones( std::integral auto n ) noexcept + [[nodiscard]] matrix> ones( std::integral auto n ) noexcept { return ones< T >( n, n ); } template < typename T, Allocator A> - matrix< T, A > const ones( A const& alloc, std::integral auto r, std::integral auto c ) noexcept + [[nodiscard]] matrix< T, A > ones( A const& alloc, std::integral auto r, std::integral auto c ) noexcept { - return { alloc, static_cast(r) , static_cast(c), T{1} }; + return matrix< T, A >{ alloc, static_cast(r) , static_cast(c), T{1} }; } template < typename T, Allocator A> - matrix< T, A > const ones( A const& alloc, std::integral auto n ) noexcept + [[nodiscard]] matrix< T, A > ones( A const& alloc, std::integral auto n ) noexcept { - return { alloc, static_cast(n) , static_cast(n), T{1} }; + return matrix< T, A >{ alloc, static_cast(n) , static_cast(n), T{1} }; } template < Matrix Mat > - Mat const zeros_like( Mat const& mat ) noexcept + [[nodiscard]] Mat zeros_like( Mat const& mat ) noexcept { return Mat{ mat.get_allocator(), mat.row(), mat.col(), typename Mat::value_type{} }; } template < typename T > - matrix> const zeros( std::integral auto r, std::integral auto c ) noexcept + [[nodiscard]] matrix> zeros( std::integral auto r, std::integral auto c ) noexcept { matrix< T > ans{ static_cast(r), static_cast(c), T{} }; return ans; } template < typename T > - matrix> const zeros( std::integral auto n ) noexcept + [[nodiscard]] matrix> zeros( std::integral auto n ) noexcept { return zeros< T >( n, n ); } template < typename T, Allocator A> - matrix< T, A > const zeros( A const& alloc, std::integral auto r, std::integral auto c ) noexcept + [[nodiscard]] matrix< T, A > zeros( A const& alloc, std::integral auto r, std::integral auto c ) noexcept { - return { alloc, static_cast(r) , static_cast(c), T{} }; + return matrix< T, A >{ alloc, static_cast(r) , static_cast(c), T{} }; } template < typename T, Allocator A> - matrix< T, A > const zeros( A const& alloc, std::integral auto n ) noexcept + [[nodiscard]] matrix< T, A > zeros( A const& alloc, std::integral auto n ) noexcept { - return { alloc, static_cast(n) , static_cast(n), T{} }; + return matrix< T, A >{ alloc, static_cast(n) , static_cast(n), T{} }; } template < typename T > - auto const empty( const std::integral auto r, const std::integral auto c ) noexcept + [[nodiscard]] auto empty( const std::integral auto r, const std::integral auto c ) noexcept { return matrix< T >{ static_cast(r), static_cast(c) }; } template < typename T, Allocator A> - matrix< T, A > const empty( A const& alloc, std::integral auto r, std::integral auto c ) noexcept + [[nodiscard]] matrix< T, A > empty( A const& alloc, std::integral auto r, std::integral auto c ) noexcept { - return { alloc, r , c }; + return matrix< T, A >{ alloc, static_cast(r), static_cast(c) }; } - template < typename T, typename A_ = std::allocator< T >> - std::uint_least64_t - singular_value_decomposition( matrix const& A, - matrix& u, - matrix& w, - matrix& v, - std::uint_least64_t const max_its = 1000 ) - { - typedef T value_type; - const value_type zero{ 0 }; - const value_type one{ 1 }; - auto const [m, n] = A.shape(); - u = A; - w.resize( n, n ); - v.resize( n, n ); - std::uint_least64_t i{ 0 }, l{ 0 }; - value_type c{ 0 }, f{ 0 }, h{ 0 }; - std::vector< value_type > arr( n ); - value_type g = zero; - value_type s = zero; - value_type scale = zero; - value_type anorm = zero; - - for ( i = 0; i < n; ++i ) - { - l = i + 2; - arr[i] = scale * g; - g = zero; - s = zero; - scale = zero; - - if ( i < m ) - { - scale = std::accumulate( u.col_begin( i ) + i, u.col_end( i ), value_type( 0 ), []( value_type v1, value_type v2 ) { return v1 + std::abs( v2 ); } ); - - if ( scale != zero ) - { - matrix_details::for_each( u.col_begin( i ) + i, u.col_end( i ), [scale]( value_type & v ) { v /= scale; } ); - const value_type tmp_s = std::inner_product( u.col_begin( i ) + i, u.col_end( i ), u.col_begin( i ) + i, value_type( 0 ) ); - g = ( u[i][i] >= zero ) ? -std::sqrt( tmp_s ) : std::sqrt( tmp_s ); - const value_type tmp_h = u[i][i] * g - tmp_s; - u[i][i] -= g; - - for ( std::uint_least64_t j = l - 1; j < n; ++j ) - { - const value_type tmp_ss = std::inner_product( u.col_begin( i ) + i, u.col_end( i ), u.col_begin( j ) + i, value_type( 0 ) ); - std::transform( u.col_begin( j ) + i, u.col_end( j ), u.col_begin( i ) + i, u.col_begin( j ) + i, [tmp_ss, tmp_h]( value_type v1, value_type v2 ) { return v1 + tmp_ss * v2 / tmp_h; } ); - } - - matrix_details::for_each( u.col_begin( i ) + i, u.col_end( i ), [scale]( value_type & v ) { v *= scale; } ); - } - } - - w[i][i] = scale * g; - g = zero; - s = zero; - scale = zero; - - if ( i + 1 <= m && i != n ) - { - scale = std::accumulate( u.row_begin( i ) + l - 1, u.row_end( i ), value_type( 0 ), []( value_type v1, value_type v2 ) { return v1 + std::abs( v2 ); } ); + // S7-R3 (F13, D-026, D-028): one-sided Hestenes Jacobi SVD of an m×n A, real or complex. The kernel runs on + // B = A (m ≥ n) or B = Aᴴ (m < n, the roles of U and V are then swapped), with complex rotations for complex T; + // a sweep visits every column pair and B has converged when every pair has |b_iᴴb_j| ≤ ε·‖b_i‖‖b_j‖. Thin + // factors: u() is m×k, v() is n×k, k = min(m, n), and s() is a std::vector of the k singular values (the real + // type of T), descending, with A = U·diag(s)·Vᴴ. A u column (or, for m < n, v column) whose singular value is + // exactly zero is completed by Gram–Schmidt on unit vectors, so UᴴU = VᴴV = I on all k columns. status() is ok, + // not_converged (the sweep limit was reached; the factors hold the last iterate) or nonfinite (a NaN or inf in + // A; the factors are then empty). A is scaled by its largest |a_ij| for the iteration and s is scaled back. + template < typename T, Allocator A_ = std::allocator< T > > requires linalg_element< T > + class svd_factorization + { + public: + using matrix_type = matrix< T, A_ >; + using real_type = matrix_details::linalg_real_t< T >; - if ( scale != zero ) - { - matrix_details::for_each( u.row_begin( i ) + l - 1, u.row_end( i ), [scale]( value_type & v ) { v /= scale; } ); - auto const tmp_s = std::inner_product( u.row_begin( i ) + l - 1, u.row_end( i ), u.row_begin( i ) + l - 1, value_type( 0 ) ); - g = ( u[i][l - 1] >= zero ) ? -std::sqrt( tmp_s ) : std::sqrt( tmp_s ); - auto const tmp_h = u[i][l - 1] * g - tmp_s; - u[i][l - 1] -= g; - std::transform( u.row_begin( i ) + l - 1, u.row_end( i ), arr.begin() + l - 1, [tmp_h]( value_type v ) { return v / tmp_h; } ); - - for ( std::uint_least64_t j = l - 1; j < m; ++j ) - { - const value_type tmp_ss = std::inner_product( u.row_begin( j ) + l - 1, u.row_end( j ), u.row_begin( i ) + l - 1, value_type( 0 ) ); - std::transform( u.row_begin( j ) + l - 1, u.row_end( j ), arr.begin() + l - 1, u.row_begin( j ) + l - 1, [tmp_ss]( value_type v1, value_type v2 ) { return v1 + tmp_ss * v2; } ); - } + svd_factorization() noexcept = default; + explicit svd_factorization( matrix_type const& a, std::size_t max_sweeps = 64 ) noexcept : m_( a.row() ), n_( a.col() ) + { + factor( a, max_sweeps ); + } - matrix_details::for_each( u.row_begin( i ) + l - 1, u.row_end( i ), [scale]( value_type & v ) - { - v *= scale; - } ); - } - } + [[nodiscard]] linalg_status status() const noexcept { return status_; } + [[nodiscard]] bool ok() const noexcept { return status_ == linalg_status::ok; } + [[nodiscard]] std::size_t sweeps() const noexcept { return sweeps_; } + [[nodiscard]] matrix_type const& u() const noexcept { return u_; } + [[nodiscard]] matrix_type const& v() const noexcept { return v_; } + [[nodiscard]] std::vector< real_type > const& s() const noexcept { return s_; } - anorm = std::max( anorm, ( std::fabs( w[i][i] ) + std::fabs( arr[i] ) ) ); + // the relative cutoff: rtol < 0 selects the default max(m, n)·ε (numpy 2 pinv, D-028) + [[nodiscard]] real_type cutoff( real_type rtol = real_type( -1 ) ) const noexcept + { + if ( rtol < real_type( 0 ) ) rtol = static_cast< real_type >( std::max( m_, n_ ) ) * std::numeric_limits< real_type >::epsilon(); + return s_.empty() ? real_type( 0 ) : rtol * s_[0]; } - - for ( i = n - 1;; --i ) + // the number of s_i > rtol·s_1 + [[nodiscard]] std::size_t rank( real_type rtol = real_type( -1 ) ) const noexcept { - if ( i < n - 1 ) - { - if ( g != zero ) - { - auto const tmp_uil = u[i][l]; - std::transform( u.row_begin( i ) + l, u.row_end( i ), v.col_begin( i ) + l, [g, tmp_uil]( value_type val ) { return val / ( tmp_uil * g ); } ); - - for ( std::uint_least64_t j = l; j < n; j++ ) - { - const auto tmp_s = std::inner_product( u.row_begin( i ) + l, u.row_end( i ), v.col_begin( j ) + l, value_type( 0 ) ); - std::transform( v.col_begin( j ) + l, v.col_end( j ), v.col_begin( i ) + l, v.col_begin( j ) + l, [tmp_s]( value_type v1, value_type v2 ) { return v1 + v2 * tmp_s; } ); - } - } - - std::fill( v.row_begin( i ) + l, v.row_end( i ), zero ); - std::fill( v.col_begin( i ) + l, v.col_end( i ), zero ); - } - - v[i][i] = one; - g = arr[i]; - l = i; - - if ( !i ) - break; + real_type const tol = cutoff( rtol ); + return static_cast< std::size_t >( std::count_if( s_.begin(), s_.end(), [tol]( real_type x ) { return x > tol; } ) ); } - - for ( i = std::min( m, n ) - 1;; --i ) + // A⁺ = V·diag(1/s_i)·Uᴴ (n×m) with every s_i ≤ rtol·s_1 zeroed; a status other than ok and an empty value + // when the factorization is not ok + [[nodiscard]] linalg_result< matrix_type > pinverse( real_type rtol = real_type( -1 ) ) const noexcept { - auto const tmp_l = i + 1; - auto const tmp_g = w[i][i]; - std::fill( u.row_begin( i ) + tmp_l, u.row_end( i ), zero ); - - if ( tmp_g != zero ) - { - for ( std::uint_least64_t j = tmp_l; j < n; j++ ) + if ( !ok() ) return { matrix_type{}, status_ }; + std::size_t const r = rank( rtol ); + matrix_type x( n_, m_ ); + std::fill( x.begin(), x.end(), T{ 0 } ); + for ( std::size_t i = 0; i != n_; ++i ) + for ( std::size_t c = 0; c != r; ++c ) { - auto const tmp_s = std::inner_product( u.col_begin( i ) + tmp_l, u.col_end( i ), u.col_begin( j ) + tmp_l, value_type( 0 ) ); - auto const tmp_f = tmp_s / ( u[i][i] * tmp_g ); - std::transform( u.col_begin( j ) + i, u.col_end( j ), u.col_begin( i ) + i, u.col_begin( j ) + i, [tmp_f]( value_type v1, value_type v2 ) { return v1 + tmp_f * v2; } ); + T const vi = v_[i][c] / s_[c]; + T* const xi = x.data() + i * m_; + for ( std::size_t j = 0; j != m_; ++j ) xi[j] += vi * conj_( u_[j][c] ); } + if ( !matrix_details::all_finite( x.data(), x.size() ) ) return { matrix_type{}, linalg_status::nonfinite }; + return { std::move( x ), linalg_status::ok }; + } - matrix_details::for_each( u.col_begin( i ) + i, u.col_end( i ), [tmp_g]( value_type & v ) { v /= tmp_g; } ); - } - else - std::fill( u.col_begin( i ) + i, u.col_end( i ), zero ); - - ++u[i][i]; - - if ( !i ) - break; + private: + static T conj_( T const& x ) noexcept + { + if constexpr ( matrix_private::is_std_complex_v< T > ) return std::conj( x ); + else return x; } - for ( std::uint_least64_t k = n - 1;; --k ) + // makes the k-th column of the column-major M×k array `q` (columns 0..k-1 orthonormal) a unit vector + // orthogonal to them: the first unit vector e_t whose residual after two Gram–Schmidt passes exceeds 1/2 + static void complete_( std::vector< T >& q, std::size_t M, std::size_t k ) noexcept { - for ( std::uint_least64_t its = 0; its < max_its; its++ ) + T* const x = q.data() + k * M; + for ( std::size_t t = 0; t != M; ++t ) { - bool flag = true; - std::uint_least64_t tmp_nm = 0; - - for ( l = k;; l-- ) - { - tmp_nm = l - 1; - - if ( std::fabs( arr[l] ) + anorm == anorm ) - { - flag = false; - break; - } - - if ( std::fabs( w[l - 1][l - 1] ) + anorm == anorm ) + std::fill( x, x + M, T{ 0 } ); + x[t] = T{ 1 }; + for ( int pass = 0; pass != 2; ++pass ) + for ( std::size_t c = 0; c != k; ++c ) { - break; + T const* const qc = q.data() + c * M; + T d{ 0 }; + for ( std::size_t i = 0; i != M; ++i ) d += conj_( qc[i] ) * x[i]; + for ( std::size_t i = 0; i != M; ++i ) x[i] -= d * qc[i]; } - - if ( l == 0 ) - break; + real_type nrm{ 0 }; + for ( std::size_t i = 0; i != M; ++i ) nrm += std::norm( x[i] ); + nrm = std::sqrt( nrm ); + if ( nrm > real_type( 0.5 ) ) + { + for ( std::size_t i = 0; i != M; ++i ) x[i] /= nrm; + return; } + } + } - if ( flag ) + void factor( matrix_type const& a, std::size_t max_sweeps ) noexcept + { + using R = real_type; + T const* const ad = a.data(); + if ( !matrix_details::all_finite( ad, m_ * n_ ) ) { status_ = linalg_status::nonfinite; return; } + bool const tall = m_ >= n_; + std::size_t const M = tall ? m_ : n_; // B is M×N, M ≥ N + std::size_t const N = tall ? n_ : m_; + R scale{ 0 }; + for ( std::size_t i = 0; i != m_ * n_; ++i ) scale = std::max( scale, R( std::abs( ad[i] ) ) ); + R const inv_scale = scale > R( 0 ) ? R( 1 ) / scale : R( 1 ); + // column j of B at b[j·M ..), column j of W at w[j·N ..); B·W stays the rotated B and W is unitary + std::vector< T > b( M * N ), w( N * N, T{ 0 } ); + for ( std::size_t i = 0; i != m_; ++i ) + for ( std::size_t j = 0; j != n_; ++j ) { - c = zero; - s = one; + T const x = ad[i * n_ + j] * inv_scale; + if ( tall ) b[j * M + i] = x; + else b[i * M + j] = conj_( x ); + } + for ( std::size_t j = 0; j != N; ++j ) w[j * N + j] = T{ 1 }; - for ( i = l - 1; i < k + 1; ++i ) + R const eps = std::numeric_limits< R >::epsilon(); + bool converged = N < 2; + while ( !converged && sweeps_ < max_sweeps ) + { + ++sweeps_; + bool rotated = false; + for ( std::size_t p = 0; p + 1 < N; ++p ) + for ( std::size_t q = p + 1; q != N; ++q ) { - f = s * arr[i]; - arr[i] = c * arr[i]; - - if ( std::fabs( f ) + anorm == anorm ) + T* const bp = b.data() + p * M; + T* const bq = b.data() + q * M; + R alpha{ 0 }, beta{ 0 }; + T gamma{ 0 }; + for ( std::size_t i = 0; i != M; ++i ) { - break; + alpha += std::norm( bp[i] ); + beta += std::norm( bq[i] ); + gamma += conj_( bp[i] ) * bq[i]; } - - g = w[i][i]; - h = std::hypot( f, g ); - w[i][i] = h; - h = one / h; - c = g * h; - s = -f * h; - - for ( std::uint_least64_t j = 0; j < m; ++j ) + R const g = std::abs( gamma ); + if ( g == R( 0 ) || g <= eps * std::sqrt( alpha ) * std::sqrt( beta ) ) continue; + rotated = true; + // with e = γ/|γ| the pair (b_p, ē·b_q) has the real inner product |γ|: rotate it by (c, s) + R const zeta = ( beta - alpha ) / ( R( 2 ) * g ); + R const t = ( zeta >= R( 0 ) ? R( 1 ) : R( -1 ) ) / ( std::abs( zeta ) + std::hypot( R( 1 ), zeta ) ); + R const c = R( 1 ) / std::sqrt( R( 1 ) + t * t ); + R const s = c * t; + T const ebar = conj_( gamma / g ); + auto const rotate = [c, s, ebar]( T* x, T* y, std::size_t len ) noexcept { - value_type y = u[j][tmp_nm]; - value_type z = u[j][i]; - u[j][tmp_nm] = y * c + z * s; - u[j][i] = z * c - y * s; - } + for ( std::size_t i = 0; i != len; ++i ) + { + T const xi = x[i]; + T const yi = ebar * y[i]; + x[i] = c * xi - s * yi; + y[i] = s * xi + c * yi; + } + }; + rotate( bp, bq, M ); + rotate( w.data() + p * N, w.data() + q * N, N ); } - } + converged = !rotated; + } + status_ = converged ? linalg_status::ok : linalg_status::not_converged; - value_type z = w[k][i]; + // singular values, descending order, then normalized left vectors of B + std::vector< R > norms( N ); + for ( std::size_t j = 0; j != N; ++j ) + { + R acc{ 0 }; + for ( std::size_t i = 0; i != M; ++i ) acc += std::norm( b[j * M + i] ); + norms[j] = std::sqrt( acc ); + } + std::vector< std::size_t > idx( N ); + for ( std::size_t j = 0; j != N; ++j ) idx[j] = j; + std::stable_sort( idx.begin(), idx.end(), [&norms]( std::size_t x, std::size_t y ) { return norms[x] > norms[y]; } ); + std::vector< T > ub( M * N ), vb( N * N ); + s_.resize( N ); + for ( std::size_t r = 0; r != N; ++r ) + { + std::size_t const j = idx[r]; + s_[r] = norms[j] * scale; + std::copy( w.begin() + j * N, w.begin() + j * N + N, vb.begin() + r * N ); + if ( norms[j] > R( 0 ) ) + for ( std::size_t i = 0; i != M; ++i ) ub[r * M + i] = b[j * M + i] / norms[j]; + } + for ( std::size_t r = 0; r != N; ++r ) // exact zeros sort last + if ( norms[idx[r]] == R( 0 ) ) complete_( ub, M, r ); - if ( l == k ) - { - if ( z < zero ) - { - w[k][k] = -z; - matrix_details::for_each( v.col_begin( k ), v.col_end( k ), []( value_type & v ) { v = -v; } ); - } - - break; - } - - if ( ( its + 1 ) == max_its ) - return 1; - - value_type x = w[l][l]; - value_type y = w[k - 1][k - 1]; - g = arr[k - 1]; - h = arr[k]; - f = ( ( y - z ) * ( y + z ) + ( g - h ) * ( g + h ) ) / ( 2.0 * h * y ); - g = std::hypot( f, one ); - g = ( f >= zero ) ? g : -g; - f = ( ( x - z ) * ( x + z ) + h * ( ( y / ( f + g ) ) ) - h ) / x; - c = s = one; - - for ( std::uint_least64_t j = l; j <= k - 1; j++ ) - { - g = arr[j + 1]; - y = w[j + 1][j + 1]; - h = s * g; - g = c * g; - z = std::hypot( f, h ); - arr[j] = z; - c = f / z; - s = h / z; - f = x * c + g * s; - g = g * c - x * s; - h = y * s; - y *= c; - matrix_details::for_each( v.col_begin( j ), v.col_end( j ), v.col_begin( j + 1 ), [c, s]( value_type & v1, value_type & v2 ) - { - const auto vv1( v1 ); - const auto vv2( v2 ); - v1 = vv1 * c + vv2 * s; - v2 = vv2 * c - vv1 * s; - } ); - w[j][j] = std::hypot( f, h ); + // A = B (tall): U = U_B, V = W; A = Bᴴ (wide): U = W, V = U_B + auto const fill = []( matrix_type& out, std::vector< T > const& src, std::size_t rows, std::size_t k ) + { + out.resize( rows, k ); + for ( std::size_t i = 0; i != rows; ++i ) + for ( std::size_t c = 0; c != k; ++c ) out[i][c] = src[c * rows + i]; + }; + fill( u_, tall ? ub : vb, m_, N ); + fill( v_, tall ? vb : ub, n_, N ); + } - if ( value_type( 0 ) != w[j][j] ) - { - c = f / w[j][j]; - s = h / w[j][j]; - } + matrix_type u_; + matrix_type v_; + std::vector< real_type > s_; + linalg_status status_ = linalg_status::ok; + std::size_t sweeps_ = 0; + std::size_t m_ = 0; + std::size_t n_ = 0; + }; - matrix_details::for_each( u.col_begin( j ), u.col_end( j ), u.col_begin( j + 1 ), [c, s]( value_type & v1, value_type & v2 ) - { - const auto vv1( v1 ); - const auto vv2( v2 ); - v1 = vv1 * c + vv2 * s; - v2 = vv2 * c - vv1 * s; - } ); - f = c * g + s * y; - x = c * y - s * g; - } + template < typename T, Allocator A > requires linalg_element< T > + [[nodiscard]] svd_factorization< T, A > svd_factor( matrix< T, A > const& a, std::size_t max_sweeps = 64 ) noexcept + { + return svd_factorization< T, A >( a, max_sweeps ); + } - arr[l] = zero; - arr[k] = f; - w[k][k] = x; - i = k; - } + // The one pseudoinverse (S7-R3, D-026, D-028): out = A⁺ (n×m) with s_i ≤ rtol·s_1 zeroed, rtol < 0 selecting + // max(m, n)·ε; out is assigned only when the status is ok. + template < typename T, Allocator A > requires linalg_element< T > + [[nodiscard]] linalg_status pinverse( matrix< T, A > const& a, matrix< T, A >& out, matrix_details::linalg_real_t< T > rtol = -1 ) noexcept + { + auto r = svd_factor( a ).pinverse( rtol ); + if ( r.ok() ) out = std::move( r.value ); + return r.status; + } - if ( !k ) - break; - } + // value-returning forms: an empty 0×0 matrix when the SVD is not ok (D-026) + template < typename T, Allocator A > requires linalg_element< T > + [[nodiscard]] matrix< T, A > pinverse( matrix< T, A > const& a, matrix_details::linalg_real_t< T > rtol = -1 ) noexcept + { + matrix< T, A > out; + (void)pinverse( a, out, rtol ); // the status is reported as the empty result + return out; + } + template < typename T, Allocator A > requires linalg_element< T > + [[nodiscard]] matrix< T, A > pinv( matrix< T, A > const& a, matrix_details::linalg_real_t< T > rtol = -1 ) noexcept + { + return pinverse( a, rtol ); + } + template < typename T, Allocator A > requires linalg_element< T > + [[nodiscard]] matrix< T, A > svd_inverse( matrix< T, A > const& a ) noexcept + { + return pinverse( a ); + } + // Legacy SVD: max_its is the sweep limit. On success returns 0 with u m×k, w k×k diagonal (the singular values, + // descending) and v n×k, k = min(m, n), so A = u·w·vᴴ; otherwise returns 1 with u, w and v unchanged. + template < typename T, typename A_ = std::allocator< T > > requires linalg_element< T > + [[nodiscard]] std::uint_least64_t + singular_value_decomposition( matrix const& A, + matrix& u, + matrix& w, + matrix& v, + std::uint_least64_t const max_its = 64 ) noexcept + { + auto const f = svd_factor( A, static_cast< std::size_t >( max_its ) ); + if ( !f.ok() ) + return 1; + std::size_t const k = f.s().size(); + matrix d( k, k ); + std::fill( d.begin(), d.end(), T{ 0 } ); + for ( std::size_t i = 0; i != k; ++i ) d[i][i] = T( f.s()[i] ); + u = f.u(); + v = f.v(); + w = std::move( d ); return 0; } - template< typename T, Allocator A> - std::optional< std::tuple, matrix, matrix> > + // (u, w, v) as above, or nullopt when the SVD is not ok + template< typename T, Allocator A> requires linalg_element< T > + [[nodiscard]] std::optional< std::tuple, matrix, matrix> > singular_value_decomposition( matrix const& a ) noexcept { - auto const [row, col] = a.shape(); - auto const max_iteration = std::max( std::uint_least64_t{100}, std::max( row, col ) ); - if ( matrix u, w, v; singular_value_decomposition( a, u, w, v, max_iteration ) ) // fail + if ( matrix u, w, v; singular_value_decomposition( a, u, w, v ) ) // fail return {}; else // success - return std::forward_as_tuple( u, w, v ); + return std::make_tuple( std::move( u ), std::move( w ), std::move( v ) ); } - template< typename T, Allocator A> - auto svd( matrix const& a ) noexcept + template< typename T, Allocator A> requires linalg_element< T > + [[nodiscard]] auto svd( matrix const& a ) noexcept { return singular_value_decomposition( a ); } - template < typename T, Allocator A> - matrix const svd_inverse( matrix const& a ) + // S6-R3 (F17, D-024): random matrices from a caller-owned std::uniform_random_bit_generator. Elements are + // drawn in row-major order from the engine alone (no global state). Floating parts lie in the open interval + // (0, 1): a draw equal to 0 or 1 is redrawn. Integral elements come from std::uniform_int_distribution over + // [numeric_limits::min(), max()]; types the distribution does not accept (bool and the character types) + // are drawn as int or unsigned (long long or unsigned long long when wider) and narrowed. + namespace matrix_details { - matrix u; - matrix w; - matrix v; - singular_value_decomposition( a, u, v, w ); - matrix_details::for_each( v.begin(), v.end(), []( auto & val ) { if ( std::abs( val ) > 1.0e-10 ) val = 1.0 / val; }); - return w * v.transpose() * u.transpose(); + template< typename T > + inline constexpr bool random_dependent_false_v = false; + + template< typename T > + inline constexpr bool uniform_int_type_v = + std::is_same_v< T, short > || std::is_same_v< T, int > || std::is_same_v< T, long > || std::is_same_v< T, long long > || + std::is_same_v< T, unsigned short > || std::is_same_v< T, unsigned int > || std::is_same_v< T, unsigned long > || + std::is_same_v< T, unsigned long long >; + + template< typename T > + using uniform_draw_t = std::conditional_t< uniform_int_type_v< T >, T, + std::conditional_t< ( sizeof( T ) <= sizeof( int ) ), + std::conditional_t< std::is_signed_v< T >, int, unsigned >, + std::conditional_t< std::is_signed_v< T >, long long, unsigned long long > > >; + + template< std::floating_point X, typename G > + X real_open_unit( G& g ) noexcept + { + std::uniform_real_distribution< X > dist{ X{ 0 }, X{ 1 } }; + for ( ;; ) + { + X const x = dist( g ); + if ( X{ 0 } < x && x < X{ 1 } ) return x; + } + } + + // a seed for the legacy overloads: the nonzero seed as given, else a steady_clock count mixed with an + // object address (no hardware random device: it may throw) + inline std::uint_least64_t legacy_seed( unsigned int seed, void const* address ) noexcept + { + if ( 0 != seed ) return seed; + auto const ticks = static_cast< std::uint_least64_t >( std::chrono::steady_clock::now().time_since_epoch().count() ); + auto const where = static_cast< std::uint_least64_t >( reinterpret_cast< std::uintptr_t >( address ) ); + return ticks ^ ( where * 0x9E3779B97F4A7C15ULL ); + } } - template < typename Matrix > - Matrix const pinverse( const Matrix& m ) + + template < typename T = double, typename A = std::allocator< T >, typename G > + requires std::uniform_random_bit_generator< std::remove_cvref_t< G > > + [[nodiscard]] matrix< T, A > random_impl( const std::uint_least64_t r, const std::uint_least64_t c, G& g ) noexcept + { + matrix< T, A > ans{ r, c }; + if constexpr ( std::is_floating_point_v< T > ) + { + for ( auto& x : ans ) x = matrix_details::real_open_unit< T >( g ); + } + else if constexpr ( matrix_private::is_std_complex_v< T > ) + { + using X = typename T::value_type; + for ( auto& x : ans ) + { + X const re = matrix_details::real_open_unit< X >( g ); + X const im = matrix_details::real_open_unit< X >( g ); + x = T{ re, im }; + } + } + else if constexpr ( std::is_integral_v< T > ) + { + using D = matrix_details::uniform_draw_t< T >; + std::uniform_int_distribution< D > dist{ static_cast< D >( std::numeric_limits< T >::min() ), static_cast< D >( std::numeric_limits< T >::max() ) }; + for ( auto& x : ans ) x = static_cast< T >( dist( g ) ); + } + else + { + static_assert( matrix_details::random_dependent_false_v< T >, "feng::random/rand: the element type must be a floating-point, std::complex or integral type" ); + } + return ans; + } + + // engine overloads + template < typename T = double, typename A = std::allocator< T >, typename G > + requires std::uniform_random_bit_generator< std::remove_cvref_t< G > > + [[nodiscard]] matrix< T, A > random( std::integral auto r, std::integral auto c, G&& g ) noexcept { - Matrix u, w, v; - singular_value_decomposition( m, u, w, v ); - return v * w * u.transpose(); + return random_impl< T, A >( r, c, g ); } - template < typename Matrix > - Matrix const pinv( const Matrix& m ) + template < typename T = double, typename A = std::allocator< T >, typename G > + requires std::uniform_random_bit_generator< std::remove_cvref_t< G > > + [[nodiscard]] matrix< T, A > random( std::integral auto n, G&& g ) noexcept + { + return random_impl< T, A >( n, n, g ); + } + template < typename T = double, typename A = std::allocator< T >, typename G > + requires std::uniform_random_bit_generator< std::remove_cvref_t< G > > + [[nodiscard]] matrix< T, A > rand( const std::uint_least64_t r, const std::uint_least64_t c, G&& g ) noexcept + { + return random_impl< T, A >( r, c, g ); + } + template < typename T = double, typename A = std::allocator< T >, typename G > + requires std::uniform_random_bit_generator< std::remove_cvref_t< G > > + [[nodiscard]] matrix< T, A > rand( const std::uint_least64_t n, G&& g ) noexcept + { + return random_impl< T, A >( n, n, g ); + } + template < typename T, Allocator A, typename G > + requires std::uniform_random_bit_generator< std::remove_cvref_t< G > > + [[nodiscard]] matrix< T, A > rand_like( matrix< T, A > const& mat, G&& g ) noexcept { - return pinverse( m ); + auto const [row, col] = mat.shape(); + return random_impl< T, A >( row, col, g ); + } + template < typename T, Allocator A, typename G > + requires std::uniform_random_bit_generator< std::remove_cvref_t< G > > + [[nodiscard]] matrix< T, A > random_like( matrix< T, A > const& mat, G&& g ) noexcept + { + auto const [row, col] = mat.shape(); + return random_impl< T, A >( row, col, g ); } - //generating a matrix uniformly in (0, 1) + // legacy overloads: a local std::mt19937_64 seeded per matrix_details::legacy_seed template < typename T = double, typename A = std::allocator< T > > - matrix< T, A > const rand( const std::uint_least64_t r, const std::uint_least64_t c, unsigned int seed = 0 ) noexcept + [[nodiscard]] matrix< T, A > rand( const std::uint_least64_t r, const std::uint_least64_t c, unsigned int seed = 0 ) noexcept { - matrix< T, A > ans{ r, c }; - if ( 0 == seed ) - std::srand( static_cast< unsigned int >( static_cast< std::uint_least64_t >( std::time( nullptr ) ) + reinterpret_cast< std::uint_least64_t >( &ans ) ) ); - else - std::srand( seed ); - - auto const& generator = []() noexcept - { - return ( static_cast( std::rand() ) + 1 ) / ( static_cast( RAND_MAX ) + 2 ); // make sure in open bounds range (0, 1) - }; - std::generate( ans.begin(), ans.end(), generator ); - return ans; + std::mt19937_64 engine{ 0 }; + engine.seed( matrix_details::legacy_seed( seed, &engine ) ); + return random_impl< T, A >( r, c, engine ); } template < typename T = double, typename A = std::allocator< T > > - matrix< T, A > const rand( const std::uint_least64_t n ) + [[nodiscard]] matrix< T, A > rand( const std::uint_least64_t n ) noexcept { return rand< T, A >( n, n ); } template < typename T = double, typename A = std::allocator< T > > - matrix< T, A > const random( std::integral auto r, std::integral auto c ) + [[nodiscard]] matrix< T, A > random( std::integral auto r, std::integral auto c ) noexcept { return rand< T, A >( r, c ); } template < typename T = double, typename A = std::allocator< T > > - matrix< T, A > const random( const std::integral auto n ) + [[nodiscard]] matrix< T, A > random( const std::integral auto n ) noexcept { return rand< T, A >( n ); } template < typename T, Allocator A> - matrix< T, A > const rand_like( matrix const& mat ) noexcept + [[nodiscard]] matrix< T, A > rand_like( matrix const& mat ) noexcept { auto const[row, col] = mat.shape(); return random( row, col ); } template < typename T, Allocator A> - matrix< T, A > const random_like( matrix const& mat ) noexcept + [[nodiscard]] matrix< T, A > random_like( matrix const& mat ) noexcept { return rand_like(mat); } - template < typename T, Allocator A> //pytorch style - matrix< T, A > const randn_like( matrix const& mat ) noexcept + template < typename T, Allocator A> //pytorch style; uniform in (0, 1), an alias of rand_like + [[nodiscard]] matrix< T, A > randn_like( matrix const& mat ) noexcept { return rand_like(mat); } template < typename T, Allocator A> - const matrix< T, A > - repmat( const matrix< T, A >& m, const std::uint_least64_t r, const std::uint_least64_t c ) + [[nodiscard]] matrix< T, A > + repmat( const matrix< T, A >& m, const std::uint_least64_t r, const std::uint_least64_t c ) noexcept { better_assert( r ); better_assert( c ); @@ -5304,8 +6951,8 @@ namespace feng } template < typename Itor1, typename Itor2, - typename A = std::allocator< typename std::remove_const< typename std::remove_reference< typename std::iterator_traits< Itor1 >::value_type >::result_type >::result_type >> - matrix< typename std::iterator_traits< Itor1 >::value_type, A > const toeplitz( Itor1 i1_, Itor1 _i1, Itor2 i2_, Itor2 _i2 ) + typename A = std::allocator< std::remove_cvref_t< typename std::iterator_traits< Itor1 >::value_type > >> + [[nodiscard]] matrix< typename std::iterator_traits< Itor1 >::value_type, A > toeplitz( Itor1 i1_, Itor1 _i1, Itor2 i2_, Itor2 _i2 ) noexcept { std::uint_least64_t r = std::distance( i1_, _i1 ); std::uint_least64_t c = std::distance( i2_, _i2 ); @@ -5321,23 +6968,23 @@ namespace feng } template < typename Itor, - typename A = std::allocator< typename std::remove_const< typename std::remove_reference< typename std::iterator_traits< Itor >::value_type >::result_type >::result_type >> - matrix< typename std::iterator_traits< Itor >::value_type, A > const toeplitz( Itor i_, Itor _i ) + typename A = std::allocator< std::remove_cvref_t< typename std::iterator_traits< Itor >::value_type > >> + [[nodiscard]] matrix< typename std::iterator_traits< Itor >::value_type, A > toeplitz( Itor i_, Itor _i ) noexcept { return toeplitz( i_, _i, i_, _i ); } template < typename Matrix > - typename Matrix::value_type tr( const Matrix& m ) + [[nodiscard]] typename Matrix::value_type tr( const Matrix& m ) noexcept { return m.tr(); } template < typename Matrix > - Matrix const transpose( const Matrix& m ) + [[nodiscard]] Matrix transpose( const Matrix& m ) noexcept { return m.transpose(); } template < typename T, Allocator A> - matrix< T, A > const tril( const matrix< T, A >& m ) + [[nodiscard]] matrix< T, A > tril( const matrix< T, A >& m ) noexcept { matrix< T, A > ans{ m.row(), m.col() }; @@ -5347,7 +6994,7 @@ namespace feng return ans; } template < typename T, Allocator A> - matrix< T, A > const triu( const matrix< T, A >& m ) + [[nodiscard]] matrix< T, A > triu( const matrix< T, A >& m ) noexcept { matrix< T, A > ans{ m.row(), m.col() }; @@ -5356,155 +7003,79 @@ namespace feng return ans; } - template < typename T, Allocator A> - const matrix< std::complex< T >, A> - operator+( const matrix< std::complex< T >, A>& lhs, const T& rhs ) - { - matrix< std::complex< T >, A> ans( lhs ); - std::transform( ans.begin(), ans.end(), ans.begin(), [rhs]( std::complex< T > const & x ) - { - return rhs + x; - } ); - return ans; - } - template < typename T, Allocator A> - const matrix< std::complex< T >, A> - operator+( const T& lhs, const matrix< std::complex< T >, A>& rhs ) - { - return rhs + lhs; - } - template < typename T, Allocator A> - const matrix< std::complex< T >, A> - operator-( const matrix< std::complex< T >, A>& lhs, const T& rhs ) - { - matrix< std::complex< T >, A> ans( lhs ); - std::transform( ans.begin(), ans.end(), ans.begin(), [rhs]( std::complex< T > const & x ) - { - return x - rhs; - } ); - return ans; - } - template < typename T, Allocator A> - const matrix< std::complex< T >, A> - operator-( const T& lhs, const matrix< std::complex< T >, A>& rhs ) - { - matrix< std::complex< T >, A> ans( rhs ); - std::transform( ans.begin(), ans.end(), ans.begin(), [lhs]( std::complex< T > const & x ) - { - return -lhs + x; - } ); - return ans; - } - template < typename T, Allocator A> - const matrix< std::complex< T >, A> - operator*( const matrix< std::complex< T >, A>& lhs, const T& rhs ) - { - matrix< std::complex< T >, A> ans( lhs ); - std::transform( ans.begin(), ans.end(), ans.begin(), [rhs]( std::complex< T > const & x ) - { - return x * rhs; - } ); - return ans; - } - template < typename T, Allocator A> - const matrix< std::complex< T >, A> - operator*( const T& lhs, const matrix< std::complex< T >, A>& rhs ) - { - return rhs * lhs; - } - template < typename T, Allocator A> - const matrix< std::complex< T >, A> - operator/( const matrix< std::complex< T >, A>& lhs, const T& rhs ) + // S6-R5 (D-023): one template per operator and side. The result is matrix< scalar_result_t< T, S >, A rebound >: + // T (and lhs's allocator, as a copy selects it) when the scalar's kind is not higher than T's, the scalar + // converted to T, or to T's value type for complex T, first; otherwise the matrix is promoted to the result type. + // Signed overflow and integer division by zero are preconditions; unsigned arithmetic wraps. + namespace matrix_details { - matrix< std::complex< T >, A> ans( lhs ); - std::transform( ans.begin(), ans.end(), ans.begin(), [rhs]( std::complex< T > const & x ) + template< typename T, Allocator A, typename S, typename Op > + promoted_matrix_t< scalar_result_t< T, S >, T, A > scalar_apply( matrix< T, A > const& m, S const& s, Op op ) noexcept { - return x / rhs; - } ); - return ans; - } - template < typename T, Allocator A> - const matrix< std::complex< T >, A> - operator/( const T& lhs, const matrix< std::complex< T >, A>& rhs ) - { - return lhs * rhs.inverse(); + using R = scalar_result_t< T, S >; + auto ans = promote< R >( m ); + auto const v = scalar_as< R >( s ); + std::transform( ans.begin(), ans.end(), ans.begin(), [&v, &op]( R const& x ) noexcept { return static_cast< R >( op( x, v ) ); } ); + return ans; + } } - template < typename T, Allocator A> - const matrix< T, A > - operator+( const matrix< T, A >& lhs, const T& rhs ) + + template < typename T, Allocator A, typename S > requires matrix_details::scalar_operand_for< S, T > + matrix_details::promoted_matrix_t< matrix_details::scalar_result_t< T, S >, T, A > + operator+( const matrix< T, A >& lhs, const S& rhs ) noexcept { - matrix< T, A > ans( lhs ); - std::transform( ans.begin(), ans.end(), ans.begin(), [rhs]( T x ) - { - return rhs + x; - } ); - return ans; + return matrix_details::scalar_apply( lhs, rhs, []( auto const& x, auto const& y ) noexcept { return x + y; } ); } - template < typename T, Allocator A> - const matrix< T, A > - operator+( const T& lhs, const matrix< T, A >& rhs ) + template < typename T, Allocator A, typename S > requires matrix_details::scalar_operand_for< S, T > + matrix_details::promoted_matrix_t< matrix_details::scalar_result_t< T, S >, T, A > + operator+( const S& lhs, const matrix< T, A >& rhs ) noexcept { - return rhs + lhs; + return matrix_details::scalar_apply( rhs, lhs, []( auto const& x, auto const& y ) noexcept { return y + x; } ); } - template < typename T, Allocator A> - const matrix< T, A > - operator-( const matrix< T, A >& lhs, const T& rhs ) + template < typename T, Allocator A, typename S > requires matrix_details::scalar_operand_for< S, T > + matrix_details::promoted_matrix_t< matrix_details::scalar_result_t< T, S >, T, A > + operator-( const matrix< T, A >& lhs, const S& rhs ) noexcept { - matrix< T, A > ans( lhs ); - std::transform( ans.begin(), ans.end(), ans.begin(), [rhs]( T x ) - { - return x - rhs; - } ); - return ans; + return matrix_details::scalar_apply( lhs, rhs, []( auto const& x, auto const& y ) noexcept { return x - y; } ); } - template < typename T, Allocator A> - const matrix< T, A > - operator-( const T& lhs, const matrix< T, A >& rhs ) + template < typename T, Allocator A, typename S > requires matrix_details::scalar_operand_for< S, T > + matrix_details::promoted_matrix_t< matrix_details::scalar_result_t< T, S >, T, A > + operator-( const S& lhs, const matrix< T, A >& rhs ) noexcept { - matrix< T, A > ans( rhs ); - std::transform( ans.begin(), ans.end(), ans.begin(), [lhs]( T x ) - { - return -lhs + x; - } ); - return ans; + return matrix_details::scalar_apply( rhs, lhs, []( auto const& x, auto const& y ) noexcept { return y - x; } ); } - template < typename T, Allocator A> - const matrix< T, A > - operator*( const matrix< T, A >& lhs, const T& rhs ) + template < typename T, Allocator A, typename S > requires matrix_details::scalar_operand_for< S, T > + matrix_details::promoted_matrix_t< matrix_details::scalar_result_t< T, S >, T, A > + operator*( const matrix< T, A >& lhs, const S& rhs ) noexcept { - matrix< T, A > ans( lhs ); - std::transform( ans.begin(), ans.end(), ans.begin(), [rhs]( T x ) - { - return x * rhs; - } ); - return ans; + return matrix_details::scalar_apply( lhs, rhs, []( auto const& x, auto const& y ) noexcept { return x * y; } ); } - template < typename T, Allocator A> - const matrix< T, A > - operator*( const T& lhs, const matrix< T, A >& rhs ) + template < typename T, Allocator A, typename S > requires matrix_details::scalar_operand_for< S, T > + matrix_details::promoted_matrix_t< matrix_details::scalar_result_t< T, S >, T, A > + operator*( const S& lhs, const matrix< T, A >& rhs ) noexcept { - return rhs * lhs; + return matrix_details::scalar_apply( rhs, lhs, []( auto const& x, auto const& y ) noexcept { return y * x; } ); } - template < typename T, Allocator A> - const matrix< T, A > - operator/( const matrix< T, A >& lhs, const T& rhs ) + template < typename T, Allocator A, typename S > requires matrix_details::scalar_operand_for< S, T > + matrix_details::promoted_matrix_t< matrix_details::scalar_result_t< T, S >, T, A > + operator/( const matrix< T, A >& lhs, const S& rhs ) noexcept { - matrix< T, A > ans( lhs ); - std::transform( ans.begin(), ans.end(), ans.begin(), [rhs]( T x ) - { - return x / rhs; - } ); - return ans; + return matrix_details::scalar_apply( lhs, rhs, []( auto const& x, auto const& y ) noexcept { return x / y; } ); } - template < typename T, Allocator A> - const matrix< T, A > - operator/( const T& lhs, const matrix< T, A >& rhs ) + // s / m keeps its meaning, s times m.inverse(); only its result type follows D-023. + template < typename T, Allocator A, typename S > requires matrix_details::scalar_operand_for< S, T > + matrix_details::promoted_matrix_t< matrix_details::scalar_result_t< T, S >, T, A > + operator/( const S& lhs, const matrix< T, A >& rhs ) noexcept { - return lhs * rhs.inverse(); + using R = matrix_details::scalar_result_t< T, S >; + if constexpr ( std::same_as< R, T > ) + return rhs.inverse() * matrix_details::scalar_as< R >( lhs ); + else + return matrix_details::promote< R >( rhs ).inverse() * matrix_details::scalar_as< R >( lhs ); } template < typename T, Allocator A> - const matrix< T, A > - operator||( const matrix< T, A >& lhs, const T& rhs ) + matrix< T, A > + operator||( const matrix< T, A >& lhs, const T& rhs ) noexcept { matrix< T, A > ans( lhs.row(), lhs.col() + 1 ); @@ -5515,8 +7086,8 @@ namespace feng return ans; } template < typename T, Allocator A> - const matrix< T, A > - operator||( const T& lhs, const matrix< T, A >& rhs ) + matrix< T, A > + operator||( const T& lhs, const matrix< T, A >& rhs ) noexcept { matrix< T, A > ans( rhs.row(), rhs.col() + 1 ); @@ -5527,8 +7098,8 @@ namespace feng return ans; } template < typename T, Allocator A> - const matrix< T, A > - operator&&( const matrix< T, A >& lhs, const T& rhs ) + matrix< T, A > + operator&&( const matrix< T, A >& lhs, const T& rhs ) noexcept { matrix< T, A > ans( lhs.row() + 1, lhs.col() ); @@ -5539,8 +7110,8 @@ namespace feng return ans; } template < typename T, Allocator A> - const matrix< T, A > - operator&&( const T& lhs, const matrix< T, A >& rhs ) + matrix< T, A > + operator&&( const T& lhs, const matrix< T, A >& rhs ) noexcept { matrix< T, A > ans( rhs.row() + 1, rhs.col() ); @@ -5551,8 +7122,8 @@ namespace feng return ans; } template < typename T, Allocator A> - const matrix< T, A > - operator^( const matrix< T, A >& lhs, std::uint_least64_t n ) + matrix< T, A > + operator^( const matrix< T, A >& lhs, std::uint_least64_t n ) noexcept { better_assert( lhs.row() == lhs.col() ); auto const r = lhs.row(); @@ -5564,15 +7135,15 @@ namespace feng return lhs; if ( n & 1 ) - return lhs ^ ( n - 1 ) * lhs; + return ( lhs ^ ( n - 1 ) ) * lhs; auto const& lhs_2 = lhs ^ ( n >> 1 ); return lhs_2 * lhs_2; } template < typename T1, Allocator A1, typename T2, Allocator A2, typename T3, Allocator A3 > - int backward_substitution( const matrix< T1, A1 >& A, + [[nodiscard]] int backward_substitution( const matrix< T1, A1 >& A, matrix< T2, A2 >& x, - const matrix< T3, A3 >& b ) + const matrix< T3, A3 >& b ) noexcept { typedef matrix< T1, A1 > matrix_type; typedef typename matrix_type::value_type value_type; @@ -5596,177 +7167,316 @@ namespace feng return 0; } + namespace iterative_private + { + // Shared set-up of cgs and bicgstab (S7-R5, D-029): the starting iterate is x when x is n×1 and nonzero, + // else b (the legacy rule); the vectors are plain arrays of A's element type. + template < typename T > + struct krylov + { + std::size_t n; + std::vector< T > a, b, x; + + template < typename M1, typename M2, typename M3 > + krylov( M1 const& A, M2 const& x0, M3 const& b0 ) : n( A.row() ), a( n * n ), b( n ), x( n ) + { + for ( std::size_t i = 0; i != n; ++i ) + { + for ( std::size_t j = 0; j != n; ++j ) a[i * n + j] = static_cast< T >( A[i][j] ); + b[i] = static_cast< T >( b0[i][0] ); + } + bool const use_x = x0.row() == n && x0.col() == 1 && std::any_of( x0.begin(), x0.end(), []( auto v ) { return v != decltype( v )( 0 ); } ); + for ( std::size_t i = 0; i != n; ++i ) x[i] = use_x ? static_cast< T >( x0[i][0] ) : b[i]; + } + bool finite_input() const noexcept + { + return matrix_details::all_finite( a.data(), a.size() ) && matrix_details::all_finite( b.data(), b.size() ); + } + void mul( std::vector< T > const& v, std::vector< T >& out ) const noexcept // out = A·v + { + for ( std::size_t i = 0; i != n; ++i ) + { + T s{ 0 }; + for ( std::size_t j = 0; j != n; ++j ) s += a[i * n + j] * v[j]; + out[i] = s; + } + } + static T dot( std::vector< T > const& u, std::vector< T > const& v ) noexcept + { + T s{ 0 }; + for ( std::size_t i = 0; i != u.size(); ++i ) s += u[i] * v[i]; + return s; + } + static T norm( std::vector< T > const& v ) noexcept + { + T s{ 0 }; + for ( auto const e : v ) s += e * e; + return std::sqrt( s ); + } + T residual( std::vector< T > const& xv, std::vector< T >& tmp ) const noexcept // ‖b − A·xv‖₂ + { + mul( xv, tmp ); + for ( std::size_t i = 0; i != n; ++i ) tmp[i] = b[i] - tmp[i]; + return norm( tmp ); + } + static bool usable( T v ) noexcept { return v != T( 0 ) && std::isfinite( v ); } + template < typename M > + void store( M& out, std::vector< T > const& v ) const noexcept + { + if ( out.row() != n || out.col() != 1 ) out.resize( n, 1 ); + for ( std::size_t i = 0; i != n; ++i ) out[i][0] = static_cast< typename M::value_type >( v[i] ); + } + }; + }//namespace iterative_private + + // S7-R5 (D-029): BiCGSTAB for A·x = b. Returns 0 when ‖b − A·x‖₂ ≤ eps·‖b‖₂ (a zero b gives x = 0), and 1 for a + // nonfinite A or b (x unchanged), a breakdown (a zero or nonfinite denominator) or max_loops iterations without + // convergence; x then holds the last finite iterate. template < typename T1, Allocator A1, typename T2, Allocator A2, typename T3, Allocator A3 > - int biconjugate_gradient_stabilized_method( const matrix< T1, A1 >& A, + [[nodiscard]] int biconjugate_gradient_stabilized_method( const matrix< T1, A1 >& A, matrix< T2, A2 >& x, const matrix< T3, A3 >& b, const std::uint_least64_t max_loops = 100, - const T1 eps = 1.0e-10 ) + const T1 eps = 1.0e-10 ) noexcept { typedef T1 value_type; - better_assert( A.row() == A.col() ); - better_assert( A.row() == b.row() ); - better_assert( b.col() == 1 ); - auto const n = A.row(); - - if ( ( n != x.row() ) || ( 1 != x.col() ) || ( 0 == std::count_if( x.begin(), x.end(), []( T2 const v ) - { - return v != T2( 0 ); - } ) ) ) - x = b; - auto r = b - A * x; - auto const r_ = r; - auto p = r; - auto s = p; - auto ap = p; - auto as = p; - auto new_r = r; - auto rem = r; - auto const EPS = eps * n; - value_type const zero = value_type( 0 ); - - if ( dot( r, r ) < EPS ) + better_assert( A.row() == A.col(), "bicgstab: expecting a square A, but got ", A.row(), "x", A.col() ); + better_assert( A.row() == b.row() && b.col() == 1, "bicgstab: expecting a column b with ", A.row(), " rows, but got ", b.row(), "x", b.col() ); + iterative_private::krylov< value_type > k( A, x, b ); + if ( !k.finite_input() ) return 1; + std::size_t const n = k.n; + value_type const bnorm = k.norm( k.b ); + if ( bnorm == value_type( 0 ) ) + { + std::fill( k.x.begin(), k.x.end(), value_type( 0 ) ); + k.store( x, k.x ); return 0; - + } + value_type const tol = eps * bnorm; + std::vector< value_type > r( n ), tmp( n ), p( n, 0 ), v( n, 0 ), s( n ), t( n ), xn( n ); + if ( !std::isfinite( k.residual( k.x, r ) ) ) return 1; + if ( k.norm( r ) <= tol ) { k.store( x, k.x ); return 0; } + std::vector< value_type > const r_hat = r; + value_type rho_old( 1 ), alpha( 1 ), omega( 1 ); + int result = 1; for ( std::uint_least64_t loops = 0; loops != max_loops; ++loops ) { - ap = A * p; - auto const alpha = dot( r, r_ ) / dot( ap, r_ ); - - if ( zero == alpha ) - return 1; - - if ( std::isinf( alpha ) || std::isnan( alpha ) ) - return 1; - - s = r - alpha * ap; - as = A * s; - auto const omega = dot( as, s ) / dot( as, as ); - - if ( std::isinf( omega ) || std::isnan( omega ) ) - return 1; - - x += alpha * p + omega * s; - new_r = s - omega * as; - auto const beta = dot( new_r, r_ ) * alpha / dot( r, r_ ) / omega; - - if ( std::isinf( beta ) || std::isnan( beta ) ) - return 1; - - r = new_r; - p = r + beta * ( p - omega * ap ); - rem = A * x - b; - - if ( dot( rem, rem ) <= EPS ) - return 0; - } - - return 0; + value_type const rho = k.dot( r_hat, r ); + if ( !k.usable( rho ) ) break; + if ( loops == 0 ) + p = r; + else + { + value_type const beta = ( rho / rho_old ) * ( alpha / omega ); + if ( !std::isfinite( beta ) ) break; + for ( std::size_t i = 0; i != n; ++i ) p[i] = r[i] + beta * ( p[i] - omega * v[i] ); + } + k.mul( p, v ); + value_type const den = k.dot( r_hat, v ); + if ( !k.usable( den ) ) break; + alpha = rho / den; + if ( !std::isfinite( alpha ) ) break; + for ( std::size_t i = 0; i != n; ++i ) { s[i] = r[i] - alpha * v[i]; xn[i] = k.x[i] + alpha * p[i]; } + if ( !matrix_details::all_finite( xn.data(), n ) ) break; + if ( k.residual( xn, tmp ) <= tol ) { k.x = xn; result = 0; break; } + k.mul( s, t ); + value_type const tt = k.dot( t, t ); + if ( !k.usable( tt ) ) { k.x = xn; break; } + omega = k.dot( t, s ) / tt; + if ( !k.usable( omega ) ) { k.x = xn; break; } + for ( std::size_t i = 0; i != n; ++i ) { xn[i] += omega * s[i]; r[i] = s[i] - omega * t[i]; } + if ( !matrix_details::all_finite( xn.data(), n ) ) break; + k.x = xn; + if ( k.residual( k.x, tmp ) <= tol ) { result = 0; break; } + rho_old = rho; + } + k.store( x, k.x ); + return result; } template < typename T1, Allocator A1, typename T2, Allocator A2, typename T3, Allocator A3 > - int bicgstab( const matrix< T1, A1 >& A, + [[nodiscard]] int bicgstab( const matrix< T1, A1 >& A, matrix< T2, A2 >& x, const matrix< T3, A3 >& b, const std::uint_least64_t max_loops = 100, - const T1 eps = 1.0e-10 ) + const T1 eps = 1.0e-10 ) noexcept { - return biconjugate_gradient_stablized_method( A, x, b, max_loops, eps ); + return biconjugate_gradient_stabilized_method( A, x, b, max_loops, eps ); } - template < typename Matrix1, typename Matrix2 > - void cholesky_decomposition( const Matrix1& m, Matrix2& a ) + // S7-R4 (F13, D-026, D-027): Cholesky factorization A = L·Lᴴ of a real symmetric or complex Hermitian + // positive definite A. status() is ok or not_positive_definite; not_positive_definite also covers a + // non-symmetric (non-Hermitian) A, |a_ij − conj(a_ji)| > n·ε·‖A‖∞, a nonfinite input and a singular or + // indefinite A: a pivot d_k = a_kk − Σ|l_kj|² that is not > n·ε·max_i |a_ii| (the D-027 scale). The factor is + // computed from the lower triangle; l() is lower triangular with a real positive diagonal when ok and an + // empty 0×0 matrix otherwise (never NaN). + template < typename T, Allocator A_ = std::allocator< T > > requires linalg_element< T > + class cholesky_factorization { - typedef typename Matrix1::value_type value_type; - better_assert( m.row() == m.col() ); - a = m; - const std::uint_least64_t n = m.row(); + public: + using matrix_type = matrix< T, A_ >; + using real_type = matrix_details::linalg_real_t< T >; - for ( std::uint_least64_t i = 0; i < n; ++i ) - for ( std::uint_least64_t j = i; j < n; ++j ) + cholesky_factorization() noexcept = default; + explicit cholesky_factorization( matrix_type const& a ) noexcept + { + better_assert( a.row() == a.col(), "feng::cholesky_factor: expecting a square matrix, but got ", a.row(), "x", a.col() ); + factor( a ); + } + + [[nodiscard]] linalg_status status() const noexcept { return status_; } + [[nodiscard]] bool ok() const noexcept { return status_ == linalg_status::ok; } + [[nodiscard]] matrix_type const& l() const noexcept { return l_; } + + private: + static T conj_( T const& x ) noexcept + { + if constexpr ( matrix_private::is_std_complex_v< T > ) return std::conj( x ); + else return x; + } + + void factor( matrix_type const& a ) noexcept + { + using R = real_type; + std::size_t const n = a.row(); + T const* const ad = a.data(); + status_ = linalg_status::not_positive_definite; + if ( !matrix_details::all_finite( ad, n * n ) ) return; + R norm_inf{ 0 }, max_diag{ 0 }; + for ( std::size_t i = 0; i != n; ++i ) { - const value_type sum = a[i][j] - std::inner_product( a.row_begin( i ), a.row_begin( i ) + i, a.row_begin( j ), value_type( 0 ) ); - a[j][i] = ( i == j ) ? std::sqrt( sum ) : ( sum / a[i][i] ); + R s{ 0 }; + for ( std::size_t j = 0; j != n; ++j ) s += std::abs( ad[i * n + j] ); + norm_inf = std::max( norm_inf, s ); + max_diag = std::max( max_diag, R( std::abs( ad[i * n + i] ) ) ); + } + R const eps = std::numeric_limits< R >::epsilon(); + R const sym_tol = static_cast< R >( n ) * eps * norm_inf; + for ( std::size_t i = 0; i != n; ++i ) + for ( std::size_t j = i; j != n; ++j ) + if ( std::abs( ad[i * n + j] - conj_( ad[j * n + i] ) ) > sym_tol ) return; + R const pivot_tol = static_cast< R >( n ) * eps * max_diag; + matrix_type l( a ); + T* const ld = l.data(); + std::fill( ld, ld + n * n, T{ 0 } ); + for ( std::size_t j = 0; j != n; ++j ) + { + R d = std::real( ad[j * n + j] ); + for ( std::size_t k = 0; k != j; ++k ) d -= std::norm( ld[j * n + k] ); + if ( !( d > pivot_tol ) || !std::isfinite( d ) ) return; + R const ljj = std::sqrt( d ); + ld[j * n + j] = T( ljj ); + for ( std::size_t i = j + 1; i != n; ++i ) + { + T s = ad[i * n + j]; + for ( std::size_t k = 0; k != j; ++k ) s -= ld[i * n + k] * conj_( ld[j * n + k] ); + ld[i * n + j] = s / ljj; + } } + if ( !matrix_details::all_finite( ld, n * n ) ) return; + l_ = std::move( l ); + status_ = linalg_status::ok; + } + + matrix_type l_; + linalg_status status_ = linalg_status::ok; + }; + + template < typename T, Allocator A > requires linalg_element< T > + [[nodiscard]] cholesky_factorization< T, A > cholesky_factor( matrix< T, A > const& a ) noexcept + { + return cholesky_factorization< T, A >( a ); + } - for ( std::uint_least64_t i = 1; i < n; ++i ) - std::fill( a.upper_diag_begin( i ), a.upper_diag_end( i ), value_type() ); + // Legacy Cholesky: 0 with a = L (A = L·Lᴴ), or 1 with a unchanged when cholesky_factor is not ok (S7-R4). + template < typename Matrix1, typename Matrix2 > requires linalg_element< typename Matrix1::value_type > + [[nodiscard]] int cholesky_decomposition( const Matrix1& m, Matrix2& a ) noexcept + { + better_assert( m.row() == m.col(), "cholesky_decomposition: expecting a square matrix, but got ", m.row(), "x", m.col() ); + auto const f = cholesky_factor( m ); + if ( !f.ok() ) + return 1; + a = f.l(); + return 0; } + // S7-R5 (D-029): conjugate gradient squared for A·x = b; the same return and x rules as bicgstab. template < typename T1, Allocator A1, typename T2, Allocator A2, typename T3, Allocator A3 > - int conjugate_gradient_squared( const matrix< T1, A1 >& A, + [[nodiscard]] int conjugate_gradient_squared( const matrix< T1, A1 >& A, matrix< T2, A2 >& x, const matrix< T3, A3 >& b, const std::uint_least64_t max_loops = 100, - const T1 eps = 1.0e-10 ) + const T1 eps = 1.0e-10 ) noexcept { - typedef matrix< T1, A1 > matrix_type; - typedef typename matrix_type::value_type value_type; - typedef typename matrix_type::size_type size_type; - better_assert( A.row() == A.col() ); - better_assert( A.row() == b.row() ); - better_assert( b.col() == 1 ); - size_type const n = A.row(); - x.resize( n, 1 ); - - if ( dot( x, x ) == value_type() ) - x = b; - - matrix_type const r_ = b - A * x; - matrix_type r = r_; - matrix_type p = r_; - matrix_type u = r_; - matrix_type ap = r_; - matrix_type q = r_; - matrix_type new_r = r_; - matrix_type uq = r_; - matrix_type rem = r_; - value_type const EPS = n * eps * eps; - - if ( dot( r_, r_ ) < EPS ) + typedef T1 value_type; + better_assert( A.row() == A.col(), "cgs: expecting a square A, but got ", A.row(), "x", A.col() ); + better_assert( A.row() == b.row() && b.col() == 1, "cgs: expecting a column b with ", A.row(), " rows, but got ", b.row(), "x", b.col() ); + iterative_private::krylov< value_type > k( A, x, b ); + if ( !k.finite_input() ) return 1; + std::size_t const n = k.n; + value_type const bnorm = k.norm( k.b ); + if ( bnorm == value_type( 0 ) ) + { + std::fill( k.x.begin(), k.x.end(), value_type( 0 ) ); + k.store( x, k.x ); return 0; - - for ( std::uint_least64_t i = 0; i != max_loops; ++i ) + } + value_type const tol = eps * bnorm; + std::vector< value_type > r( n ), tmp( n ), u( n ), p( n ), q( n ), uq( n ), v( n ), xn( n ); + if ( !std::isfinite( k.residual( k.x, r ) ) ) return 1; + if ( k.norm( r ) <= tol ) { k.store( x, k.x ); return 0; } + std::vector< value_type > const r_hat = r; + value_type rho_old( 1 ); + int result = 1; + for ( std::uint_least64_t loops = 0; loops != max_loops; ++loops ) { - ap = A * p; - value_type const alpha = dot( r, r_ ) / dot( ap, r_ ); - - if ( std::isinf( alpha ) || std::isnan( alpha ) ) + value_type const rho = k.dot( r_hat, r ); + if ( !k.usable( rho ) ) break; + if ( loops == 0 ) { - return 1; + u = r; + p = u; } - - q = u - alpha * ap; - uq = u + q; - x += alpha * uq; - rem = A * x - b; - - if ( dot( rem, rem ) < EPS ) - return 0; - - new_r = r - alpha * A * uq; - value_type const beta = dot( new_r, r_ ) / dot( r, r_ ); - - if ( std::isinf( beta ) || std::isnan( beta ) ) + else { - return 1; + value_type const beta = rho / rho_old; + if ( !std::isfinite( beta ) ) break; + for ( std::size_t i = 0; i != n; ++i ) + { + u[i] = r[i] + beta * q[i]; + p[i] = u[i] + beta * ( q[i] + beta * p[i] ); + } } - - r = new_r; - u = r + beta * q; - p = u + beta * ( q + beta * p ); + k.mul( p, v ); + value_type const den = k.dot( r_hat, v ); + if ( !k.usable( den ) ) break; + value_type const alpha = rho / den; + if ( !std::isfinite( alpha ) ) break; + for ( std::size_t i = 0; i != n; ++i ) + { + q[i] = u[i] - alpha * v[i]; + uq[i] = u[i] + q[i]; + xn[i] = k.x[i] + alpha * uq[i]; + } + if ( !matrix_details::all_finite( xn.data(), n ) ) break; + k.x = xn; + k.mul( uq, tmp ); + for ( std::size_t i = 0; i != n; ++i ) r[i] -= alpha * tmp[i]; + if ( k.residual( k.x, tmp ) <= tol ) { result = 0; break; } + rho_old = rho; } - - return 0; + k.store( x, k.x ); + return result; } template < typename T1, Allocator A1, typename T2, Allocator A2, typename T3, Allocator A3 > - int cgs( const matrix< T1, A1 >& A, + [[nodiscard]] int cgs( const matrix< T1, A1 >& A, matrix< T2, A2 >& x, const matrix< T3, A3 >& b, const std::uint_least64_t max_loops = 100, - const T1 eps = 1.0e-10 ) + const T1 eps = 1.0e-10 ) noexcept { return conjugate_gradient_squared( A, x, b, max_loops, eps ); } + // Experimental (S7, D-029): not qualified; householder has no status and is not covered by the S7 tests. template < typename Matrix1, typename Matrix2, typename Matrix3 > - void householder( const Matrix1& A, Matrix2& Q, Matrix3& D ) + void householder( const Matrix1& A, Matrix2& Q, Matrix3& D ) noexcept { typedef Matrix1 matrix_type; typedef typename matrix_type::value_type value_type; @@ -5774,9 +7484,8 @@ namespace feng better_assert( A.row() == A.col() ); size_type const n = A.row(); value_type const zero = value_type( 0 ); - value_type const one = value_type( 1 ); value_type const two = value_type( 2 ); - Matrix1 const I = eye( n, A ); + Matrix1 const I = eye< value_type >( n ); Q = eye< value_type >( n ); D = A; Matrix1 x( n, 1 ); @@ -5817,7 +7526,7 @@ namespace feng namespace eigen_jacobi_private { template < typename Matrix > - typename Matrix::value_type norm( const Matrix& A ) + typename Matrix::value_type norm( const Matrix& A ) noexcept { typedef typename Matrix::value_type value_type; auto A_ = abs( A ); @@ -5832,7 +7541,7 @@ namespace feng return std::sqrt( sum ) * max_elem; } template < typename Matrix1, typename Matrix2 > - void rotate( Matrix1& A, Matrix2& V, const std::uint_least64_t p, const std::uint_least64_t q ) + void rotate( Matrix1& A, Matrix2& V, const std::uint_least64_t p, const std::uint_least64_t q ) noexcept { typedef typename Matrix1::value_type value_type; auto const one = value_type( 1 ); @@ -5863,27 +7572,28 @@ namespace feng } } } + // Experimental (S7, D-029): not qualified; eigen_jacobi has an unbounded rotation loop and no status. template < typename Matrix1, typename Matrix2, typename T = double > - std::uint_least64_t eigen_jacobi( const Matrix1& A, Matrix2& V, std::vector< T >& Lambda, const T eps = T( 1.0e-10 ) ) + std::uint_least64_t eigen_jacobi( const Matrix1& A, Matrix2& V, std::vector< T >& Lambda, const T eps = T( 1.0e-10 ) ) noexcept { Lambda.resize( A.row() ); return eigen_jacobi( A, V, Lambda.begin(), eps ); } template < typename Matrix1, typename Matrix2, typename T = double > - std::uint_least64_t eigen_jacobi( const Matrix1& A, Matrix2& V, std::valarray< T >& Lambda, const T eps = T( 1.0e-10 ) ) + std::uint_least64_t eigen_jacobi( const Matrix1& A, Matrix2& V, std::valarray< T >& Lambda, const T eps = T( 1.0e-10 ) ) noexcept { Lambda.resize( A.row() ); - return eigen_jacobi( A, V, Lambda.begin(), eps ); + return eigen_jacobi( A, V, std::begin( Lambda ), eps ); } template < typename Matrix1, typename Matrix2, typename T, typename A_, typename T_ = double > - std::uint_least64_t eigen_jacobi( const Matrix1& A, Matrix2& V, matrix< T, A_ >& Lambda, const T_ eps = T_( 1.0e-10 ) ) + std::uint_least64_t eigen_jacobi( const Matrix1& A, Matrix2& V, matrix< T, A_ >& Lambda, const T_ eps = T_( 1.0e-10 ) ) noexcept { Lambda.resize( A.row(), A.col() ); Lambda = T( 0 ); return eigen_jacobi( A, V, Lambda.diag_begin(), eps ); } template < typename Matrix1, typename Matrix2, typename Otor, typename T = double > - std::uint_least64_t eigen_jacobi( const Matrix1& A, Matrix2& V, Otor o, const T eps = T( 1.0e-10 ) ) + std::uint_least64_t eigen_jacobi( const Matrix1& A, Matrix2& V, Otor o, const T eps = T( 1.0e-10 ) ) noexcept { typedef typename Matrix1::value_type value_type; typedef typename Matrix1::size_type size_type; @@ -5926,15 +7636,12 @@ namespace feng return size_type( -1 ); } + // Experimental (S7, D-029): not qualified; cyclic_eigen_jacobi ignores eps and reports no status. template < typename Matrix1, typename Matrix2, typename Otor, typename T = double > - std::uint_least64_t cyclic_eigen_jacobi( const Matrix1& A, Matrix2& V, Otor o, std::uint_least64_t max_rot = 80, const T eps = T( 1.0e-10 ) ) + std::uint_least64_t cyclic_eigen_jacobi( const Matrix1& A, Matrix2& V, Otor o, std::uint_least64_t max_rot = 80, [[maybe_unused]] const T eps = T( 1.0e-10 ) ) noexcept { typedef typename Matrix1::value_type value_type; typedef typename Matrix1::size_type size_type; - auto const compare_func = [eps]( const value_type lhs, const value_type rhs ) - { - return std::abs( lhs - rhs ) < eps; - }; better_assert( A.row() == A.col() ); size_type i = 0; auto a = A; @@ -5961,45 +7668,46 @@ namespace feng return i * n * n; } template < typename Matrix1, typename Matrix2, typename T = double > - std::uint_least64_t cyclic_eigen_jacobi( const Matrix1& A, Matrix2& V, std::vector< T >& Lambda, std::uint_least64_t const max_rot = 80, const T eps = T( 1.0e-10 ) ) + std::uint_least64_t cyclic_eigen_jacobi( const Matrix1& A, Matrix2& V, std::vector< T >& Lambda, std::uint_least64_t const max_rot = 80, const T eps = T( 1.0e-10 ) ) noexcept { Lambda.resize( A.row() ); return cyclic_eigen_jacobi( A, V, Lambda.begin(), max_rot, eps ); } template < typename Matrix1, typename Matrix2, typename T = double > - std::uint_least64_t cyclic_eigen_jacobi( const Matrix1& A, Matrix2& V, std::valarray< T >& Lambda, std::uint_least64_t const max_rot = 80, const T eps = T( 1.0e-10 ) ) + std::uint_least64_t cyclic_eigen_jacobi( const Matrix1& A, Matrix2& V, std::valarray< T >& Lambda, std::uint_least64_t const max_rot = 80, const T eps = T( 1.0e-10 ) ) noexcept { Lambda.resize( A.row() ); - return cyclic_eigen_jacobi( A, V, Lambda.begin(), max_rot, eps ); + return cyclic_eigen_jacobi( A, V, std::begin( Lambda ), max_rot, eps ); } template < typename Matrix1, typename Matrix2, typename T, typename A_, typename T_ = double > - std::uint_least64_t cyclic_eigen_jacobi( const Matrix1& A, Matrix2& V, matrix< T, A_ >& Lambda, std::uint_least64_t const max_rot = 80, const T_ eps = T_( 1.0e-10 ) ) + std::uint_least64_t cyclic_eigen_jacobi( const Matrix1& A, Matrix2& V, matrix< T, A_ >& Lambda, std::uint_least64_t const max_rot = 80, const T_ eps = T_( 1.0e-10 ) ) noexcept { Lambda.resize( A.row(), A.col() ); Lambda = T( 0 ); return cyclic_eigen_jacobi( A, V, Lambda.diag_begin(), max_rot, eps ); } + // Experimental (S7, D-029): not qualified; eigen_real_symmetric rests on householder and eigen_jacobi. template < typename Matrix1, typename Matrix2, typename T = double > - void eigen_real_symmetric( const Matrix1& A, Matrix2& V, std::vector< T >& Lambda, const T eps = T( 1.0e-10 ) ) + void eigen_real_symmetric( const Matrix1& A, Matrix2& V, std::vector< T >& Lambda, const T eps = T( 1.0e-10 ) ) noexcept { Lambda.resize( A.row() ); return eigen_real_symmetric( A, V, Lambda.begin(), eps ); } template < typename Matrix1, typename Matrix2, typename T = double > - void eigen_real_symmetric( const Matrix1& A, Matrix2& V, std::valarray< T >& Lambda, const T eps = T( 1.0e-10 ) ) + void eigen_real_symmetric( const Matrix1& A, Matrix2& V, std::valarray< T >& Lambda, const T eps = T( 1.0e-10 ) ) noexcept { Lambda.resize( A.row() ); - return eigen_real_symmetric( A, V, Lambda.begin(), eps ); + return eigen_real_symmetric( A, V, std::begin( Lambda ), eps ); } template < typename Matrix1, typename Matrix2, typename T, typename A_, typename T_ = double > - void eigen_real_symmetric( const Matrix1& A, Matrix2& V, matrix< T, A_ >& Lambda, const T_ eps = T_( 1.0e-10 ) ) + void eigen_real_symmetric( const Matrix1& A, Matrix2& V, matrix< T, A_ >& Lambda, const T_ eps = T_( 1.0e-10 ) ) noexcept { Lambda.resize( A.row(), A.col() ); Lambda = T( 0 ); return eigen_real_symmetric( A, V, Lambda.diag_begin(), eps ); } template < typename Matrix1, typename Matrix2, typename Otor, typename T = double > - void eigen_real_symmetric( const Matrix1& A, Matrix2& V, Otor o, const T eps = T( 1.0e-10 ) ) + void eigen_real_symmetric( const Matrix1& A, Matrix2& V, Otor o, const T eps = T( 1.0e-10 ) ) noexcept { better_assert( A.row() == A.col() ); Matrix1 D( A ); @@ -6008,35 +7716,36 @@ namespace feng eigen_jacobi( D, V, o, eps ); V = Q * V; } + // Experimental (S7, D-029): not qualified; eigen_hermitian rests on eigen_real_symmetric. template < typename Complex_Matrix1, typename Complex_Matrix2, typename T = double > - void eigen_hermitian( const Complex_Matrix1& A, Complex_Matrix2& V, std::vector< T >& Lambda, const T eps = T( 1.0e-20 ) ) + void eigen_hermitian( const Complex_Matrix1& A, Complex_Matrix2& V, std::vector< T >& Lambda, const T eps = T( 1.0e-20 ) ) noexcept { Lambda.resize( A.row() ); return eigen_hermitian_impl( A, V, Lambda.begin(), eps ); } template < typename Complex_Matrix1, typename Complex_Matrix2, typename T = double > - void eigen_hermitian( const Complex_Matrix1& A, Complex_Matrix2& V, std::valarray< T >& Lambda, const T eps = T( 1.0e-20 ) ) + void eigen_hermitian( const Complex_Matrix1& A, Complex_Matrix2& V, std::valarray< T >& Lambda, const T eps = T( 1.0e-20 ) ) noexcept { Lambda.resize( A.row() ); - return eigen_hermitian_impl( A, V, Lambda.begin(), eps ); + return eigen_hermitian_impl( A, V, std::begin( Lambda ), eps ); } template < typename Complex_Matrix1, typename Complex_Matrix2, typename T, typename A_, typename T_ = double > - void eigen_hermitian( const Complex_Matrix1& A, Complex_Matrix2& V, matrix< T, A_ >& Lambda, const T_ eps = T_( 1.0e-20 ) ) + void eigen_hermitian( const Complex_Matrix1& A, Complex_Matrix2& V, matrix< T, A_ >& Lambda, const T_ eps = T_( 1.0e-20 ) ) noexcept { Lambda.resize( A.row(), A.col() ); Lambda = T( 0 ); return eigen_hermitian_impl( A, V, Lambda.diag_begin(), eps ); } template < typename T1, Allocator A1, typename T2, Allocator A2, typename Otor, typename T = double > - void eigen_hermitian_impl( const matrix< std::complex< T1 >, A1 >& A, matrix< std::complex< T2 >, A2 >& V, Otor o, const T eps = T( 1.0e-20 ) ) + void eigen_hermitian_impl( const matrix< std::complex< T1 >, A1 >& A, matrix< std::complex< T2 >, A2 >& V, Otor o, const T eps = T( 1.0e-20 ) ) noexcept { better_assert( A.row() == A.col() ); std::uint_least64_t const n = A.row(); auto const A_ = real( A ); auto const B_ = imag( A ); auto const AA = ( A_ || ( -B_ ) ) && ( B_ || A_ ); - matrix< T1, A1 > VV( n + n, n + n ); - matrix< T1, A1 > LL( n + n, n + n ); + matrix< T1, typename std::allocator_traits< A1 >::template rebind_alloc< T1 > > VV( n + n, n + n ); + matrix< T1, typename std::allocator_traits< A1 >::template rebind_alloc< T1 > > LL( n + n, n + n ); eigen_real_symmetric( AA, VV, LL, eps ); std::vector< T1 > vec( n + n ); std::copy( LL.diag_begin(), LL.diag_end(), vec.begin() ); @@ -6058,8 +7767,9 @@ namespace feng *o++ = vec[i + i]; } } + // Experimental (S7, D-029): not qualified; eigen_power_iteration has an unbounded loop and no status. template < typename T, typename A_, typename O > - T eigen_power_iteration( const matrix< T, A_ >& A, O output, const T eps = T( 1.0e-5 ) ) + T eigen_power_iteration( const matrix< T, A_ >& A, O output, const T eps = T( 1.0e-5 ) ) noexcept { better_assert( A.row() == A.col() ); matrix< T, A_ > b( A.col(), 1 ); @@ -6087,13 +7797,13 @@ namespace feng return T( 0 ); } template < typename T, typename A_ > - T eigen_power_iteration( const matrix< T, A_ >& A, const T eps = T( 1.0e-5 ) ) + T eigen_power_iteration( const matrix< T, A_ >& A, const T eps = T( 1.0e-5 ) ) noexcept { matrix< T, A_ > b( A.col(), 1 ); return eigen_power_iteration( A, b.begin(), eps ); } template < typename T, typename A_, typename O > - T eigen_power_iteration( const matrix< std::complex< T >, A_ >& A, O output, const T eps = T( 1.0e-5 ) ) + T eigen_power_iteration( const matrix< std::complex< T >, A_ >& A, O output, const T eps = T( 1.0e-5 ) ) noexcept { better_assert( A.row() == A.col() ); matrix< std::complex< T >, A_ > b( A.col(), 1 ); @@ -6143,14 +7853,14 @@ namespace feng return T( 0 ); } template < typename T, typename A_ > - T eigen_power_iteration( const matrix< std::complex< T >, A_ >& A, const T eps = T( 1.0e-5 ) ) + T eigen_power_iteration( const matrix< std::complex< T >, A_ >& A, const T eps = T( 1.0e-5 ) ) noexcept { matrix< std::complex< T >, A_ > b( A.col(), 1 ); return eigen_power_iteration( A, b.begin(), eps ); } /* template < typename Matrix > - typename Matrix::value_type + [[nodiscard]] typename Matrix::value_type norm( const Matrix& A ) { typedef typename Matrix::value_type value_type; @@ -6165,7 +7875,7 @@ namespace feng } template < typename Matrix > - typename Matrix::value_type + [[nodiscard]] typename Matrix::value_type norm( const Matrix& A, const std::uint_least64_t n ) { typedef typename Matrix::value_type value_type; @@ -6192,7 +7902,7 @@ namespace feng } */ template < Matrix Mat > - auto norm_1( Mat const& A ) + [[nodiscard]] auto norm_1( Mat const& A ) noexcept { typedef typename Mat::value_type value_type; std::vector< value_type > m( A.col() ); @@ -6206,7 +7916,7 @@ namespace feng } template < typename T, typename A_ > - T norm_1( const matrix< std::complex< T >, A_ >& A ) + [[nodiscard]] T norm_1( const matrix< std::complex< T >, A_ >& A ) noexcept { std::vector< T > m( A.col() ); @@ -6218,7 +7928,7 @@ namespace feng return *( std::max_element( m.begin(), m.end() ) ); } template < Matrix Mat > - auto norm_2( Mat const& A ) + [[nodiscard]] auto norm_2( Mat const& A ) noexcept { return std::sqrt( eigen_power_iteration( A ) ); } @@ -6236,35 +7946,41 @@ namespace feng typedef typename fix_complex_value_type< T >::value_type value_type; }; } - template < Matrix Mat > - Mat expm( Mat const& A ) + // S7-R5 (D-029): matrix exponential by scaling and squaring with the degree-13 Padé approximant (Higham 2005). + // expm(0) = I and a diagonal A gives diag(exp(a_ii)) exactly (shortcut); the scaling 2^-s is applied with + // std::ldexp; the Padé quotient F = (V − U)⁻¹(V + U) is solved with lu_factor; a nonfinite A (or a failed + // Padé solve) gives an all-NaN matrix of A's shape. Real and complex elements. + template < Matrix Mat > requires linalg_element< typename Mat::value_type > + [[nodiscard]] Mat expm( Mat const& A ) noexcept { typedef Mat matrix_type; - typedef typename matrix_type::value_type value_type_; - typedef typename expm_private::fix_complex_value_type< value_type_ >::value_type value_type; - typedef typename matrix_type::size_type size_type; - better_assert( A.row() == A.col() ); - static const value_type theta[] = { 0.000000000000000e+000, - 3.650024139523051e-008, - 5.317232856892575e-004, - 1.495585217958292e-002, - 8.536352760102745e-002, - 2.539398330063230e-001, - 5.414660951208968e-001, - 9.504178996162932e-001, - 1.473163964234804e+000, - 2.097847961257068e+000, - 2.811644121620263e+000, - 3.602330066265032e+000, - 4.458935413036850e+000, - 5.371920351148152e+000 - }; - value_type const norm_A = norm_1( A ); - value_type const ratio = norm_A / theta[13]; - size_type const s = ratio < value_type( 1 ) ? 0 : static_cast< size_type >( std::ceil( std::log2( ratio ) ) ); - value_type const s__2 = s ? value_type( 1 << s ) : value_type( 1 ); - matrix_type const& _A = A / s__2; - size_type const n = _A.row(); + typedef typename matrix_type::value_type T; + typedef typename expm_private::fix_complex_value_type< T >::value_type value_type; + better_assert( A.row() == A.col(), "feng::expm: expecting a square matrix, but got ", A.row(), "x", A.col() ); + std::size_t const n = A.row(); + auto nan_result = [&A]() + { + matrix_type ans( A ); + value_type const q = std::numeric_limits< value_type >::quiet_NaN(); + if constexpr ( matrix_private::is_std_complex_v< T > ) std::fill( ans.begin(), ans.end(), T( q, q ) ); + else std::fill( ans.begin(), ans.end(), T( q ) ); + return ans; + }; + if ( !matrix_details::all_finite( A.data(), n * n ) ) return nan_result(); + + bool diagonal = true; + for ( std::size_t i = 0; i != n && diagonal; ++i ) + for ( std::size_t j = 0; j != n; ++j ) + if ( i != j && A[i][j] != T( 0 ) ) { diagonal = false; break; } + if ( diagonal ) + { + matrix_type ans( A ); + std::fill( ans.begin(), ans.end(), T( 0 ) ); + for ( std::size_t i = 0; i != n; ++i ) ans[i][i] = std::exp( A[i][i] ); + return ans; + } + + static const value_type theta13 = 5.371920351148152e+000; static value_type const c[] = { 0.000000000000000, 6.4764752532480000e+16, 3.2382376266240000e+16, @@ -6281,91 +7997,358 @@ namespace feng 1.82e+2, 1 }; - matrix_type const& _A2 = _A * _A; - matrix_type const& _A4 = _A2 * _A2; - matrix_type const& _A6 = _A2 * _A4; - matrix_type const& U = _A * ( _A6 * ( c[14] * _A6 + c[12] * _A4 + c[10] * _A2 ) + c[8] * _A6 + c[6] * _A4 + c[4] * _A2 + c[2] * eye< value_type >( n, n ) ); - matrix_type const& V = _A6 * ( c[13] * _A6 + c[11] * _A4 + c[9] * _A2 ) + c[7] * _A6 + c[5] * _A4 + c[3] * _A2 + c[1] * eye< value_type >( n, n ); - matrix_type const& VU = V + U; - matrix_type const& UV = V - U; - matrix_type F = VU / UV; - - for ( size_type i = 0; i != s; ++i ) - F *= F; - + value_type norm_A( 0 ); // ‖A‖₁ + for ( std::size_t j = 0; j != n; ++j ) + { + value_type sum( 0 ); + for ( std::size_t i = 0; i != n; ++i ) sum += std::abs( A[i][j] ); + norm_A = std::max( norm_A, sum ); + } + value_type const ratio = norm_A / theta13; + int const s = ratio > value_type( 1 ) ? static_cast< int >( std::ceil( std::log2( ratio ) ) ) : 0; + matrix_type _A( A ); + if ( s ) + for ( auto& v : _A ) + { + if constexpr ( matrix_private::is_std_complex_v< T > ) v = T( std::ldexp( v.real(), -s ), std::ldexp( v.imag(), -s ) ); + else v = std::ldexp( v, -s ); + } + matrix_type I( A ); + std::fill( I.begin(), I.end(), T( 0 ) ); + for ( std::size_t i = 0; i != n; ++i ) I[i][i] = T( 1 ); + matrix_type const _A2 = _A * _A; + matrix_type const _A4 = _A2 * _A2; + matrix_type const _A6 = _A2 * _A4; + matrix_type const U = _A * ( _A6 * ( c[14] * _A6 + c[12] * _A4 + c[10] * _A2 ) + c[8] * _A6 + c[6] * _A4 + c[4] * _A2 + c[2] * I ); + matrix_type const V = _A6 * ( c[13] * _A6 + c[11] * _A4 + c[9] * _A2 ) + c[7] * _A6 + c[5] * _A4 + c[3] * _A2 + c[1] * I; + auto quotient = lu_factor( matrix_type( V - U ) ).solve( matrix_type( V + U ) ); + if ( !quotient.ok() ) return nan_result(); + matrix_type F = std::move( quotient.value ); + for ( int i = 0; i != s; ++i ) + F = F * F; return F; } - namespace fft_private + + // S7-R5 (D-026): exp(A) into out; nonfinite (out unchanged) for a nonfinite A or a nonfinite result, else ok + template < typename T, Allocator A_ > requires linalg_element< T > + [[nodiscard]] linalg_status expm( matrix< T, A_ > const& a, matrix< T, A_ >& out ) noexcept + { + if ( !matrix_details::all_finite( a.data(), a.size() ) ) return linalg_status::nonfinite; + matrix< T, A_ > e = expm( a ); + if ( !matrix_details::all_finite( e.data(), e.size() ) ) return linalg_status::nonfinite; + out = std::move( e ); + return linalg_status::ok; + } + // S8-R2, S8-R3 (F14, D-007, D-031): numpy.fft.fft2/ifft2 by a separable in-house FFT (radix-2 for power-of-two + // lengths, Bluestein through a power-of-two convolution otherwise) and fftshift/ifftshift as numpy rolls. + namespace matrix_details { template < typename T > - struct add_complex + struct fft_complex; + template < std::integral T > + struct fft_complex< T > { - typedef std::complex< T > result_type; + using type = std::complex< double >; + }; + template < std::floating_point T > + struct fft_complex< T > + { + using type = std::complex< T >; }; template < typename T > - struct add_complex< std::complex< T >> + struct fft_complex< std::complex< T > > { - typedef std::complex< T > result_type; + using type = std::complex< T >; }; - } + // D-031: float, double, long double -> complex of the same real type; complex stays; integers -> complex + template < typename T > + using fft_complex_t = typename fft_complex< T >::type; - template < Matrix Mat > - auto fft( Mat const& x ) - { - typedef typename Mat::value_type value_type; - typedef typename fft_private::add_complex< value_type >::result_type complex_type; - matrix< complex_type > X( x.row(), x.col() ); - auto make_omege = []( auto k, auto n, auto N ) + template < typename R > + inline constexpr R fft_pi = static_cast< R >( 3.141592653589793238462643383279502884L ); + + // twiddle table for a power-of-two length n and a sign: tw[j] = exp(sign 2 pi i j / n), j < n/2, each by std::polar + template < typename R > + void fft_twiddles( std::vector< std::complex< R > >& tw, std::size_t n, int sign ) { - double const pi = 3.1415926535897932384626433; - double const theta = -pi * 2.0 * k * n / static_cast< double >( N ); - return complex_type{ std::cos( theta ), std::sin( theta ) }; - }; - std::uint_least64_t const R = X.row(); - std::uint_least64_t const C = X.col(); + tw.resize( n / 2 ); + R const step = static_cast< R >( sign ) * R( 2 ) * fft_pi< R > / static_cast< R >( n ); + for ( std::size_t j = 0; j != n / 2; ++j ) + tw[j] = std::polar( R( 1 ), step * static_cast< R >( j ) ); + } - for ( std::uint_least64_t r = 0; r != R; ++r ) - for ( std::uint_least64_t c = 0; c != C; ++c ) + // in-place iterative radix-2 transform of a power-of-two length n with a precomputed twiddle table tw + template < typename R > + void fft_radix2_apply( std::complex< R >* data, std::size_t n, std::complex< R > const* tw ) noexcept + { + for ( std::size_t i = 1, j = 0; i < n; ++i ) + { + std::size_t bit = n >> 1; + for ( ; j & bit; bit >>= 1 ) + j ^= bit; + j ^= bit; + if ( i < j ) std::swap( data[i], data[j] ); + } + for ( std::size_t len = 2; len <= n; len <<= 1 ) { - complex_type X_rc{ 0.0, 0.0 }; + std::size_t const half = len >> 1; + std::size_t const stride = n / len; + for ( std::size_t i = 0; i < n; i += len ) + for ( std::size_t j = 0; j != half; ++j ) + { + std::complex< R > const u = data[i + j]; + std::complex< R > const v = data[i + j + half] * tw[j * stride]; + data[i + j] = u + v; + data[i + j + half] = u - v; + } + } + } - for ( std::uint_least64_t r_ = 0; r_ != R; ++r_ ) - { - complex_type tmp{ 0.0, 0.0 }; + // in-place iterative radix-2 transform of a power-of-two length n; sign -1 forward, +1 inverse (unscaled); + // twiddles exp(sign 2 pi i j / n) computed per index with std::polar, no recurrence (per-call tables: the + // reference path behind fft2_reference) + template < typename R > + void fft_radix2( std::complex< R >* data, std::size_t n, int sign ) noexcept + { + std::vector< std::complex< R > > tw; + fft_twiddles( tw, n, sign ); + fft_radix2_apply( data, n, tw.data() ); + } + + // Bluestein chirp w_k = exp(sign i pi (k^2 mod 2n) / n), k < n + template < typename R > + void fft_chirp( std::vector< std::complex< R > >& w, std::size_t n, int sign ) + { + w.resize( n ); + std::uint_least64_t const two_n = 2 * static_cast< std::uint_least64_t >( n ); + std::uint_least64_t q = 0; // k^2 mod 2n, exact in 64-bit integers: (k+1)^2 = k^2 + 2k + 1 + for ( std::size_t k = 0; k != n; ++k ) + { + w[k] = std::polar( R( 1 ), static_cast< R >( sign ) * fft_pi< R > * static_cast< R >( q ) / static_cast< R >( n ) ); + q = ( q + 2 * static_cast< std::uint_least64_t >( k ) + 1 ) % two_n; + } + } + + // Bluestein: X_k = w_k sum_j (x_j w_j) conj(w_{k-j}), w_k = exp(sign i pi (k^2 mod 2n) / n), the + // convolution done by radix-2 transforms of length M, the power of two >= 2n - 1 (reference path) + template < typename R > + void fft_bluestein( std::complex< R >* data, std::size_t n, int sign ) noexcept + { + std::size_t const m = std::bit_ceil( 2 * n - 1 ); + std::vector< std::complex< R > > w; + fft_chirp( w, n, sign ); + std::vector< std::complex< R > > a( m ), b( m ); + for ( std::size_t k = 0; k != n; ++k ) + a[k] = data[k] * w[k]; + b[0] = std::conj( w[0] ); + for ( std::size_t k = 1; k != n; ++k ) + b[k] = b[m - k] = std::conj( w[k] ); + fft_radix2( a.data(), m, -1 ); + fft_radix2( b.data(), m, -1 ); + for ( std::size_t k = 0; k != m; ++k ) + a[k] *= b[k]; + fft_radix2( a.data(), m, +1 ); + R const inv_m = R( 1 ) / static_cast< R >( m ); // m is a power of two: exact + for ( std::size_t k = 0; k != n; ++k ) + data[k] = w[k] * ( a[k] * inv_m ); + } + + // 1-D transform of a contiguous sequence of length n (reference path) + template < typename R > + void fft_1d( std::complex< R >* data, std::size_t n, int sign ) noexcept + { + if ( n < 2 ) return; + if ( std::has_single_bit( n ) ) + fft_radix2( data, n, sign ); + else + fft_bluestein( data, n, sign ); + } - for ( std::uint_least64_t c_ = 0; c_ != C; ++c_ ) - tmp += x[r][c] * make_omege( c, c_, C ); + // S10-R3 (B-042, fft-plans): everything a length-n transform of one sign needs, built once per fft2 call + // by the same expressions as fft_radix2 / fft_bluestein: the twiddle table (radix-2 n), or the chirp w, the + // forward transform of b and the length-m tables for -1 and +1 (Bluestein n), plus the length-m scratch a + template < typename R > + struct fft_plan + { + std::size_t n = 0; + std::size_t m = 0; // 0 for radix-2 (or n < 2), else the Bluestein convolution length + std::vector< std::complex< R > > tw; // radix-2: exp(sign 2 pi i j / n) + std::vector< std::complex< R > > w; // Bluestein chirp + std::vector< std::complex< R > > b_hat; // forward radix-2 transform of b + std::vector< std::complex< R > > tw_fwd; // length-m tables, sign -1 + std::vector< std::complex< R > > tw_inv; // and +1 + std::vector< std::complex< R > > a; // scratch - X_rc += tmp * make_omege( r, r_, R ); + fft_plan( std::size_t n_, int sign ) : n{ n_ } + { + if ( n < 2 ) return; + if ( std::has_single_bit( n ) ) + { + fft_twiddles( tw, n, sign ); + return; } + m = std::bit_ceil( 2 * n - 1 ); + fft_chirp( w, n, sign ); + fft_twiddles( tw_fwd, m, -1 ); + fft_twiddles( tw_inv, m, +1 ); + b_hat.assign( m, std::complex< R >{} ); + b_hat[0] = std::conj( w[0] ); + for ( std::size_t k = 1; k != n; ++k ) + b_hat[k] = b_hat[m - k] = std::conj( w[k] ); + fft_radix2_apply( b_hat.data(), m, tw_fwd.data() ); + a.resize( m ); + } - X[r][c] = X_rc; + void operator()( std::complex< R >* data ) noexcept + { + if ( n < 2 ) return; + if ( m == 0 ) + { + fft_radix2_apply( data, n, tw.data() ); + return; + } + for ( std::size_t k = 0; k != n; ++k ) + a[k] = data[k] * w[k]; + std::fill( a.begin() + static_cast< std::ptrdiff_t >( n ), a.end(), std::complex< R >{} ); + fft_radix2_apply( a.data(), m, tw_fwd.data() ); + for ( std::size_t k = 0; k != m; ++k ) + a[k] *= b_hat[k]; + fft_radix2_apply( a.data(), m, tw_inv.data() ); + R const inv_m = R( 1 ) / static_cast< R >( m ); // m is a power of two: exact + for ( std::size_t k = 0; k != n; ++k ) + data[k] = w[k] * ( a[k] * inv_m ); } + }; - return X; - } + // the input converted to the complex result type + template < Matrix Mat > + auto fft2_load( Mat const& x ) + { + using complex_type = fft_complex_t< typename Mat::value_type >; + using real_type = typename complex_type::value_type; + std::size_t const r = x.row(); + std::size_t const c = x.col(); + matrix< complex_type > X( r, c ); + if ( r == 0 || c == 0 ) return X; + for ( std::size_t i = 0; i != r; ++i ) + for ( std::size_t j = 0; j != c; ++j ) + { + auto const& v = x[i][j]; + if constexpr ( std::is_same_v< typename Mat::value_type, complex_type > ) + X[i][j] = v; + else + X[i][j] = complex_type{ static_cast< real_type >( v ), real_type( 0 ) }; + } + return X; + } - template < Matrix Mat > - auto fftshift( Mat const& x ) - { - auto X = fft( x ); - std::uint_least64_t const R = X.row(); - std::uint_least64_t const C = X.col(); - std::uint_least64_t const row_starter = ( R >> 1 ) + ( R & 1 ); + // ifft (sign +1) divides by R*C + template < typename C > + void fft2_scale( matrix< C >& X, int sign ) noexcept + { + using real_type = typename C::value_type; + if ( sign > 0 ) + { + real_type const n = static_cast< real_type >( X.row() ) * static_cast< real_type >( X.col() ); + for ( auto& v : X ) v /= n; + } + } - for ( std::uint_least64_t index = 0; row_starter + index < R; ++index ) - std::swap_ranges( X.row_begin( index ), X.row_end( index ), X.row_begin( row_starter + index ) ); + // 2-D transform: rows in place, then each column gathered, transformed and scattered, through one plan per + // length (the row plan reused for the columns when r == c) + template < Matrix Mat > + auto fft2( Mat const& x, int sign ) noexcept + { + using complex_type = fft_complex_t< typename Mat::value_type >; + using real_type = typename complex_type::value_type; + std::size_t const r = x.row(); + std::size_t const c = x.col(); + auto X = fft2_load( x ); + if ( r == 0 || c == 0 ) return X; + fft_plan< real_type > row_plan( c, sign ); + for ( std::size_t i = 0; i != r; ++i ) + row_plan( X.data() + i * c ); + if ( r > 1 ) + { + std::optional< fft_plan< real_type > > own; + if ( r != c ) own.emplace( r, sign ); + fft_plan< real_type >& col_plan = r == c ? row_plan : *own; + std::vector< complex_type > col( r ); + for ( std::size_t j = 0; j != c; ++j ) + { + for ( std::size_t i = 0; i != r; ++i ) col[i] = X[i][j]; + col_plan( col.data() ); + for ( std::size_t i = 0; i != r; ++i ) X[i][j] = col[i]; + } + } + fft2_scale( X, sign ); + return X; + } - std::uint_least64_t const col_starter = ( C >> 1 ) + ( C & 1 ); + // S10-R3 reference: the per-row transform fft2 replaced (tables rebuilt for every row and column); tests only + template < Matrix Mat > + auto fft2_reference( Mat const& x, int sign ) noexcept + { + std::size_t const r = x.row(); + std::size_t const c = x.col(); + auto X = fft2_load( x ); + if ( r == 0 || c == 0 ) return X; + for ( std::size_t i = 0; i != r; ++i ) + fft_1d( X.data() + i * c, c, sign ); + if ( r > 1 ) + { + std::vector< typename decltype( X )::value_type > col( r ); + for ( std::size_t j = 0; j != c; ++j ) + { + for ( std::size_t i = 0; i != r; ++i ) col[i] = X[i][j]; + fft_1d( col.data(), r, sign ); + for ( std::size_t i = 0; i != r; ++i ) X[i][j] = col[i]; + } + } + fft2_scale( X, sign ); + return X; + } + + // out[(i + dr) mod R][(j + dc) mod C] = x[i][j], a new matrix of x's type and allocator + template < Matrix Mat > + std::remove_cvref_t< Mat > roll2( Mat const& x, std::ptrdiff_t dr, std::ptrdiff_t dc ) noexcept + { + using out_type = std::remove_cvref_t< Mat >; + std::size_t const r = x.row(); + std::size_t const c = x.col(); + out_type out{ x.get_allocator(), r, c }; + if ( r == 0 || c == 0 ) return out; + std::ptrdiff_t const sr = static_cast< std::ptrdiff_t >( r ); + std::ptrdiff_t const sc = static_cast< std::ptrdiff_t >( c ); + std::size_t const sdr = static_cast< std::size_t >( ( dr % sr + sr ) % sr ); + std::size_t const sdc = static_cast< std::size_t >( ( dc % sc + sc ) % sc ); + for ( std::size_t i = 0; i != r; ++i ) + { + auto const src = x.row_begin( i ); + auto const dst = out.row_begin( ( i + sdr ) % r ); + std::copy( src, src + static_cast< std::ptrdiff_t >( c - sdc ), dst + static_cast< std::ptrdiff_t >( sdc ) ); + std::copy( src + static_cast< std::ptrdiff_t >( c - sdc ), src + sc, dst ); + } + return out; + } + }//namespace matrix_details - for ( std::uint_least64_t index = 0; col_starter + index < C; ++index ) - std::swap_ranges( X.col_begin( index ), X.col_end( index ), X.col_begin( col_starter + index ) ); + // numpy.fft.fft2: X[k][l] = sum x[m][n] exp(-2 pi i (km/R + ln/C)) + template < Matrix Mat > + [[nodiscard]] auto fft( Mat const& x ) noexcept + { + return matrix_details::fft2( x, -1 ); + } - return X; + // numpy.fft.fftshift: roll by (R/2, C/2), no transform + template < Matrix Mat > + [[nodiscard]] auto fftshift( Mat const& x ) noexcept + { + return matrix_details::roll2( x, static_cast< std::ptrdiff_t >( x.row() / 2 ), static_cast< std::ptrdiff_t >( x.col() / 2 ) ); } template < Matrix Mat > - int forward_substitution( Mat const& A, Mat& x, Mat const& b ) + [[nodiscard]] int forward_substitution( Mat const& A, Mat& x, Mat const& b ) noexcept { typedef Mat matrix_type; typedef typename matrix_type::value_type value_type; @@ -6389,147 +8372,274 @@ namespace feng return 0; } - template < Matrix Mat > - std::optional gauss_jordan_elimination( Mat const& m ) noexcept + // S7-R4 (F15, D-026): reduced row echelon form of any m×n A. Columns c = 0..n−1 are visited in turn; the pivot + // of column c is the first entry of largest |·| in rows r..m−1 and counts only when it exceeds + // max(m, n)·ε·‖A‖∞ (MATLAB's rref tolerance); the row index r advances only on a pivot, so zero and dependent + // columns are skipped (their entries in rows r..m−1 are set to 0). Pivot entries are exactly 1 and the other + // entries of a pivot column exactly 0. status is ok, or nonfinite (r = A, no pivots) for a NaN or inf in A. + template < typename T, Allocator A_ = std::allocator< T > > + struct rref_result { - auto const& [row, col] = m.shape(); - better_assert( row < col && "matrix row must be less than colum to execut a Gauss-Jordan Elimination" ); - - auto a = m; - - for ( auto i : matrix_details::range(row) ) - { - auto const p = std::distance( a.col_begin( i ), std::max_element( a.col_begin( i ) + i, a.col_end( i ), [](auto x, auto y){ return std::abs(x) < std::abs(y); } ) ); - - if ( p != static_cast(i) ) - std::swap_ranges( a.row_begin( i ) + i, a.row_end( i ), a.row_begin( p ) + i ); - - auto const factor = a[i][i]; - - if ( std::abs(factor) < 1.0e-10) return {}; - - matrix_details::for_each( a.row_rbegin( i ), a.row_rend( i ) - i, [factor](auto& v){ v /= factor; } ); - - for ( auto j : matrix_details::range(row) ) - { - if ( i == j ) continue; + matrix< T, A_ > r; + std::vector< std::size_t > pivot_columns; + std::size_t rank = 0; + linalg_status status = linalg_status::ok; + }; - auto const ratio = a[j][i]; - std::transform( a.row_rbegin( j ), a.row_rend( j ) - i, a.row_rbegin( i ), a.row_rbegin( j ), [ratio](auto x, auto y){ return x - y*ratio; } ); + template < typename T, Allocator A_ > requires linalg_element< T > + [[nodiscard]] rref_result< T, A_ > row_echelon( matrix< T, A_ > const& m ) noexcept + { + using R = matrix_details::linalg_real_t< T >; + rref_result< T, A_ > ans{ m, {}, 0, linalg_status::ok }; + std::size_t const rows = m.row(); + std::size_t const cols = m.col(); + T* const a = ans.r.data(); + if ( !matrix_details::all_finite( a, rows * cols ) ) { ans.status = linalg_status::nonfinite; return ans; } + R norm_inf{ 0 }; + for ( std::size_t i = 0; i != rows; ++i ) + { + R s{ 0 }; + for ( std::size_t j = 0; j != cols; ++j ) s += std::abs( a[i * cols + j] ); + norm_inf = std::max( norm_inf, s ); + } + R const tol = static_cast< R >( std::max( rows, cols ) ) * std::numeric_limits< R >::epsilon() * norm_inf; + std::size_t r = 0; + for ( std::size_t c = 0; c != cols && r != rows; ++c ) + { + std::size_t p = r; + R best = std::abs( a[r * cols + c] ); + for ( std::size_t i = r + 1; i != rows; ++i ) + if ( R const v = std::abs( a[i * cols + c] ); v > best ) { best = v; p = i; } + if ( !( best > tol ) ) + { + for ( std::size_t i = r; i != rows; ++i ) a[i * cols + c] = T( 0 ); + continue; + } + if ( p != r ) std::swap_ranges( a + r * cols, a + r * cols + cols, a + p * cols ); + T* const pr = a + r * cols; + T const pivot = pr[c]; + for ( std::size_t j = c + 1; j != cols; ++j ) pr[j] /= pivot; + pr[c] = T( 1 ); + for ( std::size_t i = 0; i != rows; ++i ) + { + if ( i == r ) continue; + T* const ri = a + i * cols; + T const f = ri[c]; + if ( f == T( 0 ) ) continue; + for ( std::size_t j = c + 1; j != cols; ++j ) ri[j] -= f * pr[j]; + ri[c] = T( 0 ); } + ans.pivot_columns.push_back( c ); + ++r; } + // columns left once every row holds a pivot need no zeroing: rows r..m−1 do not exist + ans.rank = r; + return ans; + } - return a; + // Legacy spellings (S7-R4): the RREF of any shape, nullopt only for a nonfinite input + template < Matrix Mat > requires linalg_element< typename Mat::value_type > + [[nodiscard]] std::optional gauss_jordan_elimination( Mat const& m ) noexcept + { + auto res = row_echelon( m ); + if ( res.status != linalg_status::ok ) + return {}; + return std::optional{ std::move( res.r ) }; } // matlab alias - template < Matrix Mat > - std::optional rref( Mat const& m ) noexcept + template < Matrix Mat > requires linalg_element< typename Mat::value_type > + [[nodiscard]] std::optional rref( Mat const& m ) noexcept { return gauss_jordan_elimination( m ); } - namespace ifft_private + // numpy.fft.ifft2: exp(+2 pi i (km/R + ln/C)), divided by R*C + template < typename T, Allocator A > + [[nodiscard]] auto ifft( matrix const& x ) noexcept { - template < typename T > - struct add_complex - { - typedef std::complex< T > result_type; - }; - template < typename T > - struct add_complex< std::complex< T >> - { - typedef std::complex< T > result_type; - }; + return matrix_details::fft2( x, +1 ); } - template < typename T, Allocator A > - auto ifft( matrix const& x ) + // numpy.fft.ifftshift: roll by (-(R/2), -(C/2)), no transform + template < Matrix Mat > + [[nodiscard]] auto ifftshift( Mat const& x ) noexcept { - typedef typename ifft_private::add_complex< T >::result_type complex_type; - matrix< complex_type > X( x.row(), x.col() ); - auto make_omege = []( auto k, auto n, auto N ) - { - double const pi = 3.1415926535897932384626433; - double const theta = pi * 2.0 * k * n / static_cast< double >( N ); - return complex_type{ std::cos( theta ), std::sin( theta ) }; - }; - std::uint_least64_t const R = X.row(); - std::uint_least64_t const C = X.col(); - - for ( std::uint_least64_t r = 0; r != R; ++r ) - for ( std::uint_least64_t c = 0; c != C; ++c ) - { - complex_type X_rc{ 0.0, 0.0 }; + return matrix_details::roll2( x, -static_cast< std::ptrdiff_t >( x.row() / 2 ), -static_cast< std::ptrdiff_t >( x.col() / 2 ) ); + } - for ( std::uint_least64_t r_ = 0; r_ != R; ++r_ ) - { - complex_type tmp{ 0.0, 0.0 }; + // S7-R1 (F13, D-026, D-027): the partial-pivoting LU object. Factors of a square A with P·A = L·U; row i of + // P·A is row pivots()[i] of A; status ok, singular (rank < n) or nonfinite. + template < typename T, Allocator A_ = std::allocator< T > > requires linalg_element< T > + class lu_factorization + { + public: + using matrix_type = matrix< T, A_ >; - for ( std::uint_least64_t c_ = 0; c_ != C; ++c_ ) - tmp += x[r][c] * make_omege( c, c_, C ); + lu_factorization() noexcept = default; + explicit lu_factorization( matrix_type const& a ) noexcept : lu_( a ), piv_( a.row() ) + { + better_assert( a.row() == a.col(), "feng::lu_factor: expecting a square matrix, but got ", a.row(), "x", a.col() ); + info_ = matrix_details::lu_in_place( lu_.data(), lu_.row(), piv_.data() ); + } - X_rc += tmp * make_omege( r, r_, R ); - } + [[nodiscard]] linalg_status status() const noexcept { return info_.status; } + [[nodiscard]] bool ok() const noexcept { return info_.status == linalg_status::ok; } + [[nodiscard]] std::size_t rank() const noexcept { return info_.rank; } + [[nodiscard]] std::size_t size() const noexcept { return lu_.row(); } + [[nodiscard]] std::vector< std::size_t > const& pivots() const noexcept { return piv_; } - X[r][c] = X_rc; + // unit lower triangular factor + [[nodiscard]] matrix_type l() const noexcept + { + std::size_t const n = size(); + matrix_type ans( n, n ); + std::fill( ans.begin(), ans.end(), T{ 0 } ); + for ( std::size_t i = 0; i != n; ++i ) + { + for ( std::size_t j = 0; j != i; ++j ) ans[i][j] = lu_[i][j]; + ans[i][i] = T{ 1 }; } + return ans; + } + // upper triangular factor + [[nodiscard]] matrix_type u() const noexcept + { + std::size_t const n = size(); + matrix_type ans( n, n ); + std::fill( ans.begin(), ans.end(), T{ 0 } ); + for ( std::size_t i = 0; i != n; ++i ) + for ( std::size_t j = i; j != n; ++j ) ans[i][j] = lu_[i][j]; + return ans; + } + // permutation matrix: P[i][pivots()[i]] = 1 + [[nodiscard]] matrix_type p() const noexcept + { + std::size_t const n = size(); + matrix_type ans( n, n ); + std::fill( ans.begin(), ans.end(), T{ 0 } ); + for ( std::size_t i = 0; i != n; ++i ) ans[i][piv_[i]] = T{ 1 }; + return ans; + } + // sign(P)·∏u_kk; exactly 0 when rank < n, 1 for 0×0, NaN for a nonfinite input + [[nodiscard]] T det() const noexcept + { + return matrix_details::lu_det( lu_.data(), size(), info_ ); + } + // X with A·X = B (B n×k); singular or nonfinite: that status and an empty value + [[nodiscard]] linalg_result< matrix_type > solve( matrix_type const& b ) const noexcept + { + better_assert( b.row() == size(), "feng::lu_factorization::solve: B must have ", size(), " rows, but got ", b.row(), "x", b.col() ); + if ( !ok() ) return { matrix_type{}, info_.status }; + matrix_type x( size(), b.col() ); + matrix_details::lu_solve( lu_.data(), size(), piv_.data(), b.data(), b.col(), x.data() ); + if ( !matrix_details::all_finite( x.data(), x.size() ) ) return { matrix_type{}, linalg_status::nonfinite }; + return { std::move( x ), linalg_status::ok }; + } + [[nodiscard]] linalg_result< matrix_type > inverse() const noexcept + { + matrix_type id( size(), size() ); + std::fill( id.begin(), id.end(), T{ 0 } ); + for ( std::size_t i = 0; i != size(); ++i ) id[i][i] = T{ 1 }; + return solve( id ); + } - return X; - } - template < Matrix Mat > - auto ifftshift( Mat const& x ) + private: + matrix_type lu_; + std::vector< std::size_t > piv_; + matrix_details::lu_info info_{ linalg_status::ok, 0, 1 }; + }; + + template < typename T, Allocator A > requires linalg_element< T > + [[nodiscard]] lu_factorization< T, A > lu_factor( matrix< T, A > const& a ) noexcept { - auto X = ifft( x ); - std::uint_least64_t const R = X.row(); - std::uint_least64_t const C = X.col(); - std::uint_least64_t const row_starter = ( R >> 1 ) + ( R & 1 ); + return lu_factorization< T, A >( a ); + } - for ( std::uint_least64_t index = 0; row_starter + index < R; ++index ) - std::swap_ranges( X.row_begin( index ), X.row_end( index ), X.row_begin( row_starter + index ) ); + template < typename T, Allocator A > requires linalg_element< T > + [[nodiscard]] linalg_result< matrix< T, A > > solve( matrix< T, A > const& a, matrix< T, A > const& b ) noexcept + { + better_assert( a.row() == a.col() && a.row() == b.row(), "feng::solve: expecting a square A and B with as many rows, but got A ", a.row(), "x", a.col(), " and B ", b.row(), "x", b.col() ); + return lu_factor( a ).solve( b ); + } - std::uint_least64_t const col_starter = ( C >> 1 ) + ( C & 1 ); + template < typename T, Allocator A > requires linalg_element< T > + [[nodiscard]] linalg_result< matrix< T, A > > try_inverse( matrix< T, A > const& a ) noexcept + { + better_assert( a.row() == a.col(), "feng::try_inverse: expecting a square matrix, but got ", a.row(), "x", a.col() ); + return lu_factor( a ).inverse(); + } - for ( std::uint_least64_t index = 0; col_starter + index < C; ++index ) - std::swap_ranges( X.col_begin( index ), X.col_end( index ), X.col_begin( col_starter + index ) ); + // S9-R5 (D-032): the S5 loaders' outcome and file formats, for load_expected; declared in every mode. + enum class io_status { ok, failed }; + enum class io_format { txt, binary, npy }; - return X; +#if defined( __cpp_lib_expected ) + // S9-R5: std::expected views of the status results: the value (or factorization) when ok, else the status. + template < typename V > + [[nodiscard]] std::expected< V, linalg_status > to_expected( linalg_result< V > r ) noexcept + { + if ( !r.ok() ) return std::unexpected( r.status ); + return std::expected< V, linalg_status >{ std::in_place, std::move( r.value ) }; } - template< Matrix Mat > - int lu_decomposition( Mat const& A, Mat& L, Mat& U ) + template < typename F > requires requires( F const& f ) { { f.status() } -> std::same_as< linalg_status >; } + [[nodiscard]] std::expected< F, linalg_status > to_expected( F f ) noexcept { - typedef typename Mat::value_type value_type; - better_assert( A.row() == A.col() && "Square Matrix Requred!" ); - - const std::uint_least64_t n = A.row(); - L.resize( n, n ); - std::fill( L.begin(), L.end(), value_type{0} ); - std::fill( L.diag_begin(), L.diag_end(), value_type( 1 ) ); - - U.resize( n, n ); - std::fill( U.begin(), U.end(), value_type{0} ); + if ( f.status() != linalg_status::ok ) return std::unexpected( f.status() ); + return std::expected< F, linalg_status >{ std::in_place, std::move( f ) }; + } - for ( std::uint_least64_t j = 0; j < n; ++j ) - { - for ( std::uint_least64_t i = 0; i < j + 1; ++i ) - { - U[i][j] = A[i][j] - std::inner_product( L.row_begin( i ), L.row_begin( i ) + i, U.col_begin( j ), value_type() ); - } + template < typename T, Allocator A > + [[nodiscard]] std::expected< rref_result< T, A >, linalg_status > to_expected( rref_result< T, A > r ) noexcept + { + if ( r.status != linalg_status::ok ) return std::unexpected( r.status ); + return std::expected< rref_result< T, A >, linalg_status >{ std::in_place, std::move( r ) }; + } - for ( std::uint_least64_t i = j + 1; i < n; ++i ) - { - L[i][j] = ( A[i][j] - std::inner_product( L.row_begin( i ), L.row_begin( i ) + j, U.col_begin( j ), value_type() ) ) / U[j][j]; + // S9-R5: the matrix load_txt, load_binary or load_npy reads from path, or io_status::failed (the loader has + // printed its one stderr line). + template < typename T, Allocator A = std::allocator< T > > + [[nodiscard]] std::expected< matrix< T, A >, io_status > load_expected( std::string const& path, io_format format ) noexcept + { + matrix< T, A > m; + bool const ok = format == io_format::npy ? m.load_npy( path ) : format == io_format::txt ? m.load_txt( path ) : m.load_binary( path ); + if ( !ok ) return std::unexpected( io_status::failed ); + return std::expected< matrix< T, A >, io_status >{ std::in_place, std::move( m ) }; + } +#endif - if ( std::isinf( L[i][j] ) || std::isnan( L[i][j] ) ) - return 1; - } - } + // out is assigned only when the status is ok (D-026) + template < typename T, Allocator A > requires linalg_element< T > + [[nodiscard]] linalg_status inverse( matrix< T, A > const& a, matrix< T, A >& out ) noexcept + { + better_assert( a.row() == a.col(), "feng::inverse: expecting a square matrix, but got ", a.row(), "x", a.col() ); + return matrix_details::lu_inverse( a, out ); + } + // Legacy two-output LU (MATLAB form): L = Pᵀ·L₀ is a permuted unit lower triangle and A = L·U. Returns 0, or 1 + // with L and U unchanged when the factorization is not ok (singular or nonfinite). + template< Matrix Mat > requires linalg_element< typename Mat::value_type > + [[nodiscard]] int lu_decomposition( Mat const& A, Mat& L, Mat& U ) noexcept + { + typedef typename Mat::value_type value_type; + better_assert( A.row() == A.col(), "lu_decomposition: expecting a square matrix, but got ", A.row(), "x", A.col() ); + auto const f = lu_factor( A ); + if ( !f.ok() ) + return 1; + std::size_t const n = A.row(); + auto const& piv = f.pivots(); + Mat l0 = f.l(); + Mat l( n, n ); + std::fill( l.begin(), l.end(), value_type{ 0 } ); + for ( std::size_t i = 0; i != n; ++i ) // row piv[i] of Pᵀ·L₀ is row i of L₀ + std::copy( l0.row_begin( i ), l0.row_end( i ), l.row_begin( piv[i] ) ); + U = f.u(); + L = std::move( l ); return 0; } - template< Matrix Mat > - std::optional> lu_decomposition( Mat const& A ) + template< Matrix Mat > requires linalg_element< typename Mat::value_type > + [[nodiscard]] std::optional> lu_decomposition( Mat const& A ) noexcept { if ( Mat L, U; lu_decomposition( A, L, U ) == 1 ) return {}; @@ -6537,31 +8647,20 @@ namespace feng return std::make_tuple( L, U ); } - template< Matrix Mat > - int lu_solver( Mat const& A, Mat& x, Mat const& b ) + // Legacy solver for a column b: 0 with x = A⁻¹b, or 1 with x unchanged when A is singular or nonfinite. + template< Matrix Mat > requires linalg_element< typename Mat::value_type > + [[nodiscard]] int lu_solver( Mat const& A, Mat& x, Mat const& b ) noexcept { - typedef Mat matrix_type; - better_assert( A.row() == A.col() ); - better_assert( A.row() == b.row() ); - better_assert( b.col() == 1 ); - matrix_type L, U; - - if ( lu_decomposition( A, L, U ) ) - return 1; - - matrix_type Y; - - if ( forward_substitution( L, Y, b ) ) + better_assert( A.row() == A.col() && A.row() == b.row() && b.col() == 1, "lu_solver: expecting a square A and a column b with as many rows, but got A ", A.row(), "x", A.col(), " and b ", b.row(), "x", b.col() ); + auto r = lu_factor( A ).solve( b ); + if ( !r.ok() ) return 1; - - if ( backward_substitution( U, x, Y ) ) - return 1; - + x = std::move( r.value ); return 0; } - template< Matrix Mat > - std::optional lu_solver( Mat const& A, Mat const& b ) + template< Matrix Mat > requires linalg_element< typename Mat::value_type > + [[nodiscard]] std::optional lu_solver( Mat const& A, Mat const& b ) noexcept { if ( Mat x; lu_solver(A, x, b) == 1 ) return {}; @@ -6569,133 +8668,181 @@ namespace feng return x; } + namespace matrix_details + { + // S8-R1 (F15, D-030): the full 2-D convolution of non-empty A (ra x ca) and B (rb x cb), written directly: + // out[i][j] = sum of A[p][q] * B[i-p][j-q] over the overlap (the kernel reversed), no padded copy. Rows of + // the result are split with parallel; small problems stay on the calling thread. + template< Matrix Mat > + Mat conv_full( Mat const& A, Mat const& B ) noexcept + { + using value_type = typename Mat::value_type; + std::size_t const ra = A.row(), ca = A.col(), rb = B.row(), cb = B.col(); + std::size_t const ro = ra + rb - 1, co = ca + cb - 1; + Mat out{ A.get_allocator(), ro, co }; + value_type const* const a = A.data(); + value_type const* const b = B.data(); + value_type* const o = out.data(); + auto const row = [=]( std::size_t i ) noexcept + { + std::size_t const p0 = i + 1 > rb ? i + 1 - rb : 0; + std::size_t const p1 = std::min( i + 1, ra ); + for ( std::size_t j = 0; j != co; ++j ) + { + std::size_t const q0 = j + 1 > cb ? j + 1 - cb : 0; + std::size_t const q1 = std::min( j + 1, ca ); + value_type acc{ 0 }; + for ( std::size_t p = p0; p != p1; ++p ) + { + value_type const* const ap = a + p * ca; + value_type const* const bp = b + ( i - p ) * cb; + for ( std::size_t q = q0; q != q1; ++q ) + acc += ap[q] * bp[j - q]; + } + o[i * co + j] = acc; + } + }; + std::size_t const work = ro * co * std::min( ra * ca, rb * cb ); + parallel( row, std::size_t{ 0 }, ro, work < ( std::size_t{ 1 } << 16 ) ? std::numeric_limits< unsigned long >::max() : 0UL ); + return out; + } + } + + // S8-R1 (F15): scipy.signal.convolve2d( A, B, "full" ); an operand with a zero dimension gives 0x0 (D-030). template< Matrix Mat > - Mat conv( Mat const& A, Mat const& B ) noexcept + [[nodiscard]] Mat conv( Mat const& A, Mat const& B ) noexcept { if ( ( 0 == A.size() ) || ( 0 == B.size() ) ) - return Mat{0, 0}; - - if ( A.size() > B.size() ) - return conv( B, A ); - - Mat padded_B{ B.row()+2*A.row()-2, B.col()+2*A.col()-2 }; - for ( auto row : matrix_details::range(B.row()) ) - std::copy( B.row_begin(row), B.row_end(row), padded_B.row_begin(row+A.row()-1)+A.col()-1 ); - - Mat ans{ A.row()+B.row()-1, A.col()+B.col()-1 }; - - auto const& product = []( Mat const& a, auto const& b, auto row, auto const col ) noexcept - { - typename Mat::value_type ans{0}; - for ( auto r : matrix_details::range(row) ) - for ( auto c : matrix_details::range(col) ) - ans += a[r][c] * b[r][c]; - return ans; - }; - - auto const& func = [&]( std::uint_least64_t row ) - { - for ( auto col : matrix_details::range(ans.col() )) - { - auto const& view = make_view( padded_B, {row, row+A.row()}, {col, col+A.col()} ); - ans[row][col] = product( A, view, A.row(), A.col() ); - } - }; - - matrix_details::parallel( func, 0UL, ans.row(), 0UL ); - - return ans; + return Mat{ A.get_allocator(), 0, 0 }; + return matrix_details::conv_full( A, B ); } + // S8-R1 (F15): scipy.signal.convolve2d( A, B, mode ) for mode "full", "same" or "valid"; any other mode aborts + // (D-011). same is A's shape cropped from full at ((rb-1)/2, (cb-1)/2); valid needs one operand to contain the + // other in both dimensions and aborts otherwise. Empty operands (D-030): full and valid give 0x0, same gives + // zeros of A's shape. template< Matrix Mat > - Mat conv( Mat const& A, Mat const& B, std::string const& mode ) noexcept + [[nodiscard]] Mat conv( Mat const& A, Mat const& B, std::string const& mode ) noexcept { - auto const& default_conv = conv( A, B ); + bool const full = mode == "full", same = mode == "same", valid = mode == "valid"; + better_assert( full || same || valid, "conv: unknown mode '", mode, "'; expecting \"full\", \"same\" or \"valid\"" ); auto const [ra, ca] = A.shape(); auto const [rb, cb] = B.shape(); - if ( mode == std::string{"same"} ) + if ( ( 0 == A.size() ) || ( 0 == B.size() ) ) + return same ? Mat{ A.get_allocator(), ra, ca } : Mat{ A.get_allocator(), 0, 0 }; + + if ( full ) + return matrix_details::conv_full( A, B ); + + if ( same ) { - better_assert( rb > 1, " For a convolution in 'same' mode, the row of the second matrix is at least 1, but now has ", rb ); - better_assert( rb > 1, " For a convolution in 'same' mode, the column of the second matrix is at least 1, but now has ", cb ); - return { default_conv, { (rb-1)>>1, ra + ((rb-1)>>1) }, { (cb-1)>>1, ca + ((cb-1)>>1) } }; + Mat const f = matrix_details::conv_full( A, B ); + return { f, { (rb-1)/2, ra + (rb-1)/2 }, { (cb-1)/2, ca + (cb-1)/2 } }; } - if ( mode == std::string{"valid"} ) + bool const a_contains_b = ra >= rb && ca >= cb; + bool const b_contains_a = rb >= ra && cb >= ca; + better_assert( a_contains_b || b_contains_a, "conv: 'valid' mode needs one operand to contain the other in both dimensions, but got ", ra, "x", ca, " and ", rb, "x", cb ); + if ( a_contains_b ) { - better_assert( ra >= rb-1, " For a convolution in 'valid' mode, the row of first matrix is supposed to be at least larger than the second matrix's row by 1", - " but the first matrix is with row ", ra, " and the second matrix is with row ", rb ); - better_assert( ca >= cb-1, " For a convolution in 'valid' mode, the column of first matrix is supposed to be at least larger than the second matrix's column by 1", - " but the first matrix is with column ", ca, " and the second matrix is with column ", cb ); - return { default_conv, { rb-1, ra }, { cb-1, ca } }; + Mat const f = matrix_details::conv_full( A, B ); + return { f, { rb-1, ra }, { cb-1, ca } }; } - - return default_conv; + Mat const f = matrix_details::conv_full( B, A ); + return { f, { ra-1, rb }, { ca-1, cb } }; } template< typename ... Args > - auto conv2( Args const& ... args ) + [[nodiscard]] auto conv2( Args const& ... args ) noexcept { return conv( args... ); } - inline std::optional,3>> load_bmp( std::string const& file_path ) + namespace matrix_details { - std::ifstream ifs{file_path, std::ios::binary}; - if ( !ifs ) - { - std::cerr << "load_bmp::Failed to open file " << file_path << "\n"; - return {}; // <- failed to load file - } - - std::vector const file_content{(std::istreambuf_iterator(ifs)), (std::istreambuf_iterator())}; - if( file_content.size() <= 54 ) - { - std::cerr << "load_bmp::file " << file_path << " has contents less than 54 byte.\n"; - return {}; // <- failed for bad file - } - - std::uint_least64_t const col = std::uint_least64_t{file_content[18]} | (std::uint_least64_t{file_content[19]} << 8) | - (std::uint_least64_t{file_content[20]} << 16) | (std::uint_least64_t{file_content[21]} << 24); - std::uint_least64_t const row = std::uint_least64_t{file_content[22]} | (std::uint_least64_t{file_content[23]} << 8) | - (std::uint_least64_t{file_content[24]} << 16) | (std::uint_least64_t{file_content[25]} << 24); - std::uint_least64_t const padding = ( 4 - ( ( col * 3 ) & 0x3 ) ) & 0x3; - - if ( 54+(3*col+padding)*row != file_content.size() ) - { - std::cerr << "load_bmp::file " << file_path << " has insistant contents with its header.\n"; - std::cerr << "Additional Information: col-" << col << ", row-" << row << ", padding-" << padding << ", file size-" << file_content.size() << "\n"; - std::cerr << "Expected size-" << 54+(3*col+padding)*row << ", but the file has " << file_content.size() << " bytes\n"; - return {}; //<- error with the data size - } - - std::array,3> ans; - for ( auto& mat : ans ) mat.resize( row, col ); - auto& [mat_r, mat_g, mat_b] = ans; - - auto pos_itor = file_content.begin()+54; - for ( auto r : matrix_details::range( row ) ) + // S5-R3 (F08): parses a complete BMP file image of `size` bytes into its red, green and blue channels. + // Accepts the 14-byte file header with signature BM, an info header of size 40, 52, 56, 108 or 124 inside + // the file, planes 1, compression 0 (BI_RGB), 24 or 32 bits per pixel (alpha ignored), width > 0, height + // != 0 and != INT32_MIN, a pixel-data offset at or past the end of the info header and offset + + // stride * |height| <= size in checked 64-bit arithmetic (trailing bytes allowed). A positive height is + // stored bottom-up, a negative one top-down. On failure returns an empty optional and sets *why when given. + inline std::optional,3>> parse_bmp( std::uint8_t const* bytes, std::size_t size, char const** why = nullptr ) noexcept { - for ( auto c : matrix_details::range( col ) ) + auto fail = [why]( char const* reason ) noexcept -> std::optional,3>> { - mat_b[row-r-1][c] = *pos_itor++; - mat_g[row-r-1][c] = *pos_itor++; - mat_r[row-r-1][c] = *pos_itor++; + if ( why ) *why = reason; + return std::nullopt; + }; + auto u16 = [bytes]( std::size_t at ) noexcept { return static_cast( bytes[at] ) | ( static_cast( bytes[at+1] ) << 8 ); }; + auto u32 = [u16]( std::size_t at ) noexcept { return u16( at ) | ( u16( at+2 ) << 16 ); }; + + if ( bytes == nullptr || size < 18 ) return fail( "file too short for the BMP headers" ); + if ( bytes[0] != 'B' || bytes[1] != 'M' ) return fail( "bad BMP signature" ); + std::uint32_t const info_size = u32( 14 ); + if ( info_size != 40 && info_size != 52 && info_size != 56 && info_size != 108 && info_size != 124 ) + return fail( "unsupported BMP info header size" ); + std::uint64_t const headers_end = 14 + std::uint64_t{ info_size }; + if ( headers_end > size ) return fail( "BMP info header extends past the end of the file" ); + std::int32_t const width = static_cast( u32( 18 ) ); + std::int32_t const height = static_cast( u32( 22 ) ); + std::uint32_t const planes = u16( 26 ); + std::uint32_t const bpp = u16( 28 ); + std::uint32_t const compression = u32( 30 ); + std::uint64_t const offset = u32( 10 ); + if ( planes != 1 ) return fail( "BMP planes is not 1" ); + if ( compression != 0 ) return fail( "compressed BMP is not supported" ); + if ( bpp != 24 && bpp != 32 ) return fail( "unsupported BMP bit depth" ); + if ( width <= 0 ) return fail( "BMP width is not positive" ); + if ( height == 0 || height == std::numeric_limits::min() ) return fail( "bad BMP height" ); + if ( offset < headers_end ) return fail( "BMP pixel data overlaps the headers" ); + + std::uint64_t const cols = static_cast( width ); + std::uint64_t const rows = height < 0 ? static_cast( -static_cast( height ) ) : static_cast( height ); + std::uint64_t bits = 0, stride = 0, pixels = 0, pixels_end = 0; + if ( !checked_mul( cols, std::uint64_t{ bpp }, bits ) || !checked_add( bits, std::uint64_t{ 31 }, bits ) ) + return fail( "BMP row size overflows" ); + stride = bits / 32 * 4; + if ( !checked_mul( stride, rows, pixels ) || !checked_add( offset, pixels, pixels_end ) ) + return fail( "BMP pixel array size overflows" ); + if ( pixels_end > size ) return fail( "BMP pixel data extends past the end of the file" ); + + std::array,3> ans{ matrix( rows, cols ), matrix( rows, cols ), matrix( rows, cols ) }; + auto& [mat_r, mat_g, mat_b] = ans; + std::uint64_t const step = bpp / 8; + for ( std::uint64_t k = 0; k != rows; ++k ) + { + std::uint64_t const r = height > 0 ? rows - 1 - k : k; + std::uint8_t const* p = bytes + offset + k * stride; + for ( std::uint64_t c = 0; c != cols; ++c, p += step ) + { + mat_b[r][c] = p[0]; + mat_g[r][c] = p[1]; + mat_r[r][c] = p[2]; + } } - std::advance( pos_itor, padding ); + return ans; } + }//namespace matrix_details - return {ans}; + // S5-R3/S5-R4 (D-011): reads the whole file and parses it with matrix_details::parse_bmp; on failure returns an + // empty optional and prints one stderr line naming load_bmp and the reason. Never aborts. + [[nodiscard]] inline std::optional,3>> load_bmp( std::string const& file_path ) noexcept + { + std::vector bytes; + if ( !matrix_details::read_file( file_path.c_str(), bytes, "feng::load_bmp" ) ) + return std::nullopt; + char const* why = "invalid BMP file"; + auto ans = matrix_details::parse_bmp( bytes.data(), bytes.size(), &why ); + if ( !ans ) + std::cerr << "feng::load_bmp -- " << why << ": " << file_path << "\n"; + return ans; } template< Matrix Mat > - auto pooling( Mat const& mat, std::uint_least64_t dim_r, std::uint_least64_t dim_c, std::string const& pooling_action = "mean" ) + [[nodiscard]] auto pooling( Mat const& mat, std::uint_least64_t dim_r, std::uint_least64_t dim_c, std::string const& pooling_action = "mean" ) noexcept { - if (dim_r == 0 || dim_c == 0) return Mat{}; - - if (dim_r==1 && dim_c==1) return mat; - typedef typename Mat::value_type T; std::map< std::string, std::pair > > function_list = @@ -6741,6 +8888,10 @@ namespace feng auto iterator = function_list.find( pooling_action ); better_assert( iterator != function_list.end(), "Error: Unknow pooling action [[", pooling_action, "]], only [mean], [max], [min] supported!" ); + if (dim_r == 0 || dim_c == 0) return Mat{}; + + if (dim_r==1 && dim_c==1) return mat; + auto init_value =(*iterator).second.first; auto const& the_function =(*iterator).second.second; auto const [row, col] = mat.shape(); @@ -6757,19 +8908,22 @@ namespace feng } }; - matrix_details::parallel( make_pooling, 0UL, new_row, 0UL ); + matrix_details::parallel_work( make_pooling, 0UL, new_row, matrix_details::saturating_work( new_col, matrix_details::saturating_work( dim_r, dim_c ) ), matrix_details::callback_grain ); return ans; } template< Matrix Mat > - auto pooling( Mat const& mat, std::uint_least64_t dim, std::string const& pooling_action = "mean" ) + [[nodiscard]] auto pooling( Mat const& mat, std::uint_least64_t dim, std::string const& pooling_action = "mean" ) noexcept { return pooling( mat, dim, dim, pooling_action ); } template< Matrix Mat > - void save_as_bmp( std::string const& file_name, Mat const& red_channel, Mat const& green_channel, Mat const& blue_channel ) + // S5-R4 (D-011): true on success; false with one stderr line on a directory, open, write or close failure, never + // aborting (mismatched channel shapes remain a contract violation). S5-R3: the file loads back through load_bmp + // to the scaled channels in the same orientation. + [[nodiscard]] bool save_as_bmp( std::string const& file_name, Mat const& red_channel, Mat const& green_channel, Mat const& blue_channel ) noexcept { better_assert( red_channel.row()==green_channel.row(), "Row not match for red and green matrix! The row for red is ", red_channel.row(), " but for green is ", green_channel.row() ); better_assert( red_channel.row()==blue_channel.row(), "Row not match for red and blue matrix! The row for red is ", red_channel.row(), " but for blue is ", blue_channel.row() ); @@ -6785,36 +8939,40 @@ namespace feng auto const& [red_mx, red_mn] = std::make_pair( *std::max_element(red_channel.begin(), red_channel.end() ), *std::min_element(red_channel.begin(), red_channel.end() ) ); auto const& [green_mx, green_mn] = std::make_pair( *std::max_element(green_channel.begin(), green_channel.end() ), *std::min_element(green_channel.begin(), green_channel.end() ) ); auto const& [blue_mx, blue_mn] = std::make_pair( *std::max_element(blue_channel.begin(), blue_channel.end() ), *std::min_element(blue_channel.begin(), blue_channel.end() ) ); + // encode_bmp_stream writes its row k as the k-th stored (bottom-up) row, so channel row k holds image row + // row-1-k (the member save_as_bmp flips the same way) for ( auto r : matrix_details::range(row) ) for ( auto c : matrix_details::range(col) ) { - channel_r[r][c] = static_cast( 256.0 * (red_channel[r][c] - red_mn) / (red_mx-red_mn+0.1) ); - channel_g[r][c] = static_cast( 256.0 * (green_channel[r][c] - green_mn) / (green_mx-green_mn+0.1) ); - channel_b[r][c] = static_cast( 256.0 * (blue_channel[r][c] - blue_mn) / (blue_mx-blue_mn+0.1) ); + auto const k = row - 1 - r; + channel_r[k][c] = static_cast( 256.0 * (red_channel[r][c] - red_mn) / (red_mx-red_mn+0.1) ); + channel_g[k][c] = static_cast( 256.0 * (green_channel[r][c] - green_mn) / (green_mx-green_mn+0.1) ); + channel_b[k][c] = static_cast( 256.0 * (blue_channel[r][c] - blue_mn) / (blue_mx-blue_mn+0.1) ); } // encode rgb to bitmap stream auto const& encoding = matrix_details::encode_bmp_stream( channel_r, channel_g, channel_b ); - better_assert( encoding, "Failed to convert the 3 matrix to bmp stream in save_as_bmp function, where the file_name is ", file_name ); + if ( !encoding ) return matrix_details::write_failed( "feng::save_as_bmp", "failed to encode the BMP stream", file_name ); // write file stream std::string new_file_name{ file_name }; std::string const extension{ ".bmp" }; if ( ( new_file_name.size() < 4 ) || ( std::string{ new_file_name.begin() + new_file_name.size() - 4, new_file_name.end() } != extension ) ) new_file_name += extension; - std::ofstream stream( new_file_name.c_str(), std::ios_base::out | std::ios_base::binary ); - better_assert( stream, "Failed to create file ", new_file_name, " when executing save_as_bmp with file_name = ", file_name, " and row = ", row, " col = ", col ); - stream.write( reinterpret_cast((*encoding).data()), (*encoding).size() ); + return matrix_details::write_stream( "feng::save_as_bmp", new_file_name, std::ios_base::out | std::ios_base::binary, + [&encoding]( std::ofstream& stream ) + { stream.write( reinterpret_cast((*encoding).data()), static_cast( (*encoding).size() ) ); } ); } + // S5-R4 (D-011): the member save_as_bmp's result. template< Matrix Mat > - void save_as_bmp( std::string const& file_name, Mat const& mat, std::string const& colormap = std::string{"parula"} ) + [[nodiscard]] bool save_as_bmp( std::string const& file_name, Mat const& mat, std::string const& colormap = std::string{"parula"} ) noexcept { - mat.save_as_bmp( file_name, colormap ); + return mat.save_as_bmp( file_name, colormap ); } template< typename T > - auto meshgrid( T const& x, T const& y ) noexcept + [[nodiscard]] auto meshgrid( T const& x, T const& y ) noexcept { unsigned long const row = static_cast( y ); unsigned long const col = static_cast( x ); @@ -6831,359 +8989,126 @@ namespace feng } }; - matrix_details::parallel( parallel_func, 0UL, row, 0UL ); + matrix_details::parallel_work( parallel_func, 0UL, row, matrix_details::saturating_work( col, 2 ), matrix_details::elementwise_grain ); return std::make_pair( mat_y, mat_x ); } + namespace matrix_details + { + // S9-R6: the one elementwise transform. Returns matrix< R, A rebound to R > of m's shape holding f( x ) for + // each element x of m, assigned (so converted) to R; R is f's result type unless given as transform< R >. + // The allocator is m's, rebound; the elements are written by the parallel for_each (S6). + // Further matrix operands ns (same shape as m; checked by transform_checked) are read alongside: f( x, y... ). + template < typename R = void, typename T, Allocator A, typename F, Matrix... Ns > + auto transform( matrix< T, A > const& m, F const& f, Ns const&... ns ) noexcept + { + using result_type = std::conditional_t< std::is_void_v< R >, std::invoke_result_t< F const&, T const&, typename Ns::value_type const&... >, R >; + using result_alloc = typename std::allocator_traits< A >::template rebind_alloc< result_type >; + matrix< result_type, result_alloc > ans{ result_alloc( m.get_allocator() ), m.row(), m.col() }; + matrix_details::for_each( ans.begin(), ans.end(), m.begin(), ns.begin()..., [&f]( result_type& v, T const& x, auto const&... y ) { v = f( x, y... ); } ); + return ans; + } + + // Two or three matrix operands: the shape check (expect_same_shape, under name), then transform. + template < typename R = void, typename F, Matrix M, Matrix... Ns > + auto transform_checked( char const* name, F const& f, M const& m, Ns const&... ns ) noexcept + { + matrix_details::expect_same_shape( name, m, ns... ); + return matrix_details::transform< R >( m, f, ns... ); + } + } + // // - begin of unary functions // - + // Each keeps its pre-S9 result type: m's type, or the integer type cmath returns (ilogb, lrint, llrint, lround, llround). template< Matrix Mat > - auto abs( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::abs(x); } ); - return ans; - } - + [[nodiscard]] auto abs( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::abs( x ); } ); } template< Matrix Mat > - auto exp( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::exp(x); } ); - return ans; - } - + [[nodiscard]] auto exp( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::exp( x ); } ); } template< Matrix Mat > - auto exp2( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::exp2(x); } ); - return ans; - } - + [[nodiscard]] auto exp2( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::exp2( x ); } ); } template< Matrix Mat > - auto expm1( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::expm1(x); } ); - return ans; - } - + [[nodiscard]] auto expm1( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::expm1( x ); } ); } template< Matrix Mat > - auto log( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::log(x); } ); - return ans; - } - + [[nodiscard]] auto log( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::log( x ); } ); } template< Matrix Mat > - auto log10( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::log10(x); } ); - return ans; - } - + [[nodiscard]] auto log10( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::log10( x ); } ); } template< Matrix Mat > - auto log1p( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::log1p(x); } ); - return ans; - } - + [[nodiscard]] auto log1p( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::log1p( x ); } ); } template< Matrix Mat > - auto log2( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::log2(x); } ); - return ans; - } - + [[nodiscard]] auto log2( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::log2( x ); } ); } template< Matrix Mat > - auto sqrt( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::sqrt(x); } ); - return ans; - } - + [[nodiscard]] auto sqrt( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::sqrt( x ); } ); } template< Matrix Mat > - auto cbrt( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::cbrt(x); } ); - return ans; - } - + [[nodiscard]] auto cbrt( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::cbrt( x ); } ); } template< Matrix Mat > - auto sin( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::sin(x); } ); - return ans; - } - + [[nodiscard]] auto sin( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::sin( x ); } ); } template< Matrix Mat > - auto cos( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::cos(x); } ); - return ans; - } - + [[nodiscard]] auto cos( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::cos( x ); } ); } template< Matrix Mat > - auto tan( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::tan(x); } ); - return ans; - } - + [[nodiscard]] auto tan( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::tan( x ); } ); } template< Matrix Mat > - auto asin( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::asin(x); } ); - return ans; - } - + [[nodiscard]] auto asin( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::asin( x ); } ); } template< Matrix Mat > - auto acos( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::acos(x); } ); - return ans; - } - + [[nodiscard]] auto acos( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::acos( x ); } ); } template< Matrix Mat > - auto atan( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::atan(x); } ); - return ans; - } - + [[nodiscard]] auto atan( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::atan( x ); } ); } template< Matrix Mat > - auto sinh( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::sinh(x); } ); - return ans; - } - + [[nodiscard]] auto sinh( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::sinh( x ); } ); } template< Matrix Mat > - auto cosh( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::cosh(x); } ); - return ans; - } - + [[nodiscard]] auto cosh( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::cosh( x ); } ); } template< Matrix Mat > - auto tanh( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::tanh(x); } ); - return ans; - } - + [[nodiscard]] auto tanh( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::tanh( x ); } ); } template< Matrix Mat > - auto asinh( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::asinh(x); } ); - return ans; - } - + [[nodiscard]] auto asinh( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::asinh( x ); } ); } template< Matrix Mat > - auto acosh( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::acosh(x); } ); - return ans; - } - + [[nodiscard]] auto acosh( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::acosh( x ); } ); } template< Matrix Mat > - auto atanh( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::atanh(x); } ); - return ans; - } - + [[nodiscard]] auto atanh( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::atanh( x ); } ); } template< Matrix Mat > - auto erf( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::erf(x); } ); - return ans; - } - + [[nodiscard]] auto erf( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::erf( x ); } ); } template< Matrix Mat > - auto erfc( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::erfc(x); } ); - return ans; - } - + [[nodiscard]] auto erfc( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::erfc( x ); } ); } template< Matrix Mat > - auto tgamma( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::tgamma(x); } ); - return ans; - } - + [[nodiscard]] auto tgamma( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::tgamma( x ); } ); } template< Matrix Mat > - auto lgamma( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::lgamma(x); } ); - return ans; - } - + [[nodiscard]] auto lgamma( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::lgamma( x ); } ); } template< Matrix Mat > - auto trunc( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::trunc(x); } ); - return ans; - } - + [[nodiscard]] auto trunc( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::trunc( x ); } ); } template< Matrix Mat > - auto round( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::round(x); } ); - return ans; - } - + [[nodiscard]] auto round( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::round( x ); } ); } template< Matrix Mat > - auto ceil( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::ceil(x); } ); - return ans; - } - + [[nodiscard]] auto ceil( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::ceil( x ); } ); } template< Matrix Mat > - auto floor( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::floor(x); } ); - return ans; - } - + [[nodiscard]] auto floor( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::floor( x ); } ); } template< Matrix Mat > - auto rint( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::rint(x); } ); - return ans; - } - + [[nodiscard]] auto rint( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::rint( x ); } ); } template< Matrix Mat > - auto logb( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::logb(x); } ); - return ans; - } - + [[nodiscard]] auto logb( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::logb( x ); } ); } template< Matrix Mat > - auto comp_ellint_1( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::comp_ellint_1(x); } ); - return ans; - } - + [[nodiscard]] auto comp_ellint_1( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::comp_ellint_1( x ); } ); } template< Matrix Mat > - auto comp_ellint_2( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::comp_ellint_2(x); } ); - return ans; - } - + [[nodiscard]] auto comp_ellint_2( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::comp_ellint_2( x ); } ); } template< Matrix Mat > - auto comp_ellint_3( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::comp_ellint_3(x); } ); - return ans; - } - + [[nodiscard]] auto comp_ellint_3( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::comp_ellint_3( x ); } ); } template< Matrix Mat > - auto expint( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::expint(x); } ); - return ans; - } - + [[nodiscard]] auto expint( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::expint( x ); } ); } template< Matrix Mat > - auto riemann_zeta( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::riemann_zeta(x); } ); - return ans; - } - + [[nodiscard]] auto riemann_zeta( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::riemann_zeta( x ); } ); } template< Matrix Mat > - auto nearbyint( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::nearbyint(x); } ); - return ans; - } - + [[nodiscard]] auto nearbyint( Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, []( auto const x ){ return std::nearbyint( x ); } ); } template< Matrix Mat > - auto ilogb( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::ilogb(x); } ); - return ans.template astype(); // <- in cmath, ilogb returns an int - } - + [[nodiscard]] auto ilogb( Mat const& m ) noexcept { return matrix_details::transform< int >( m, []( auto const x ){ return std::ilogb( x ); } ); } template< Matrix Mat > - auto lrint( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::lrint(x); } ); - return ans.template astype(); // <- in cmath, lrint returns a long - } - + [[nodiscard]] auto lrint( Mat const& m ) noexcept { return matrix_details::transform< long >( m, []( auto const x ){ return std::lrint( x ); } ); } template< Matrix Mat > - auto llrint( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::llrint(x); } ); - return ans.template astype(); // <- in cmath, llrint returns a long long - } - + [[nodiscard]] auto llrint( Mat const& m ) noexcept { return matrix_details::transform< long long >( m, []( auto const x ){ return std::llrint( x ); } ); } template< Matrix Mat > - auto lround( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::lround(x); } ); - return ans.template astype(); // <- in cmath, lround returns a long - } - + [[nodiscard]] auto lround( Mat const& m ) noexcept { return matrix_details::transform< long >( m, []( auto const x ){ return std::lround( x ); } ); } template< Matrix Mat > - auto llround( Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::llround(x); } ); - return ans.template astype(); // <- in cmath, llround returns a long long - } - + [[nodiscard]] auto llround( Mat const& m ) noexcept { return matrix_details::transform< long long >( m, []( auto const x ){ return std::llround( x ); } ); } // // - end of unary functions @@ -7191,435 +9116,131 @@ namespace feng // - // - begin of trinary functions + // - begin of trinary and binary functions // + // Each keeps m's type (zeros_like( m ) before S9): the matrix-matrix forms check shapes in transform_checked, + // the scalar forms capture the scalar. template< Matrix Mat > - auto fma( Mat const& m1, Mat const& m2, Mat const& m3 ) - { - auto ans = zeros_like( m1 ); - matrix_details::for_each( m1.begin(), m1.end(), m2.begin(), m3.begin(), ans.begin(), []( auto const x1, auto const x2, auto const x3, auto& v ){ v = std::fma(x1, x2, x3); } ); - return ans; - } - - // - // - end of trinary functions - // - - // - // - begin of binary functions - // + [[nodiscard]] auto fma( Mat const& m1, Mat const& m2, Mat const& m3 ) noexcept { return matrix_details::transform_checked< typename Mat::value_type >( "fma", []( auto const x, auto const y, auto const z ){ return std::fma( x, y, z ); }, m1, m2, m3 ); } template< Matrix Mat, Matrix Nat > - auto ldexp( Mat const& m, Nat const& n ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), n.begin(), ans.begin(), []( auto const xm, auto const xn, auto& v ){ v = std::ldexp(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto ldexp( Mat const& m, std::floating_point auto xn ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::ldexp(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto ldexp( std::floating_point auto xn, Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::ldexp(xn, xm); } ); - return ans; - } + [[nodiscard]] auto ldexp( Mat const& m, Nat const& n ) noexcept { return matrix_details::transform_checked< typename Mat::value_type >( "ldexp", []( auto const x, auto const y ){ return std::ldexp( x, y ); }, m, n ); } + template< Matrix Mat > + [[nodiscard]] auto ldexp( Mat const& m, std::floating_point auto y ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::ldexp( x, y ); } ); } + template< Matrix Mat > + [[nodiscard]] auto ldexp( std::floating_point auto y, Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::ldexp( y, x ); } ); } template< Matrix Mat, Matrix Nat > - auto scalbn( Mat const& m, Nat const& n ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), n.begin(), ans.begin(), []( auto const xm, auto const xn, auto& v ){ v = std::scalbn(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto scalbn( Mat const& m, std::floating_point auto xn ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::scalbn(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto scalbn( std::floating_point auto xn, Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::scalbn(xn, xm); } ); - return ans; - } + [[nodiscard]] auto scalbn( Mat const& m, Nat const& n ) noexcept { return matrix_details::transform_checked< typename Mat::value_type >( "scalbn", []( auto const x, auto const y ){ return std::scalbn( x, y ); }, m, n ); } + template< Matrix Mat > + [[nodiscard]] auto scalbn( Mat const& m, std::floating_point auto y ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::scalbn( x, y ); } ); } + template< Matrix Mat > + [[nodiscard]] auto scalbn( std::floating_point auto y, Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::scalbn( y, x ); } ); } template< Matrix Mat, Matrix Nat > - auto scalbln( Mat const& m, Nat const& n ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), n.begin(), ans.begin(), []( auto const xm, auto const xn, auto& v ){ v = std::scalbln(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto scalbln( Mat const& m, std::floating_point auto xn ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::scalbln(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto scalbln( std::floating_point auto xn, Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::scalbln(xn, xm); } ); - return ans; - } + [[nodiscard]] auto scalbln( Mat const& m, Nat const& n ) noexcept { return matrix_details::transform_checked< typename Mat::value_type >( "scalbln", []( auto const x, auto const y ){ return std::scalbln( x, y ); }, m, n ); } + template< Matrix Mat > + [[nodiscard]] auto scalbln( Mat const& m, std::floating_point auto y ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::scalbln( x, y ); } ); } + template< Matrix Mat > + [[nodiscard]] auto scalbln( std::floating_point auto y, Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::scalbln( y, x ); } ); } template< Matrix Mat, Matrix Nat > - auto pow( Mat const& m, Nat const& n ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), n.begin(), ans.begin(), []( auto const xm, auto const xn, auto& v ){ v = std::pow(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto pow( Mat const& m, std::integral auto xn ) // <- pow accept integral as the second argument - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::pow(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto pow( Mat const& m, std::floating_point auto xn ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::pow(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto pow( std::floating_point auto xn, Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::pow(xn, xm); } ); - return ans; - } + [[nodiscard]] auto pow( Mat const& m, Nat const& n ) noexcept { return matrix_details::transform_checked< typename Mat::value_type >( "pow", []( auto const x, auto const y ){ return std::pow( x, y ); }, m, n ); } + template< Matrix Mat > + [[nodiscard]] auto pow( Mat const& m, std::integral auto y ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::pow( x, y ); } ); } // <- pow accept integral as the second argument + template< Matrix Mat > + [[nodiscard]] auto pow( Mat const& m, std::floating_point auto y ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::pow( x, y ); } ); } + template< Matrix Mat > + [[nodiscard]] auto pow( std::floating_point auto y, Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::pow( y, x ); } ); } template< Matrix Mat, Matrix Nat > - auto hypot( Mat const& m, Nat const& n ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), n.begin(), ans.begin(), []( auto const xm, auto const xn, auto& v ){ v = std::hypot(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto hypot( Mat const& m, std::floating_point auto xn ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::hypot(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto hypot( std::floating_point auto xn, Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::hypot(xn, xm); } ); - return ans; - } + [[nodiscard]] auto hypot( Mat const& m, Nat const& n ) noexcept { return matrix_details::transform_checked< typename Mat::value_type >( "hypot", []( auto const x, auto const y ){ return std::hypot( x, y ); }, m, n ); } + template< Matrix Mat > + [[nodiscard]] auto hypot( Mat const& m, std::floating_point auto y ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::hypot( x, y ); } ); } + template< Matrix Mat > + [[nodiscard]] auto hypot( std::floating_point auto y, Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::hypot( y, x ); } ); } template< Matrix Mat, Matrix Nat > - auto fmod( Mat const& m, Nat const& n ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), n.begin(), ans.begin(), []( auto const xm, auto const xn, auto& v ){ v = std::fmod(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto fmod( Mat const& m, std::floating_point auto xn ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::fmod(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto fmod( std::floating_point auto xn, Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::fmod(xn, xm); } ); - return ans; - } + [[nodiscard]] auto fmod( Mat const& m, Nat const& n ) noexcept { return matrix_details::transform_checked< typename Mat::value_type >( "fmod", []( auto const x, auto const y ){ return std::fmod( x, y ); }, m, n ); } + template< Matrix Mat > + [[nodiscard]] auto fmod( Mat const& m, std::floating_point auto y ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::fmod( x, y ); } ); } + template< Matrix Mat > + [[nodiscard]] auto fmod( std::floating_point auto y, Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::fmod( y, x ); } ); } template< Matrix Mat, Matrix Nat > - auto remainder( Mat const& m, Nat const& n ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), n.begin(), ans.begin(), []( auto const xm, auto const xn, auto& v ){ v = std::remainder(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto remainder( Mat const& m, std::floating_point auto xn ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::remainder(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto remainder( std::floating_point auto xn, Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::remainder(xn, xm); } ); - return ans; - } + [[nodiscard]] auto remainder( Mat const& m, Nat const& n ) noexcept { return matrix_details::transform_checked< typename Mat::value_type >( "remainder", []( auto const x, auto const y ){ return std::remainder( x, y ); }, m, n ); } + template< Matrix Mat > + [[nodiscard]] auto remainder( Mat const& m, std::floating_point auto y ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::remainder( x, y ); } ); } + template< Matrix Mat > + [[nodiscard]] auto remainder( std::floating_point auto y, Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::remainder( y, x ); } ); } template< Matrix Mat, Matrix Nat > - auto copysign( Mat const& m, Nat const& n ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), n.begin(), ans.begin(), []( auto const xm, auto const xn, auto& v ){ v = std::copysign(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto copysign( Mat const& m, std::floating_point auto xn ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::copysign(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto copysign( std::floating_point auto xn, Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::copysign(xn, xm); } ); - return ans; - } + [[nodiscard]] auto copysign( Mat const& m, Nat const& n ) noexcept { return matrix_details::transform_checked< typename Mat::value_type >( "copysign", []( auto const x, auto const y ){ return std::copysign( x, y ); }, m, n ); } + template< Matrix Mat > + [[nodiscard]] auto copysign( Mat const& m, std::floating_point auto y ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::copysign( x, y ); } ); } + template< Matrix Mat > + [[nodiscard]] auto copysign( std::floating_point auto y, Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::copysign( y, x ); } ); } template< Matrix Mat, Matrix Nat > - auto nextafter( Mat const& m, Nat const& n ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), n.begin(), ans.begin(), []( auto const xm, auto const xn, auto& v ){ v = std::nextafter(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto nextafter( Mat const& m, std::floating_point auto xn ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::nextafter(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto nextafter( std::floating_point auto xn, Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::nextafter(xn, xm); } ); - return ans; - } + [[nodiscard]] auto nextafter( Mat const& m, Nat const& n ) noexcept { return matrix_details::transform_checked< typename Mat::value_type >( "nextafter", []( auto const x, auto const y ){ return std::nextafter( x, y ); }, m, n ); } + template< Matrix Mat > + [[nodiscard]] auto nextafter( Mat const& m, std::floating_point auto y ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::nextafter( x, y ); } ); } + template< Matrix Mat > + [[nodiscard]] auto nextafter( std::floating_point auto y, Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::nextafter( y, x ); } ); } template< Matrix Mat, Matrix Nat > - auto fdim( Mat const& m, Nat const& n ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), n.begin(), ans.begin(), []( auto const xm, auto const xn, auto& v ){ v = std::fdim(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto fdim( Mat const& m, std::floating_point auto xn ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::fdim(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto fdim( std::floating_point auto xn, Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::fdim(xn, xm); } ); - return ans; - } + [[nodiscard]] auto fdim( Mat const& m, Nat const& n ) noexcept { return matrix_details::transform_checked< typename Mat::value_type >( "fdim", []( auto const x, auto const y ){ return std::fdim( x, y ); }, m, n ); } + template< Matrix Mat > + [[nodiscard]] auto fdim( Mat const& m, std::floating_point auto y ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::fdim( x, y ); } ); } + template< Matrix Mat > + [[nodiscard]] auto fdim( std::floating_point auto y, Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::fdim( y, x ); } ); } template< Matrix Mat, Matrix Nat > - auto fmax( Mat const& m, Nat const& n ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), n.begin(), ans.begin(), []( auto const xm, auto const xn, auto& v ){ v = std::fmax(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto fmax( Mat const& m, std::floating_point auto xn ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::fmax(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto fmax( std::floating_point auto xn, Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::fmax(xn, xm); } ); - return ans; - } + [[nodiscard]] auto fmax( Mat const& m, Nat const& n ) noexcept { return matrix_details::transform_checked< typename Mat::value_type >( "fmax", []( auto const x, auto const y ){ return std::fmax( x, y ); }, m, n ); } + template< Matrix Mat > + [[nodiscard]] auto fmax( Mat const& m, std::floating_point auto y ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::fmax( x, y ); } ); } + template< Matrix Mat > + [[nodiscard]] auto fmax( std::floating_point auto y, Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::fmax( y, x ); } ); } template< Matrix Mat, Matrix Nat > - auto fmin( Mat const& m, Nat const& n ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), n.begin(), ans.begin(), []( auto const xm, auto const xn, auto& v ){ v = std::fmin(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto fmin( Mat const& m, std::floating_point auto xn ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::fmin(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto fmin( std::floating_point auto xn, Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::fmin(xn, xm); } ); - return ans; - } + [[nodiscard]] auto fmin( Mat const& m, Nat const& n ) noexcept { return matrix_details::transform_checked< typename Mat::value_type >( "fmin", []( auto const x, auto const y ){ return std::fmin( x, y ); }, m, n ); } + template< Matrix Mat > + [[nodiscard]] auto fmin( Mat const& m, std::floating_point auto y ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::fmin( x, y ); } ); } + template< Matrix Mat > + [[nodiscard]] auto fmin( std::floating_point auto y, Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::fmin( y, x ); } ); } template< Matrix Mat, Matrix Nat > - auto atan2( Mat const& m, Nat const& n ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), n.begin(), ans.begin(), []( auto const xm, auto const xn, auto& v ){ v = std::atan2(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto atan2( Mat const& m, std::floating_point auto xn ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::atan2(xm, xn); } ); - return ans; - } - - template< Matrix Mat> - auto atan2( std::floating_point auto xn, Mat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), [xn]( auto const xm, auto& v ){ v = std::atan2(xn, xm); } ); - return ans; - } - + [[nodiscard]] auto atan2( Mat const& m, Nat const& n ) noexcept { return matrix_details::transform_checked< typename Mat::value_type >( "atan2", []( auto const x, auto const y ){ return std::atan2( x, y ); }, m, n ); } + template< Matrix Mat > + [[nodiscard]] auto atan2( Mat const& m, std::floating_point auto y ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::atan2( x, y ); } ); } + template< Matrix Mat > + [[nodiscard]] auto atan2( std::floating_point auto y, Mat const& m ) noexcept { return matrix_details::transform< typename Mat::value_type >( m, [y]( auto const x ){ return std::atan2( y, x ); } ); } // - // - end of binary functions + // - end of trinary and binary functions // - - - // // - begin of Complex Functions // - + // real, imag, abs, arg and norm return the complex value type, rebinding m's allocator; conj, proj and polar keep m's type. template< ComplexMatrix CMat > - auto real( CMat const& cm ) - { - typedef typename CMat::value_type complex_type; - typedef typename complex_type::value_type value_type; - typedef typename CMat::allocator_type Alloc; - matrix:: template rebind_alloc > ans{ cm.row(), cm.col() }; - matrix_details::for_each( cm.begin(), cm.end(), ans.begin(), []( auto const& val, auto& v ){ v = std::real(val); } ); - return ans; - } - + [[nodiscard]] auto real( CMat const& cm ) noexcept { return matrix_details::transform< typename CMat::value_type::value_type >( cm, []( auto const& x ){ return std::real( x ); } ); } template< ComplexMatrix CMat > - auto imag( CMat const& cm ) - { - typedef typename CMat::value_type complex_type; - typedef typename complex_type::value_type value_type; - typedef typename CMat::allocator_type Alloc; - matrix:: template rebind_alloc > ans{ cm.row(), cm.col() }; - matrix_details::for_each( cm.begin(), cm.end(), ans.begin(), []( auto const& val, auto& v ){ v = std::imag(val); } ); - return ans; - } - + [[nodiscard]] auto imag( CMat const& cm ) noexcept { return matrix_details::transform< typename CMat::value_type::value_type >( cm, []( auto const& x ){ return std::imag( x ); } ); } template< ComplexMatrix CMat > - auto abs( CMat const& cm ) - { - typedef typename CMat::value_type complex_type; - typedef typename complex_type::value_type value_type; - typedef typename CMat::allocator_type Alloc; - matrix:: template rebind_alloc > ans{ cm.row(), cm.col() }; - matrix_details::for_each( cm.begin(), cm.end(), ans.begin(), []( auto const& val, auto& v ){ v = std::abs(val); } ); - return ans; - } - + [[nodiscard]] auto abs( CMat const& cm ) noexcept { return matrix_details::transform< typename CMat::value_type::value_type >( cm, []( auto const& x ){ return std::abs( x ); } ); } template< ComplexMatrix CMat > - auto arg( CMat const& cm ) - { - typedef typename CMat::value_type complex_type; - typedef typename complex_type::value_type value_type; - typedef typename CMat::allocator_type Alloc; - matrix:: template rebind_alloc > ans{ cm.row(), cm.col() }; - matrix_details::for_each( cm.begin(), cm.end(), ans.begin(), []( auto const& val, auto& v ){ v = std::arg(val); } ); - return ans; - } - + [[nodiscard]] auto arg( CMat const& cm ) noexcept { return matrix_details::transform< typename CMat::value_type::value_type >( cm, []( auto const& x ){ return std::arg( x ); } ); } template< ComplexMatrix CMat > - auto norm( CMat const& cm ) - { - typedef typename CMat::value_type complex_type; - typedef typename complex_type::value_type value_type; - typedef typename CMat::allocator_type Alloc; - matrix:: template rebind_alloc > ans{ cm.row(), cm.col() }; - matrix_details::for_each( cm.begin(), cm.end(), ans.begin(), []( auto const& val, auto& v ){ v = std::norm(val); } ); - return ans; - } - + [[nodiscard]] auto norm( CMat const& cm ) noexcept { return matrix_details::transform< typename CMat::value_type::value_type >( cm, []( auto const& x ){ return std::norm( x ); } ); } template< ComplexMatrix CMat > - auto conj( CMat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::conj(x); } ); - return ans; - } - + [[nodiscard]] auto conj( CMat const& m ) noexcept { return matrix_details::transform< typename CMat::value_type >( m, []( auto const x ){ return std::conj( x ); } ); } template< ComplexMatrix CMat > - auto proj( CMat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::proj(x); } ); - return ans; - } - + [[nodiscard]] auto proj( CMat const& m ) noexcept { return matrix_details::transform< typename CMat::value_type >( m, []( auto const x ){ return std::proj( x ); } ); } template< ComplexMatrix CMat > - auto polar( CMat const& m ) - { - auto ans = zeros_like( m ); - matrix_details::for_each( m.begin(), m.end(), ans.begin(), []( auto const x, auto& v ){ v = std::polar(x); } ); - return ans; - } + [[nodiscard]] auto polar( CMat const& m ) noexcept { return matrix_details::transform< typename CMat::value_type >( m, []( auto const x ){ return std::polar( x ); } ); } // // - end of Complex Functions @@ -7631,29 +9252,73 @@ namespace feng /// template< Matrix Mat > - auto sum( Mat const& m ) + [[nodiscard]] auto sum( Mat const& m ) noexcept { return matrix_details::reduce( m.begin(), m.end(), typename Mat::value_type{}, []( auto const x, auto const y ){ return x+y; } ); } + // S6-R4 (F15, D-011, D-025): mean, variance and standard_deviation compute in stat_type_t (double for + // integral T, T for floating T, complex for complex T) and return the mean's real type for the spread + // (X for complex); variance divides sum |x - mean|^2 by n - ddof. Empty input or n <= ddof aborts. + namespace matrix_details + { + template< typename T > + struct stat_type { using type = T; using real_type = T; }; + template< std::integral T > + struct stat_type< T > { using type = double; using real_type = double; }; + template< typename X > + struct stat_type< std::complex< X > > { using type = std::complex< X >; using real_type = X; }; + + template< typename T > + using stat_type_t = typename stat_type< T >::type; + template< typename T > + using stat_real_t = typename stat_type< T >::real_type; + + template< Matrix Mat > + stat_real_t< typename Mat::value_type > variance_unchecked( Mat const& m, std::uint_least64_t ddof ) noexcept + { + using S = stat_type_t< typename Mat::value_type >; + using R = stat_real_t< typename Mat::value_type >; + S total{}; + for ( auto const& x : m ) total += static_cast< S >( x ); + S const mu = total / static_cast< R >( m.size() ); + R acc{}; + for ( auto const& x : m ) + { + S const d = static_cast< S >( x ) - mu; + if constexpr ( matrix_private::is_std_complex_v< S > ) + acc += std::norm( d ); + else + acc += d * d; + } + return acc / static_cast< R >( m.size() - ddof ); + } + } + template< Matrix Mat > - auto mean( Mat const& m ) + [[nodiscard]] auto mean( Mat const& m ) noexcept { - return sum( m ) / m.size(); + using S = matrix_details::stat_type_t< typename Mat::value_type >; + using R = matrix_details::stat_real_t< typename Mat::value_type >; + FENG_MATRIX_EXPECTS( m.size() != 0, "feng::mean: empty matrix, shape ", m.row(), "x", m.col() ); + S total{}; + for ( auto const& x : m ) total += static_cast< S >( x ); + return S{ total / static_cast< R >( m.size() ) }; } template< Matrix Mat > - auto variance( Mat const& m ) + [[nodiscard]] auto variance( Mat const& m, std::uint_least64_t ddof = 0 ) noexcept { - return mean( pow( m-mean(m), 2.0 ) ); + FENG_MATRIX_EXPECTS( m.size() != 0 && m.size() > ddof, "feng::variance: needs more elements than ddof, shape ", m.row(), "x", m.col(), ", ddof ", ddof ); + return matrix_details::variance_unchecked( m, ddof ); } template< Matrix Mat > - auto standard_deviation( Mat const& m ) + [[nodiscard]] auto standard_deviation( Mat const& m, std::uint_least64_t ddof = 0 ) noexcept { - if ( m.size() <= 1 ) - return typename Mat::value_type{}; - return std::sqrt( sum( pow( m-mean( m ), 2.0 ) ) / ( m.size() - 1 ) ); + FENG_MATRIX_EXPECTS( m.size() != 0 && m.size() > ddof, "feng::standard_deviation: needs more elements than ddof, shape ", m.row(), "x", m.col(), ", ddof ", ddof ); + using std::sqrt; + return sqrt( matrix_details::variance_unchecked( m, ddof ) ); } /// @@ -7666,9 +9331,9 @@ namespace feng /// template< typename T > - auto clip( T const lower, T const upper ) + [[nodiscard]] auto clip( T const lower, T const upper ) noexcept { - return [lower, upper]( Mat const& m ) + return [lower, upper]( Mat const& m ) noexcept { auto ans{ m }; matrix_details::for_each( ans.begin(), ans.end(), [lower, upper]( auto & v ){ v = v < lower ? lower : v > upper ? upper : v; } ); @@ -7682,6 +9347,17 @@ namespace feng } //namespace feng +// S4-R2 (F06): views do not own their elements, so iterators taken from a temporary view stay valid while the +// owner lives. +namespace std::ranges +{ + template < feng::matrix_element Type, feng::Allocator Alloc > + inline constexpr bool enable_borrowed_range< feng::matrix_view< Type, Alloc > > = true; + + template < feng::matrix_element Type, feng::Allocator Alloc > + inline constexpr bool enable_borrowed_range< feng::mutable_matrix_view< Type, Alloc > > = true; +} + #undef better_assert #endif diff --git a/tests/api/access.cc b/tests/api/access.cc new file mode 100644 index 0000000..5f5000e --- /dev/null +++ b/tests/api/access.cc @@ -0,0 +1,93 @@ +// S1-R4 API family `access`: operator[], item, data, shape, size, row/col/diag/anti-diag/direct iterators, and the +// S9-R4 span, mdspan and submdspan adapters. +// Compiled alone by `tools/check.sh api`; calling each API instantiates its template. +#include "../../matrix.hpp" + +#include +#include + +namespace +{ + template < typename T, typename It > + T walk( It first, It last ) + { + T acc{}; + for ( ; first != last; ++first ) + acc += *first; + return acc; + } + +// Every begin/end flavour of one iterator family, on the mutable and the const matrix, with an index argument. +#define API_ACCESS_FAMILY( prefix, index ) \ + acc += walk( m.prefix##_begin( index ), m.prefix##_end( index ) ); \ + acc += walk( cm.prefix##_begin( index ), cm.prefix##_end( index ) ); \ + acc += walk( cm.prefix##_cbegin( index ), cm.prefix##_cend( index ) ); \ + acc += walk( m.prefix##_rbegin( index ), m.prefix##_rend( index ) ); \ + acc += walk( cm.prefix##_rbegin( index ), cm.prefix##_rend( index ) ); \ + acc += walk( cm.prefix##_crbegin( index ), cm.prefix##_crend( index ) ); + + template < typename T > + T exercise_access() + { + feng::matrix m{ 4, 5, T{ 1 } }; + feng::matrix const& cm = m; + T acc{}; + + // element access + m[1][2] = T{ 2 }; + acc += cm[1][2]; + m( 2, 3 ) = T{ 3 }; + acc += cm( 2, 3 ); + acc += feng::matrix{ 1, 1, T{ 4 } }.item(); + *m.data() = T{ 5 }; + acc += *cm.data(); + + // shape and size + auto const [r, c] = cm.shape(); + acc += static_cast( static_cast( r + c + cm.row() + cm.col() + cm.size() ) ); + + // direct iterators + acc += walk( m.begin(), m.end() ); + acc += walk( cm.begin(), cm.end() ); + acc += walk( cm.cbegin(), cm.cend() ); + acc += walk( m.rbegin(), m.rend() ); + acc += walk( cm.rbegin(), cm.rend() ); + acc += walk( cm.crbegin(), cm.crend() ); + + // row, column, diagonal and anti-diagonal iterators + API_ACCESS_FAMILY( row, 1 ) + API_ACCESS_FAMILY( col, 2 ) + API_ACCESS_FAMILY( diag, 1 ) + API_ACCESS_FAMILY( diag, -1 ) + API_ACCESS_FAMILY( upper_diag, 1 ) + API_ACCESS_FAMILY( lower_diag, 1 ) + API_ACCESS_FAMILY( anti_diag, 1 ) + API_ACCESS_FAMILY( anti_diag, -1 ) + API_ACCESS_FAMILY( upper_anti_diag, 1 ) + API_ACCESS_FAMILY( lower_anti_diag, 1 ) + + // the default-index overloads + acc += walk( m.row_begin(), m.row_end() ); + acc += walk( m.diag_begin(), m.diag_end() ); + acc += walk( m.anti_diag_begin(), m.anti_diag_end() ); + + // S9-R4: span adapters, and the mdspan / submdspan adapters where the library has them + acc += walk( feng::as_span( m ).begin(), feng::as_span( m ).end() ); + acc += walk( feng::as_span( cm ).begin(), feng::as_span( cm ).end() ); + acc += feng::row_span( m, 1 )[0] + feng::row_span( cm, 2 )[1]; +#if defined( __cpp_lib_mdspan ) + acc += feng::to_mdspan( m )[1, 2] + feng::to_mdspan( cm )[2, 3]; +#endif +#if defined( __cpp_lib_submdspan ) + acc += feng::submdspan( m, { 1, 3 }, { 0, 2 } )[0, 1] + feng::submdspan( cm, { 0, 1 }, { 0, 5 } )[0, 4]; +#endif + return acc; + } + +#undef API_ACCESS_FAMILY +} // namespace + +double api_access_double() { return exercise_access(); } +float api_access_float() { return exercise_access(); } +std::complex api_access_complex() { return exercise_access>(); } +int api_access_int() { return exercise_access(); } diff --git a/tests/api/allocator.cc b/tests/api/allocator.cc new file mode 100644 index 0000000..25ee293 --- /dev/null +++ b/tests/api/allocator.cc @@ -0,0 +1,65 @@ +// S1-R4 API family `allocator`: matrix with a non-default stateless allocator, get_allocator. Compiled alone by +// `tools/check.sh api` and never run. +#include "../../matrix.hpp" + +#include +#include +#include +#include + +namespace +{ + // A small stateless allocator that is not std::allocator: it forwards to the global operator new/delete. + template < typename T > + struct api_allocator + { + using value_type = T; + + api_allocator() noexcept = default; + template < typename U > + api_allocator( api_allocator const& ) noexcept + { + } + + T* allocate( std::size_t n ) { return static_cast( ::operator new( n * sizeof( T ) ) ); } + void deallocate( T* p, std::size_t ) noexcept { ::operator delete( p ); } + + template < typename U > + bool operator==( api_allocator const& ) const noexcept + { + return true; + } + }; + + template < typename T > + T exercise_allocator() + { + using alloc_type = api_allocator; + using matrix_type = feng::matrix; + alloc_type const alloc{}; + T acc{}; + + matrix_type a{ alloc, 3, 3, T{ 1 } }; + matrix_type b{ 3, 3, T{ 2 } }; + matrix_type c{ a }; + matrix_type d{ alloc, 3, 3 }; + d = b; + c.swap( d ); + c.resize( 2, 2 ); + + alloc_type const got = a.get_allocator(); + acc += ( got == alloc ) ? T{ 1 } : T{}; + + matrix_type const s = a + b; + matrix_type const p = a * b; + matrix_type const t = a.transpose(); + acc += s[0][0] + p[0][0] + t[0][0] + c[0][0]; + acc += feng::sum( a ); + return acc; + } +} // namespace + +double api_allocator_double() { return exercise_allocator(); } +float api_allocator_float() { return exercise_allocator(); } +std::complex api_allocator_complex() { return exercise_allocator>(); } +int api_allocator_int() { return exercise_allocator(); } diff --git a/tests/api/arithmetic.cc b/tests/api/arithmetic.cc new file mode 100644 index 0000000..dc50e6f --- /dev/null +++ b/tests/api/arithmetic.cc @@ -0,0 +1,93 @@ +// S1-R4 API family `arithmetic`: + - * / with scalars, matrices, valarray and vector; compound and prefix +// operators; `^` power. Compiled alone by `tools/check.sh api`; calling each operator instantiates its template. +#include "../../matrix.hpp" + +#include +#include +#include + +namespace +{ + template < typename T > + T sink( feng::matrix const& m ) + { + return m.size() ? m[0][0] : T{}; + } + + template < typename T > + T exercise_arithmetic() + { + feng::matrix a{ 3, 3, T{ 1 } }; + feng::matrix b{ 3, 3, T{ 2 } }; + T const s{ 3 }; + T acc{}; + + // matrix with matrix + acc += sink( a + b ); + acc += sink( a - b ); + acc += sink( a * b ); + + // matrix with scalar, both sides + acc += sink( a + s ); + acc += sink( s + a ); + acc += sink( a - s ); + acc += sink( s - a ); + acc += sink( a * s ); + acc += sink( s * a ); + acc += sink( a / s ); + + // valarray, vector and pointer products + std::valarray const va( T{ 1 }, 3 ); + std::vector const ve( 3, T{ 1 } ); + acc += sink( a * va ); + acc += sink( va * a ); + acc += sink( a * ve ); + acc += sink( ve * a ); + acc += sink( a * ve.data() ); + acc += sink( ve.data() * a ); + + // compound operators + feng::matrix c{ a }; + c += b; + c += s; + c -= b; + c -= s; + c *= b; + c *= s; + c /= s; + acc += sink( c ); + + // prefix operators + acc += sink( -a ); + acc += sink( +a ); + + // power + acc += sink( a ^ 0 ); + acc += sink( a ^ 5 ); + return acc; + } + + template < typename T > + T exercise_division() + { + // matrix / matrix and scalar / matrix go through the inverse, so floating-point and complex types only. + feng::matrix a = feng::eye( 3, 3 ); + feng::matrix b = feng::eye( 3, 3 ); + T const s{ 2 }; + T acc{}; + acc += sink( a / b ); + acc += sink( s / a ); + feng::matrix c{ a }; + c /= b; + acc += sink( c ); + return acc; + } +} // namespace + +double api_arithmetic_double() { return exercise_arithmetic() + exercise_division(); } +float api_arithmetic_float() { return exercise_arithmetic() + exercise_division(); } +std::complex api_arithmetic_complex() +{ + return exercise_arithmetic>() + exercise_division>(); +} +int api_arithmetic_int() { return exercise_arithmetic(); } diff --git a/tests/api/construction.cc b/tests/api/construction.cc new file mode 100644 index 0000000..188d94e --- /dev/null +++ b/tests/api/construction.cc @@ -0,0 +1,129 @@ +// S1-R4 API family `construction`: constructors, zeros, ones, eye, arange, linspace, magic, hilbert, toeplitz, diag, +// make_diag, blkdiag, meshgrid, *_like, rand/random/randn. Compiled alone by `tools/check.sh api`; calling each API +// instantiates its template. +#include "../../matrix.hpp" + +#include +#include +#include +#include +#include +#include +#include + +namespace +{ + template < typename M > + auto sink( M const& m ) -> typename M::value_type + { + return m.size() ? m[0][0] : typename M::value_type{}; + } + + template < typename T > + T exercise_constructors() + { + using matrix_type = feng::matrix; + T acc{}; + + // constructors: default, sized, filled, initializer list, allocator, copy, move, converting + matrix_type const a; + matrix_type const b{ 2, 3 }; + matrix_type const c{ 2, 3, T{ 1 } }; + matrix_type const d{ 2, 2, { T{ 1 }, T{ 2 }, T{ 3 }, T{ 4 } } }; + matrix_type const e{ std::allocator{}, 2, 2, T{ 1 } }; + matrix_type f{ c }; + matrix_type const g{ std::move( f ) }; + feng::matrix const h{ feng::matrix{ 2, 2, 1 } }; + matrix_type i; + i = c; + i = matrix_type{ 2, 2 }; + i = T{ 1 }; + acc += sink( a ) + sink( b ) + sink( c ) + sink( d ) + sink( e ) + sink( g ) + sink( i ); + acc += static_cast( static_cast( h.size() ) ); + + // zeros, ones, empty, eye + acc += sink( feng::zeros( 2, 3 ) ); + acc += sink( feng::zeros( 2 ) ); + acc += sink( feng::zeros( std::allocator{}, 2, 3 ) ); + acc += sink( feng::zeros( std::allocator{}, 2 ) ); + acc += sink( feng::ones( 2, 3 ) ); + acc += sink( feng::ones( 2 ) ); + acc += sink( feng::ones( std::allocator{}, 2, 3 ) ); + acc += sink( feng::ones( std::allocator{}, 2 ) ); + acc += sink( feng::empty( 2, 3 ) ); + acc += sink( feng::empty( std::allocator{}, 2UL, 3UL ) ); + acc += sink( feng::eye( 2, 3 ) ); + acc += sink( feng::eye( 3 ) ); + acc += sink( feng::eye( c ) ); + + // *_like + acc += sink( feng::zeros_like( c ) ); + acc += sink( feng::ones_like( c ) ); + + // diag, make_diag, blkdiag + acc += sink( feng::diag( d ) ); + acc += sink( feng::diag( d, 1 ) ); + acc += sink( feng::diag( std::vector( 3, T{ 1 } ) ) ); + acc += sink( feng::diag( std::deque( 3, T{ 1 } ), -1 ) ); + acc += sink( feng::diag( std::valarray( T{ 1 }, 3 ) ) ); + acc += sink( feng::make_diag( T{ 1 }, T{ 2 }, T{ 3 } ) ); + acc += sink( feng::blkdiag( c, d ) ); + acc += sink( feng::blkdiag( c, d, e ) ); + acc += sink( feng::blk_diag( c, d ) ); + acc += sink( feng::block_diag( c, d ) ); + + // toeplitz and hilbert through a prototype matrix + std::vector const col{ T{ 1 }, T{ 2 }, T{ 3 } }; + acc += sink( feng::toeplitz( col.begin(), col.end() ) ); + acc += sink( feng::toeplitz( col.begin(), col.end(), col.begin(), col.end() ) ); + acc += sink( feng::repmat( d, 2, 2 ) ); + return acc; + } + + template < typename T > + T exercise_real() + { + // APIs restricted to ordered (real or integer) element types + T acc{}; + std::set const s{ T{ 1 }, T{ 2 } }; + std::multiset const ms{ T{ 1 }, T{ 1 } }; + acc += sink( feng::diag( s ) ); + acc += sink( feng::diag( ms ) ); + acc += sink( feng::arange( 0, 6, 2 ) ); + acc += sink( feng::arange( 5 ) ); + acc += sink( feng::linspace( T{ 0 }, T{ 10 }, 5 ) ); + acc += sink( feng::linspace( T{ 0 }, T{ 10 }, 5, false ) ); + acc += sink( feng::hilbert( 3 ) ); + acc += sink( feng::hilb( 3 ) ); + acc += sink( feng::hilbert( 3, feng::matrix{} ) ); + acc += sink( feng::hilb( 3, feng::matrix{} ) ); + auto const [mx, my] = feng::meshgrid( T{ 3 }, T{ 2 } ); + acc += sink( mx ) + sink( my ); + return acc; + } + + template < typename T > + T exercise_random() + { + feng::matrix const m{ 2, 2, T{ 1 } }; + T acc{}; + acc += sink( feng::rand( 2, 3 ) ); + acc += sink( feng::rand( 2, 3, 42U ) ); + acc += sink( feng::rand( 2 ) ); + acc += sink( feng::random( 2, 3 ) ); + acc += sink( feng::random( 2 ) ); + acc += sink( feng::rand_like( m ) ); + acc += sink( feng::random_like( m ) ); + acc += sink( feng::randn_like( m ) ); + return acc; + } +} // namespace + +double api_construction_double() +{ + return exercise_constructors() + exercise_real() + exercise_random() + + static_cast( feng::magic( 4 )[0][0] ); +} +float api_construction_float() { return exercise_constructors() + exercise_real() + exercise_random(); } +std::complex api_construction_complex() { return exercise_constructors>(); } +int api_construction_int() { return exercise_constructors() + exercise_real(); } diff --git a/tests/api/elementwise.cc b/tests/api/elementwise.cc new file mode 100644 index 0000000..8fac969 --- /dev/null +++ b/tests/api/elementwise.cc @@ -0,0 +1,152 @@ +// S1-R4 API family `elementwise`: the unary and binary `` wrappers, abs, real, imag, conj, arg, polar, clip, +// astype, apply. Compiled alone by `tools/check.sh api`; calling each API instantiates its template. +#include "../../matrix.hpp" + +#include + +namespace +{ + template < typename M > + auto sink( M const& m ) -> typename M::value_type + { + return m.size() ? m[0][0] : typename M::value_type{}; + } + + template < typename T > + double exercise_unary() + { + feng::matrix const m{ 3, 3, T{ 0.5 } }; + double acc{}; + acc += sink( feng::abs( m ) ); + acc += sink( feng::exp( m ) ); + acc += sink( feng::exp2( m ) ); + acc += sink( feng::expm1( m ) ); + acc += sink( feng::log( m ) ); + acc += sink( feng::log10( m ) ); + acc += sink( feng::log1p( m ) ); + acc += sink( feng::log2( m ) ); + acc += sink( feng::sqrt( m ) ); + acc += sink( feng::cbrt( m ) ); + acc += sink( feng::sin( m ) ); + acc += sink( feng::cos( m ) ); + acc += sink( feng::tan( m ) ); + acc += sink( feng::asin( m ) ); + acc += sink( feng::acos( m ) ); + acc += sink( feng::atan( m ) ); + acc += sink( feng::sinh( m ) ); + acc += sink( feng::cosh( m ) ); + acc += sink( feng::tanh( m ) ); + acc += sink( feng::asinh( m ) ); + acc += sink( feng::acosh( m ) ); + acc += sink( feng::atanh( m ) ); + acc += sink( feng::erf( m ) ); + acc += sink( feng::erfc( m ) ); + acc += sink( feng::tgamma( m ) ); + acc += sink( feng::lgamma( m ) ); + acc += sink( feng::trunc( m ) ); + acc += sink( feng::round( m ) ); + acc += sink( feng::ceil( m ) ); + acc += sink( feng::floor( m ) ); + acc += sink( feng::rint( m ) ); + acc += sink( feng::logb( m ) ); + acc += sink( feng::comp_ellint_1( m ) ); + acc += sink( feng::comp_ellint_2( m ) ); + acc += sink( feng::expint( m ) ); + acc += sink( feng::riemann_zeta( m ) ); + acc += sink( feng::nearbyint( m ) ); + acc += sink( feng::ilogb( m ) ); + acc += static_cast( sink( feng::lrint( m ) ) ); + acc += static_cast( sink( feng::llrint( m ) ) ); + acc += static_cast( sink( feng::lround( m ) ) ); + acc += static_cast( sink( feng::llround( m ) ) ); + acc += sink( feng::fma( m, m, m ) ); + return acc; + } + + template < typename T > + double exercise_binary() + { + feng::matrix const m{ 3, 3, T{ 0.5 } }; + feng::matrix const n{ 3, 3, T{ 2 } }; + double const x = 2.0; + double acc{}; + acc += sink( feng::ldexp( m, feng::matrix{ 3, 3, 1 } ) ); + acc += sink( feng::ldexp( m, x ) ); + acc += sink( feng::scalbn( m, feng::matrix{ 3, 3, 1 } ) ); + acc += sink( feng::scalbln( m, feng::matrix{ 3, 3, 1L } ) ); + acc += sink( feng::pow( m, n ) ); + acc += sink( feng::pow( m, 2 ) ); + acc += sink( feng::pow( m, x ) ); + acc += sink( feng::pow( x, m ) ); + acc += sink( feng::hypot( m, n ) ); + acc += sink( feng::hypot( m, x ) ); + acc += sink( feng::hypot( x, m ) ); + acc += sink( feng::fmod( m, n ) ); + acc += sink( feng::fmod( m, x ) ); + acc += sink( feng::fmod( x, m ) ); + acc += sink( feng::remainder( m, n ) ); + acc += sink( feng::remainder( m, x ) ); + acc += sink( feng::remainder( x, m ) ); + acc += sink( feng::copysign( m, n ) ); + acc += sink( feng::copysign( m, x ) ); + acc += sink( feng::copysign( x, m ) ); + acc += sink( feng::nextafter( m, n ) ); + acc += sink( feng::nextafter( m, x ) ); + acc += sink( feng::nextafter( x, m ) ); + acc += sink( feng::fdim( m, n ) ); + acc += sink( feng::fdim( m, x ) ); + acc += sink( feng::fdim( x, m ) ); + acc += sink( feng::fmax( m, n ) ); + acc += sink( feng::fmax( m, x ) ); + acc += sink( feng::fmax( x, m ) ); + acc += sink( feng::fmin( m, n ) ); + acc += sink( feng::fmin( m, x ) ); + acc += sink( feng::fmin( x, m ) ); + acc += sink( feng::atan2( m, n ) ); + acc += sink( feng::atan2( m, x ) ); + acc += sink( feng::atan2( x, m ) ); + return acc; + } + + template < typename T > + T exercise_misc() + { + feng::matrix m{ 3, 3, T{ 1 } }; + T acc{}; + acc += sink( feng::abs( m ) ); + acc += sink( feng::clip( T{ 0 }, T{ 2 } )( m ) ); + acc += static_cast( sink( m.template astype() ) ); + acc += static_cast( sink( m.template astype() ) ); + m.apply( []( T& v ) { v += T{ 1 }; } ); + m.elementwise_apply( []( T& v ) { v += T{ 1 }; } ); + m.map( []( T& v ) { v += T{ 1 }; } ); + acc += sink( m ); + return acc; + } + + double exercise_complex() + { + using C = std::complex; + feng::matrix m{ 3, 3, C{ 1.0, 2.0 } }; + double acc{}; + acc += sink( feng::real( m ) ); + acc += sink( feng::imag( m ) ); + acc += sink( feng::abs( m ) ); + acc += sink( feng::arg( m ) ); + acc += sink( feng::norm( m ) ); + acc += std::real( sink( feng::conj( m ) ) ); + acc += std::real( sink( feng::proj( m ) ) ); + acc += std::real( sink( m.astype>() ) ); + m.apply( []( C& v ) { v *= 2.0; } ); + acc += std::real( sink( m ) ); + return acc; + } +} // namespace + +double api_elementwise_double() { return exercise_unary() + exercise_binary() + exercise_misc(); } +float api_elementwise_float() +{ + return static_cast( exercise_unary() + exercise_binary() ) + exercise_misc(); +} +double api_elementwise_complex() { return exercise_complex(); } +int api_elementwise_int() { return exercise_misc(); } diff --git a/tests/api/image.cc b/tests/api/image.cc new file mode 100644 index 0000000..cd0df86 --- /dev/null +++ b/tests/api/image.cc @@ -0,0 +1,57 @@ +// S1-R4 API family `image`: save_as_bmp, save_as_png, save_as_pgm, load_bmp, plot, colour maps. Compiled alone +// by `tools/check.sh api` and never run; file names point under build/tmp. +#include "../../matrix.hpp" + +#include +#include + +namespace +{ + template < typename T > + int exercise_image() + { + feng::matrix a{ 4, 4, T{ 1 } }; + std::string const base{ "build/tmp/api_image" }; + int acc = 0; + + // member writers, with the default and a named colour map + acc += a.save_as_bmp( base + "_m.bmp" ) ? 1 : 0; + acc += a.save_as_bmp( base + "_jet.bmp", "jet" ) ? 1 : 0; + acc += a.save_as_bmp( ( base + "_c.bmp" ).c_str() ) ? 1 : 0; + acc += a.save_as_png( base + "_m.png" ) ? 1 : 0; + acc += a.save_as_png( base + "_gray.png", "gray" ) ? 1 : 0; + acc += a.save_as_pgm( base + "_m.pgm" ) ? 1 : 0; + acc += a.save_as_pgm( ( base + "_c.pgm" ).c_str() ) ? 1 : 0; + acc += a.plot( base + "_plot.bmp" ) ? 1 : 0; + acc += a.plot( base + "_hot.bmp", "hot" ) ? 1 : 0; + + // free writers: one channel with a colour map, and three channels; each returns bool (S5-R4) + acc += feng::save_as_bmp( base + "_f.bmp", a ) ? 1 : 0; + acc += feng::save_as_bmp( base + "_fj.bmp", a, "jet" ) ? 1 : 0; + bool const rgb_ok = feng::save_as_bmp( base + "_rgb.bmp", a, a, a ); + acc += rgb_ok ? 1 : 0; + return acc; + } + + int exercise_load_and_maps() + { + int acc = 0; + auto const loaded = feng::load_bmp( "build/tmp/api_image_rgb.bmp" ); + if ( loaded ) + acc += static_cast( ( *loaded )[0].size() ); + + // colour maps: the named table and a custom map + auto const& maps = feng::matrix_details::bmp_details::color_maps; + auto const [r, g, b] = maps.at( "parula" )( 0.5 ); + acc += r + g + b; + auto const custom = feng::matrix_details::bmp_details::make_color_map( { 0.0, 1.0 }, { { 0, 0, 0 }, { 255, 255, 255 } } ); + auto const [cr, cg, cb] = custom( 0.25 ); + acc += cr + cg + cb; + return acc; + } +} // namespace + +int api_image_double() { return exercise_image() + exercise_load_and_maps(); } +int api_image_float() { return exercise_image(); } +int api_image_int() { return exercise_image(); } +int api_image_uint8() { return exercise_image(); } diff --git a/tests/api/io.cc b/tests/api/io.cc new file mode 100644 index 0000000..a7f84c5 --- /dev/null +++ b/tests/api/io.cc @@ -0,0 +1,75 @@ +// S1-R4 API family `io`: save_as_txt, load_txt, save_as_binary, load_binary, save_as_npy, load_npy, stream +// operators, disp/display. Compiled alone by `tools/check.sh api` and never run; file names point under build/tmp. +#include "../../matrix.hpp" + +#include +#include +#include + +namespace +{ + template < typename T > + T sink( feng::matrix const& m ) + { + return m.size() ? m[0][0] : T{}; + } + + template < typename T > + T exercise_files() + { + feng::matrix a{ 2, 3, T{ 1 } }; + feng::matrix b; + std::string const txt{ "build/tmp/api_io.txt" }; + std::string const bin{ "build/tmp/api_io.bin" }; + bool ok = true; + ok = a.save_as_txt( txt ) && ok; + ok = a.save_as_txt( txt.c_str() ) && ok; + ok = b.load_txt( txt ) && ok; + ok = b.load_txt( txt.c_str() ) && ok; + ok = a.save_as_binary( bin ) && ok; + ok = a.save_as_binary( bin.c_str() ) && ok; + ok = b.load_binary( bin ) && ok; + ok = b.load_binary( bin.c_str() ) && ok; + return ok ? sink( b ) : T{}; + } + + template < typename T > + T exercise_npy() + { + feng::matrix a{ 2, 3, T{ 1 } }; + feng::matrix b; + std::string const npy{ "build/tmp/api_io.npy" }; + bool ok = a.save_as_npy( npy ); + ok = a.save_as_npy( npy.c_str() ) && ok; + ok = b.load_npy( npy ) && ok; + ok = b.load_npy( npy.c_str() ) && ok; + return ok ? sink( b ) : T{}; + } + + template < typename T > + T exercise_streams() + { + feng::matrix a{ 2, 2, T{ 1 } }; + std::ostringstream os; + os << a; + feng::matrix b{ 2, 2 }; + std::istringstream is{ os.str() }; + is >> b; + feng::display( a ); + feng::disp( a ); + return sink( b ); + } + + template < typename T > + T exercise_io() + { + return exercise_files() + exercise_streams(); + } +} // namespace + +double api_io_double() { return exercise_io() + exercise_npy(); } +float api_io_float() { return exercise_io() + exercise_npy(); } +std::complex api_io_complex() { + return exercise_io>() + exercise_npy>(); +} +int api_io_int() { return exercise_io() + exercise_npy(); } diff --git a/tests/api/linalg.cc b/tests/api/linalg.cc new file mode 100644 index 0000000..2bbe9d2 --- /dev/null +++ b/tests/api/linalg.cc @@ -0,0 +1,206 @@ +// S1-R4 API family `linalg`: det, inverse/inv, lu_factor (S7-R1), svd_factor and status pinverse (S7-R3), solve, try_inverse, lu_decomposition, lu_solver, svd, pinv/pinverse, +// cholesky_factor, row_echelon and status expm (S7-R4, S7-R5), cholesky_decomposition, rref, gauss_jordan_elimination, householder, forward/backward_substitution, expm, eigen_*, +// conjugate gradient solvers, the S9-R5 to_expected / load_expected adapters. Compiled alone by `tools/check.sh api`; calling each API instantiates its template. +#include "../../matrix.hpp" + +#include +#include +#include + +namespace +{ + template < typename M > + auto sink( M const& m ) -> typename M::value_type + { + return m.size() ? m[0][0] : typename M::value_type{}; + } + + template < typename T > + T exercise_inverse() + { + // determinant and inverse: real and complex + feng::matrix const a = feng::eye( 3, 3 ); + T acc{}; + acc += a.det(); + acc += feng::det( a ); + acc += sink( a.inverse() ); + acc += sink( feng::inverse( a ) ); + acc += sink( feng::inv( a ) ); + acc += sink( feng::expm( a ) ); + + // S7-R1, S7-R2: the LU object, status results and the status overload of inverse + feng::matrix const b{ 3, 2, T{ 1 } }; + auto const f = feng::lu_factor( a ); + acc += static_cast( f.rank() ) + static_cast( f.pivots()[0] ) + f.det(); + acc += f.status() == feng::linalg_status::ok ? T{ 1 } : T{ 0 }; + acc += sink( f.l() ) + sink( f.u() ) + sink( f.p() ); + if ( auto const r = f.solve( b ); r ) + acc += sink( r.value ); + if ( auto const r = f.inverse(); r.ok() ) + acc += sink( r.value ); + if ( auto const r = feng::solve( a, b ); r.status == feng::linalg_status::ok ) + acc += sink( r.value ); + if ( auto const r = feng::try_inverse( a ); r ) + acc += sink( r.value ); + feng::matrix out; + if ( feng::inverse( a, out ) == feng::linalg_status::ok ) + acc += sink( out ); + + // S7-R4, S7-R5: the Cholesky object, row_echelon and the status overload of expm + auto const c = feng::cholesky_factor( a ); + acc += c.status() == feng::linalg_status::ok && c.ok() ? sink( c.l() ) : T{ 0 }; + auto const e = feng::row_echelon( b ); + acc += sink( e.r ) + static_cast( static_cast( e.rank + e.pivot_columns.size() ) ); + acc += e.status == feng::linalg_status::ok ? T{ 1 } : T{ 0 }; + if ( auto const r = feng::rref( b ); r ) + acc += sink( *r ); + if ( feng::expm( a, out ) == feng::linalg_status::ok ) + acc += sink( out ); + acc += static_cast( static_cast( feng::cholesky_decomposition( a, out ) ) ); +#if defined( __cpp_lib_expected ) + // S9-R5: std::expected adapters over the status results and the loaders + if ( auto const x = feng::to_expected( feng::try_inverse( a ) ); x ) + acc += sink( *x ); + if ( auto const x = feng::to_expected( f.solve( b ) ); x ) + acc += sink( *x ); + if ( auto const x = feng::to_expected( f.inverse() ); x ) + acc += sink( *x ); + if ( auto const x = feng::to_expected( feng::solve( a, b ) ); x ) + acc += sink( *x ); + if ( auto const x = feng::to_expected( feng::lu_factor( a ) ); x ) + acc += x->det(); + if ( auto const x = feng::to_expected( feng::svd_factor( a ) ); x ) + acc += sink( x->u() ); + if ( auto const x = feng::to_expected( feng::cholesky_factor( a ) ); x ) + acc += sink( x->l() ); + if ( auto const x = feng::to_expected( feng::row_echelon( b ) ); x ) + acc += sink( x->r ); + for ( auto const fmt : { feng::io_format::txt, feng::io_format::binary, feng::io_format::npy } ) + if ( auto const x = feng::load_expected( "missing.npy", fmt ); x ) + acc += sink( *x ); +#endif + return acc; + } + + template < typename T > + T exercise_svd() + { + // S7-R3: the SVD object, the status pinverse and the legacy SVD and pseudoinverse spellings + feng::matrix const a = feng::eye( 3, 2 ); + feng::matrix u, w, v, x; + T acc{}; + auto const f = feng::svd_factor( a ); + auto const g = feng::svd_factor( a, 8 ); + acc += sink( f.u() ) + sink( f.v() ) + T( f.s()[0] ) + T( static_cast( f.sweeps() + f.rank() + g.rank( 0.5 ) ) ); + acc += f.status() == feng::linalg_status::ok ? T{ 1 } : T{ 0 }; + if ( auto const r = f.pinverse(); r ) + acc += sink( r.value ); + if ( auto const r = f.pinverse( 1.0e-3 ); r.ok() ) + acc += sink( r.value ); +#if defined( __cpp_lib_expected ) + // S9-R5: std::expected adapters over the SVD object and its pinverse + if ( auto const x = feng::to_expected( f ); x ) + acc += sink( x->v() ); + if ( auto const x = feng::to_expected( f.pinverse() ); x ) + acc += sink( *x ); +#endif + if ( feng::pinverse( a, x ) == feng::linalg_status::ok ) + acc += sink( x ); + if ( feng::pinverse( a, x, 1.0e-3 ) == feng::linalg_status::ok ) + acc += sink( x ); + acc += static_cast( static_cast( feng::singular_value_decomposition( a, u, w, v ) ) ); + acc += static_cast( static_cast( feng::singular_value_decomposition( a, u, w, v, 16 ) ) ); + if ( auto const s = feng::singular_value_decomposition( a ); s ) + acc += sink( std::get<0>( *s ) ); + if ( auto const s = feng::svd( a ); s ) + acc += sink( std::get<1>( *s ) ); + acc += sink( feng::svd_inverse( a ) ); + acc += sink( feng::pinverse( a ) ) + sink( feng::pinverse( a, 1.0e-3 ) ); + acc += sink( feng::pinv( a ) ) + sink( feng::pinv( a, 1.0e-3 ) ); + return acc; + } + + template < typename T > + T exercise_real() + { + feng::matrix const a = feng::eye( 3, 3 ); + feng::matrix const b{ 3, 1, T{ 1 } }; + feng::matrix x, l, u, w, v, q, d; + T acc{}; + + // LU + acc += static_cast( feng::lu_decomposition( a, l, u ) ); + if ( auto const lu = feng::lu_decomposition( a ); lu ) + acc += sink( std::get<0>( *lu ) ) + sink( std::get<1>( *lu ) ); + acc += static_cast( feng::lu_solver( a, x, b ) ); + if ( auto const s = feng::lu_solver( a, b ); s ) + acc += sink( *s ); + + // substitution, Cholesky, Householder, row reduction + acc += static_cast( feng::forward_substitution( a, x, b ) ); + acc += static_cast( feng::backward_substitution( a, x, b ) ); + acc += static_cast( feng::cholesky_decomposition( a, l ) ); + feng::householder( a, q, d ); + acc += sink( l ) + sink( q ) + sink( d ); + if ( auto const r = feng::rref( a ); r ) + acc += sink( *r ); + if ( auto const g = feng::gauss_jordan_elimination( a ); g ) + acc += sink( *g ); + + // SVD and pseudo-inverse + acc += static_cast( feng::singular_value_decomposition( a, u, w, v ) ); + if ( auto const s = feng::singular_value_decomposition( a ); s ) + acc += sink( std::get<0>( *s ) ); + if ( auto const s = feng::svd( a ); s ) + acc += sink( std::get<1>( *s ) ); + acc += sink( feng::svd_inverse( a ) ); + acc += sink( feng::pinverse( a ) ); + acc += sink( feng::pinv( a ) ); + + // conjugate gradient solvers + acc += static_cast( feng::conjugate_gradient_squared( a, x, b ) ); + acc += static_cast( feng::cgs( a, x, b ) ); + acc += static_cast( feng::biconjugate_gradient_stabilized_method( a, x, b ) ); + acc += static_cast( feng::bicgstab( a, x, b ) ); + + // eigen solvers + std::vector lv; + std::valarray la; + feng::matrix lm; + acc += static_cast( feng::eigen_jacobi( a, v, lv ) ); + acc += static_cast( feng::eigen_jacobi( a, v, la ) ); + acc += static_cast( feng::eigen_jacobi( a, v, lm ) ); + acc += static_cast( feng::cyclic_eigen_jacobi( a, v, lv ) ); + acc += static_cast( feng::cyclic_eigen_jacobi( a, v, la ) ); + acc += static_cast( feng::cyclic_eigen_jacobi( a, v, lm ) ); + feng::eigen_real_symmetric( a, v, lv ); + feng::eigen_real_symmetric( a, v, la ); + feng::eigen_real_symmetric( a, v, lm ); + acc += feng::eigen_power_iteration( a ); + acc += feng::eigen_power_iteration( a, x.begin() ); + return acc; + } + + template < typename T > + T exercise_hermitian() + { + using C = std::complex; + feng::matrix const a = feng::eye( 3, 3 ); + feng::matrix v, x{ 3, 1 }; + std::vector lv; + std::valarray la; + feng::matrix lm; + feng::eigen_hermitian( a, v, lv ); + feng::eigen_hermitian( a, v, la ); + feng::eigen_hermitian( a, v, lm ); + T acc = std::real( sink( v ) ); + acc += feng::eigen_power_iteration( a ); + acc += feng::eigen_power_iteration( a, x.begin() ); + return acc; + } +} // namespace + +double api_linalg_double() { return exercise_inverse() + exercise_svd() + exercise_real() + exercise_hermitian(); } +float api_linalg_float() { return exercise_inverse() + exercise_svd() + exercise_real(); } +std::complex api_linalg_complex() { return exercise_inverse>() + exercise_svd>(); } +std::complex api_linalg_complex_float() { return exercise_svd>(); } diff --git a/tests/api/link_a.cc b/tests/api/link_a.cc new file mode 100644 index 0000000..92a461c --- /dev/null +++ b/tests/api/link_a.cc @@ -0,0 +1,33 @@ +// S1-R4 link pair, part A (with main): instantiates the same matrix templates as link_b.cc so the two objects +// carry the same inline and template definitions; linking them checks the header for ODR and duplicate-symbol +// defects. Exits 0 when both parts compute the same results. +#include "../../matrix.hpp" + +#include + +double link_b_double(); +std::complex link_b_complex(); + +namespace +{ + template < typename T > + T compute() + { + feng::matrix a = feng::eye( 4, 4 ); + a = a * T{ 2 } + a; + auto const p = a ^ 3; + auto const blk = p.clone( { 0, 2 }, { 0, 2 } ); + return blk[1][1] + p[3][3]; + } +} // namespace + +int main() +{ + if ( compute() != link_b_double() ) + return 1; + if ( compute>() != link_b_complex() ) + return 2; + if ( compute() != 54.0 ) + return 3; + return 0; +} diff --git a/tests/api/link_b.cc b/tests/api/link_b.cc new file mode 100644 index 0000000..4d195cc --- /dev/null +++ b/tests/api/link_b.cc @@ -0,0 +1,20 @@ +// S1-R4 link pair, part B: the same templates as link_a.cc, in a second translation unit. +#include "../../matrix.hpp" + +#include + +namespace +{ + template < typename T > + T compute() + { + feng::matrix a = feng::eye( 4, 4 ); + a = a * T{ 2 } + a; + auto const p = a ^ 3; + auto const blk = p.clone( { 0, 2 }, { 0, 2 } ); + return blk[1][1] + p[3][3]; + } +} // namespace + +double link_b_double() { return compute(); } +std::complex link_b_complex() { return compute>(); } diff --git a/tests/api/reductions.cc b/tests/api/reductions.cc new file mode 100644 index 0000000..382407e --- /dev/null +++ b/tests/api/reductions.cc @@ -0,0 +1,110 @@ +// S1-R4 API family `reductions`: sum, mean, variance, standard_deviation, min, max, minmax, norm, norm_1, norm_2, tr, +// is_* predicates, isequal. Compiled alone by `tools/check.sh api`; calling each API instantiates its template. +#include "../../matrix.hpp" + +#include + +namespace +{ + template < typename T > + T exercise_sums() + { + // sum, norm_1, tr: every element type + feng::matrix const m{ 3, 3, T{ 2 } }; + T acc{}; + acc += feng::sum( m ); + acc += static_cast( feng::norm_1( m ) ); + acc += feng::tr( m ); + acc += m.tr(); + return acc; + } + + template < typename T > + T exercise_statistics() + { + feng::matrix const m{ 3, 3, T{ 2 } }; + T acc{}; + acc += feng::variance( m ); + acc += feng::standard_deviation( m ); + acc += feng::norm_2( m ); + return acc; + } + + template < typename T > + T exercise_ordered() + { + // min, max, minmax, mean, ordering operators: ordered element types + feng::matrix const m{ 3, 3, T{ 2 } }; + auto const less = []( T const& x, T const& y ) { return x < y; }; + T acc{}; + acc += feng::min( m ); + acc += feng::max( m ); + acc += feng::mean( m ); // complex mean compiles since S6 (returns std::complex) + acc += m.min(); + acc += m.max(); + acc += m.min( less ); + acc += m.max( less ); + auto const [lo, hi] = m.minmax(); + auto const [lo2, hi2] = m.minmax( less ); + acc += lo + hi + lo2 + hi2; + feng::matrix const n{ m }; + acc += static_cast( ( m < n ) + ( m > n ) + ( m <= n ) + ( m >= n ) ); + return acc; + } + + template < typename T > + int exercise_predicates() + { + feng::matrix const m = feng::eye( 3, 3 ); + feng::matrix const n{ m }; + int acc{}; + acc += feng::is_column( m ) + feng::is_column_matrix( m ) + feng::iscolumn( m ); + acc += feng::is_row( m ) + feng::is_row_matrix( m ) + feng::isrow( m ); + acc += feng::is_empty( m ) + feng::is_empty_matrix( m ) + feng::isempty( m ); + acc += feng::is_equal( m, n ) + feng::is_equal( m, n, m ); + acc += feng::isequal( m, n ) + feng::isequal( m, n, m ); + acc += feng::is_symmetric( m ); + acc += feng::is_symmetric( m, []( T const& x, T const& y ) { return x == y; } ); + acc += feng::is_orthogonal( m ); + acc += feng::is_orthogonal( m, []( T const& v ) { return v == T{}; } ); + acc += ( m == n ); + return acc; + } + + template < typename T > + int exercise_real_predicates() + { + feng::matrix const m = feng::eye( 3, 3 ); + int acc{}; + acc += feng::is_positive_definite( m ); + acc += feng::is_inf( m )[0][0] + feng::isinf( m )[0][0]; + acc += feng::is_nan( m )[0][0] + feng::isnan( m )[0][0]; + return acc; + } + + double exercise_complex() + { + feng::matrix> const m{ 3, 3, std::complex{ 1.0, 1.0 } }; + double acc{}; + acc += feng::norm_1( m ); + acc += feng::norm( m )[0][0]; + acc += std::real( feng::sum( m ) + feng::tr( m ) ); + return acc; + } +} // namespace + +double api_reductions_double() +{ + return exercise_sums() + exercise_statistics() + exercise_ordered() + + exercise_predicates() + exercise_real_predicates() + exercise_complex(); +} +float api_reductions_float() +{ + return exercise_sums() + exercise_statistics() + exercise_ordered() + + static_cast( exercise_predicates() + exercise_real_predicates() ); +} +std::complex api_reductions_complex() +{ + return exercise_sums>() + static_cast( exercise_predicates>() ); +} +int api_reductions_int() { return exercise_sums() + exercise_ordered() + exercise_predicates(); } diff --git a/tests/api/shape.cc b/tests/api/shape.cc new file mode 100644 index 0000000..a13db6d --- /dev/null +++ b/tests/api/shape.cc @@ -0,0 +1,52 @@ +// S1-R4 API family `shape`: reshape, resize, transpose, ctranspose, fliplr, flipud, flipdim, tril, triu, clear, swap, +// shrink_to_size. Compiled alone by `tools/check.sh api`; calling each API instantiates its template. +#include "../../matrix.hpp" + +#include + +namespace +{ + template < typename M > + auto sink( M const& m ) -> typename M::value_type + { + return m.size() ? m[0][0] : typename M::value_type{}; + } + + template < typename T > + T exercise_shape() + { + feng::matrix m{ 4, 6, T{ 1 } }; + T acc{}; + + acc += sink( m.reshape( 6, 4 ) ); + acc += sink( m.resize( 3, 8 ) ); + acc += sink( m.resize( 5, 5 ) ); + acc += sink( m.shrink_to_size( 3, 3 ) ); + acc += sink( m.transpose() ); + acc += sink( feng::transpose( m ) ); + acc += sink( feng::fliplr( m ) ); + acc += sink( feng::flipud( m ) ); + acc += sink( feng::flipdim( m, 1 ) ); + acc += sink( feng::flipdim( m, 2 ) ); + acc += sink( feng::tril( m ) ); + acc += sink( feng::triu( m ) ); + + feng::matrix other{ 2, 2, T{ 2 } }; + m.swap( other ); + acc += sink( m ) + sink( other ); + m.clear(); + acc += static_cast( static_cast( m.size() ) ); + return acc; + } + + std::complex exercise_ctranspose() + { + feng::matrix> const m{ 2, 3, std::complex{ 1.0, 2.0 } }; + return sink( feng::ctranspose( m ) ); + } +} // namespace + +double api_shape_double() { return exercise_shape(); } +float api_shape_float() { return exercise_shape(); } +std::complex api_shape_complex() { return exercise_shape>() + exercise_ctranspose(); } +int api_shape_int() { return exercise_shape(); } diff --git a/tests/api/signal.cc b/tests/api/signal.cc new file mode 100644 index 0000000..f68ec44 --- /dev/null +++ b/tests/api/signal.cc @@ -0,0 +1,66 @@ +// S1-R4 API family `signal`: conv, conv2, fft, ifft, fftshift, ifftshift, pooling. +// Compiled alone by `tools/check.sh api`; calling each API instantiates its template. +#include "../../matrix.hpp" + +#include +#include + +namespace +{ + template < typename M > + auto sink( M const& m ) -> typename M::value_type + { + return m.size() ? m[0][0] : typename M::value_type{}; + } + + template < typename T > + T exercise_conv() + { + feng::matrix const a{ 5, 6, T{ 1 } }; + feng::matrix const k{ 3, 3, T{ 1 } }; + T acc{}; + acc += sink( feng::conv( a, k ) ); + acc += sink( feng::conv( a, k, std::string{ "same" } ) ); + acc += sink( feng::conv( a, k, std::string{ "valid" } ) ); + acc += sink( feng::conv2( a, k ) ); + acc += sink( feng::conv2( a, k, std::string{ "same" } ) ); + return acc; + } + + template < typename T > + T exercise_pooling() + { + feng::matrix const a{ 4, 6, T{ 1 } }; + T acc{}; + acc += sink( feng::pooling( a, 2, 3 ) ); + acc += sink( feng::pooling( a, 2, 2, "max" ) ); + acc += sink( feng::pooling( a, 2, 2, "min" ) ); + acc += sink( feng::pooling( a, 2 ) ); + acc += sink( feng::pooling( a, 2, "max" ) ); + return acc; + } + + // fft, ifft and the shifts on every element type of D-031: float, double, long double, int and complex + template < typename T > + std::complex exercise_fft() + { + feng::matrix const a{ 4, 4, T{ 1 } }; + std::complex acc{}; + acc += sink( feng::fft( a ) ); + acc += sink( feng::ifft( a ) ); + acc += sink( feng::ifft( feng::fft( a ) ) ); + acc += sink( feng::fftshift( a ) ); + acc += sink( feng::ifftshift( a ) ); + return acc; + } +} // namespace + +double api_signal_double() { return exercise_conv() + exercise_pooling(); } +float api_signal_float() { return exercise_conv() + exercise_pooling(); } +std::complex api_signal_complex() +{ + return exercise_conv>() + exercise_fft() + exercise_fft>() + + exercise_fft() + exercise_fft() + exercise_fft() + + exercise_fft>(); +} +int api_signal_int() { return exercise_conv() + exercise_pooling(); } diff --git a/tests/api/views.cc b/tests/api/views.cc new file mode 100644 index 0000000..4f10c6c --- /dev/null +++ b/tests/api/views.cc @@ -0,0 +1,58 @@ +// S1-R4 API family `views`: make_view, matrix_view, every clone overload, copy, slicing. +// Compiled alone by `tools/check.sh api`; calling each API instantiates its template. +#include "../../matrix.hpp" + +#include +#include +#include + +namespace +{ + template < typename T > + T exercise_views() + { + using size_type = typename feng::matrix::size_type; + feng::matrix m{ 5, 6, T{ 1 } }; + feng::matrix const& cm = m; + T acc{}; + + // make_view and matrix_view, and the view's read API + auto v = feng::make_view( cm, { 1UL, 4UL }, { 2UL, 5UL } ); + feng::matrix_view> w{ cm, std::make_pair( size_type{ 0 }, size_type{ 2 } ), + std::make_pair( size_type{ 1 }, size_type{ 3 } ) }; + auto const [vr, vc] = v.shape(); + acc += static_cast( static_cast( vr + vc + v.row() + v.col() + v.size() + w.size() ) ); + acc += v[0][0]; + acc += v( 1, 1 ); + acc += *v.row_begin( 0 ); + acc += *v.row_cbegin( 1 ); + acc += w[1][1]; + + // every clone overload + feng::matrix a; + a.clone( cm, { 1, 3 }, { 0, 2 } ); + a.clone( cm, 1, 3, 0, 2 ); + auto const b = cm.clone( { 0, 2 }, { 1, 4 } ); + auto const c = cm.clone( 0, 2, 1, 4 ); + acc += a[0][0] + b[0][0] + c[0][0]; + + // copy, whole and into a block + feng::matrix d; + d.copy( cm ); + d.copy( a, { 0, 2 }, { 0, 2 } ); + acc += d[0][0]; + + // slicing constructors + typename feng::matrix::range_type const rr{ 1, 3 }, rc{ 2, 4 }; + feng::matrix const s1{ cm, { 1, 3 }, { 2, 4 } }; + feng::matrix const s2{ cm, rr, rc }; + feng::matrix const s3{ cm, 1, 3, 2, 4 }; + acc += s1[0][0] + s2[0][0] + s3[0][0]; + return acc; + } +} // namespace + +double api_views_double() { return exercise_views(); } +float api_views_float() { return exercise_views(); } +std::complex api_views_complex() { return exercise_views>(); } +int api_views_int() { return exercise_views(); } diff --git a/tests/cases/load_npy.hpp b/tests/cases/load_npy.hpp index c1b3fe9..9d0ed8e 100644 --- a/tests/cases/load_npy.hpp +++ b/tests/cases/load_npy.hpp @@ -3,13 +3,13 @@ TEST_CASE( "Loading npy files", "[load_npy]" ) { { feng::matrix m; - m.load_npy( "./images/u8.npy" ); + (void)m.load_npy( "./images/u8.npy" ); REQUIRE( m[0][0] == 4 ); REQUIRE( m[0][1] == 1 ); REQUIRE( m[0][2] == 8 ); REQUIRE( m[1][0] == 9 ); REQUIRE( m[1][1] == 1 ); REQUIRE( m[1][2] == 5 ); } { feng::matrix m; - m.load_npy( "./images/8.npy" ); + (void)m.load_npy( "./images/8.npy" ); REQUIRE( m[0][0] == 4 ); REQUIRE( m[0][1] == 1 ); REQUIRE( m[0][2] == 8 ); REQUIRE( m[1][0] == 9 ); REQUIRE( m[1][1] == 1 ); REQUIRE( m[1][2] == 5 ); } @@ -17,13 +17,13 @@ TEST_CASE( "Loading npy files", "[load_npy]" ) // [9.510697 , 1.8137231, 5.7381544]] { feng::matrix m; - m.load_npy( "./images/32.npy" ); + (void)m.load_npy( "./images/32.npy" ); REQUIRE( std::abs(4.815519-m[0][0]) < 1.0e-5 ); REQUIRE( std::abs(1.0601262-m[0][1]) < 1.0e-5 ); REQUIRE( std::abs(8.989337-m[0][2]) < 1.0e-5 ); REQUIRE( std::abs(9.510697-m[1][0]) < 1.0e-5 ); REQUIRE( std::abs(1.8137231-m[1][1]) < 1.0e-5 ); REQUIRE( std::abs(5.7381544-m[1][2]) < 1.0e-5 ); } { feng::matrix m; - m.load_npy( "./images/64.npy" ); + (void)m.load_npy( "./images/64.npy" ); REQUIRE( std::abs(4.815519-m[0][0]) < 1.0e-5 ); REQUIRE( std::abs(1.0601262-m[0][1]) < 1.0e-5 ); REQUIRE( std::abs(8.989337-m[0][2]) < 1.0e-5 ); REQUIRE( std::abs(9.510697-m[1][0]) < 1.0e-5 ); REQUIRE( std::abs(1.8137231-m[1][1]) < 1.0e-5 ); REQUIRE( std::abs(5.7381544-m[1][2]) < 1.0e-5 ); } diff --git a/tests/cases/opencv.hpp b/tests/cases/opencv.hpp index 5e383f0..68293fd 100644 --- a/tests/cases/opencv.hpp +++ b/tests/cases/opencv.hpp @@ -1,4 +1,4 @@ -#ifdef OPENCV +#ifdef FENG_MATRIX_OPENCV TEST_CASE( "From OPENCV", "[from_opencv]" ) { diff --git a/tests/cases/s10_r3.hpp b/tests/cases/s10_r3.hpp new file mode 100644 index 0000000..e7246cb --- /dev/null +++ b/tests/cases/s10_r3.hpp @@ -0,0 +1,303 @@ +// S10-R3 (PR-13): the cache-blocked GEMM kernel behind operator*= gives results bit-identical to the strided +// inner_product kernel it replaced (matrix_details::gemm_reference), for every element type and shape, and its row +// split over forced worker counts equals the 1-worker run. +#include +#include +#include + +namespace s10_r3 +{ + template < typename T > + T value( std::size_t i, int salt ) + { + if constexpr ( std::is_same_v< T, int > ) + return static_cast< int >( ( i * 7 + static_cast< std::size_t >( salt ) * 3 ) % 11 ) - 5; + else if constexpr ( std::is_same_v< T, std::complex< double > > ) + return { 0.1 * static_cast< double >( ( i + static_cast< std::size_t >( salt ) ) % 13 ) - 0.37 * static_cast< double >( i % 7 ), + 1.0 / 3.0 + 1e-3 * static_cast< double >( i ) - 0.21 * static_cast< double >( ( i * 5 ) % 9 ) }; + else + return static_cast< T >( 0.1 * static_cast< double >( ( i + static_cast< std::size_t >( salt ) ) % 13 ) - 0.37 * static_cast< double >( i % 7 ) + 1e-3 * static_cast< double >( i ) ); + } + + template < typename T > + feng::matrix< T > sample( std::size_t r, std::size_t c, int salt ) + { + feng::matrix< T > m( r, c ); + for ( std::size_t i = 0; i != m.size(); ++i ) + m.data()[i] = value< T >( i, salt ); + return m; + } + + template < typename T > + bool same_bits( feng::matrix< T > const& x, feng::matrix< T > const& y ) + { + if ( x.row() != y.row() || x.col() != y.col() ) return false; + if ( x.size() == 0 ) return true; + return std::memcmp( x.data(), y.data(), x.size() * sizeof( T ) ) == 0; + } + + template < typename T > + void check( std::size_t m, std::size_t k, std::size_t n ) + { + INFO( m << "x" << k << " * " << k << "x" << n ); + auto const a = sample< T >( m, k, 1 ); + auto const b = sample< T >( k, n, 2 ); + auto const ref = feng::matrix_details::gemm_reference( a, b ); + REQUIRE( ref.row() == m ); + REQUIRE( ref.col() == n ); + auto c = a; + c *= b; + REQUIRE( same_bits( c, ref ) ); + REQUIRE( same_bits( a * b, ref ) ); + auto d = a; + d.direct_multiply( b ); + REQUIRE( same_bits( d, ref ) ); + } + + template < typename T > + void check_all_shapes() + { + check< T >( 0, 4, 3 ); + check< T >( 0, 0, 0 ); + check< T >( 3, 0, 5 ); + check< T >( 1, 1, 1 ); + check< T >( 1, 9, 1 ); + check< T >( 1, 1, 9 ); + check< T >( 1, 7, 9 ); + check< T >( 9, 7, 1 ); + check< T >( 16, 16, 16 ); + check< T >( 17, 17, 17 ); + check< T >( 18, 18, 18 ); + check< T >( 33, 65, 31 ); + check< T >( 300, 7, 5 ); + check< T >( 5, 7, 300 ); + check< T >( 131, 131, 1 ); + check< T >( 70, 140, 270 ); + } + + template < typename T > + void check_self_square( std::size_t n ) + { + auto a = sample< T >( n, n, 3 ); + auto const ref = feng::matrix_details::gemm_reference( a, a ); + a *= a; + REQUIRE( same_bits( a, ref ) ); + } + + template < typename T > + void check_workers( std::size_t m, std::size_t k, std::size_t n ) + { + INFO( m << "x" << k << " * " << k << "x" << n ); + auto const a = sample< T >( m, k, 4 ); + auto const b = sample< T >( k, n, 5 ); + feng::matrix< T > one( m, n ); + feng::matrix_details::gemm_blocked( a.data(), b.data(), one.data(), m, k, n, 1 ); + REQUIRE( same_bits( one, feng::matrix_details::gemm_reference( a, b ) ) ); + for ( std::size_t w : { std::size_t{ 2 }, std::size_t{ 3 }, std::size_t{ 7 } } ) + { + INFO( "workers " << w ); + feng::matrix< T > c( m, n ); + feng::matrix_details::gemm_blocked( a.data(), b.data(), c.data(), m, k, n, w ); + REQUIRE( same_bits( c, one ) ); + } + } +} + +TEST_CASE( "S10-R3 blocked GEMM equals the reference kernel", "[S10][S10-R3]" ) +{ + SECTION( "int" ) { s10_r3::check_all_shapes< int >(); } + SECTION( "float" ) { s10_r3::check_all_shapes< float >(); } + SECTION( "double" ) { s10_r3::check_all_shapes< double >(); } + SECTION( "complex" ) { s10_r3::check_all_shapes< std::complex< double > >(); } + SECTION( "a *= a" ) + { + s10_r3::check_self_square< double >( 1 ); + s10_r3::check_self_square< double >( 18 ); + s10_r3::check_self_square< std::complex< double > >( 33 ); + s10_r3::check_self_square< int >( 64 ); + } + SECTION( "forced worker counts equal the 1-worker run" ) + { + s10_r3::check_workers< double >( 37, 29, 41 ); + s10_r3::check_workers< double >( 2, 5, 3 ); + s10_r3::check_workers< double >( 50, 50, 1 ); + s10_r3::check_workers< float >( 0, 3, 4 ); + s10_r3::check_workers< std::complex< double > >( 23, 19, 17 ); + s10_r3::check_workers< int >( 13, 11, 9 ); + } +} + +// S10-R3 (D-008, par-thresholds): the default worker counts are work-based (matrix_details::work_workers); every +// thresholded path equals its serial reference. Elementwise ops visit each index once, so any split gives the +// 1-worker bits; reductions keep reduce_range's chunked fold with the worker count work_workers picks. +namespace s10_r3 +{ + template < typename T > + feng::matrix< T > serial_add( feng::matrix< T > const& a, feng::matrix< T > const& b ) + { + feng::matrix< T > c = a; + T* x = c.data(); + T const* y = b.data(); + feng::matrix_details::parallel_workers( [x, y]( std::size_t i ) { x[i] += y[i]; }, std::size_t{ 0 }, c.size(), 1 ); + return c; + } + + template < typename T > + void check_elementwise( std::size_t r, std::size_t c ) + { + INFO( "elementwise " << r << "x" << c ); + auto const a = sample< T >( r, c, 6 ); + auto const b = sample< T >( r, c, 7 ); + auto const ref_add = serial_add( a, b ); + REQUIRE( same_bits( a + b, ref_add ) ); + auto s = a; + s += b; + REQUIRE( same_bits( s, ref_add ) ); + + feng::matrix< T > ref_minus = a; + for ( std::size_t i = 0; i != ref_minus.size(); ++i ) ref_minus.data()[i] -= b.data()[i]; + auto m = a; + m -= b; + REQUIRE( same_bits( m, ref_minus ) ); + + feng::matrix< T > ref_neg = a; + for ( std::size_t i = 0; i != ref_neg.size(); ++i ) ref_neg.data()[i] = -ref_neg.data()[i]; + REQUIRE( same_bits( -a, ref_neg ) ); + + auto const twice = []( T& v ) { v = v + v; }; + feng::matrix< T > ref_apply = a; + for ( std::size_t i = 0; i != ref_apply.size(); ++i ) twice( ref_apply.data()[i] ); + auto ap = a; + ap.apply( twice ); + REQUIRE( same_bits( ap, ref_apply ) ); + + // for_each (behind transform and the unary maps) + auto const f = []( T const& x ) { return x * x - x; }; + feng::matrix< T > ref_map( r, c ); + for ( std::size_t i = 0; i != a.size(); ++i ) ref_map.data()[i] = f( a.data()[i] ); + REQUIRE( same_bits( feng::matrix_details::transform( a, f ), ref_map ) ); + + // row-wise copy into a larger matrix and clone back out + feng::matrix< T > big( r + 2, c + 3 ); + for ( std::size_t i = 0; i != big.size(); ++i ) big.data()[i] = value< T >( i, 8 ); + feng::matrix< T > ref_big = big; + for ( std::size_t i = 0; i != r; ++i ) + for ( std::size_t j = 0; j != c; ++j ) + ref_big[i + 1][j + 2] = a[i][j]; + big.copy( a, { 1, 1 + r }, { 2, 2 + c } ); + REQUIRE( same_bits( big, ref_big ) ); + if ( r != 0 && c != 0 ) // clone rejects an empty range + { + feng::matrix< T > cl; + cl.clone( big, 1, 1 + r, 2, 2 + c ); + REQUIRE( same_bits( cl, a ) ); + } + } + + template < typename T > + void check_reduce( std::size_t n ) + { + INFO( "reduce n = " << n ); + feng::matrix< T > m( 1, n ); + for ( std::size_t i = 0; i != n; ++i ) m.data()[i] = value< T >( i, 9 ); + auto const plus = []( T const& x, T const& y ) { return x + y; }; + std::size_t const w = feng::matrix_details::work_workers( n, feng::matrix_details::reduce_grain ); + auto const at = [&m]( std::size_t i ) noexcept -> T const& { return m.data()[i]; }; + T const chunked = feng::matrix_details::reduce_range< T >( at, n, T{}, plus, w ); + T serial{}; + for ( std::size_t i = 0; i != n; ++i ) serial = plus( serial, m.data()[i] ); + if ( w == 1 ) REQUIRE( std::memcmp( &chunked, &serial, sizeof( T ) ) == 0 ); + T const by_iter = feng::matrix_details::reduce( m.begin(), m.end(), T{}, plus ); + T const by_impl = feng::matrix_details::reduce_impl_private::reduce_impl( m )( plus, T{} ); + T const by_sum = feng::sum( m ); + REQUIRE( std::memcmp( &by_iter, &chunked, sizeof( T ) ) == 0 ); + REQUIRE( std::memcmp( &by_impl, &chunked, sizeof( T ) ) == 0 ); + REQUIRE( std::memcmp( &by_sum, &chunked, sizeof( T ) ) == 0 ); + } + + template < typename T > + void check_thresholded() + { + for ( auto [r, c] : { std::pair< std::size_t, std::size_t >{ 0, 0 }, { 0, 5 }, { 1, 1 }, { 16, 16 }, { 17, 17 }, { 18, 18 }, + { 70000, 5 }, { 5, 70000 } } ) + check_elementwise< T >( r, c ); + for ( std::size_t n : { std::size_t{ 0 }, std::size_t{ 1 }, std::size_t{ 16 }, std::size_t{ 17 }, std::size_t{ 18 }, + std::size_t{ 1000 }, feng::matrix_details::reduce_grain, 3 * feng::matrix_details::reduce_grain + 7 } ) + check_reduce< T >( n ); + check< T >( 16, 16, 16 ); + check< T >( 17, 17, 17 ); + check< T >( 18, 18, 18 ); + check< T >( 1, 1, 1 ); + check< T >( 0, 4, 3 ); + check< T >( 400, 64, 64 ); // tall: above the GEMM grain + check< T >( 64, 64, 400 ); // wide + } +} + +TEST_CASE( "S10-R3 thresholded parallel helpers equal the serial path", "[S10][S10-R3]" ) +{ + SECTION( "work_workers: 1 for little work or in serial builds, else min( cores, work / grain )" ) + { + using feng::matrix_details::work_workers; + REQUIRE( work_workers( 0, 1024 ) == 1 ); + REQUIRE( work_workers( 16, feng::matrix_details::elementwise_grain ) == 1 ); + REQUIRE( work_workers( feng::matrix_details::elementwise_grain - 1, feng::matrix_details::elementwise_grain ) == 1 ); + REQUIRE( work_workers( 5, 0 ) >= 1 ); + std::size_t const three = work_workers( 3 * feng::matrix_details::gemm_grain + 1, feng::matrix_details::gemm_grain ); + std::size_t const huge = work_workers( std::numeric_limits< std::size_t >::max(), 1 ); +#ifdef FENG_MATRIX_PARALLEL + unsigned const hc = std::thread::hardware_concurrency(); + std::size_t const cores = hc == 0 ? 1 : hc; + REQUIRE( three == std::min< std::size_t >( cores, 3 ) ); + REQUIRE( huge == cores ); +#else + REQUIRE( three == 1 ); + REQUIRE( huge == 1 ); +#endif + } + SECTION( "int" ) { s10_r3::check_thresholded< int >(); } + SECTION( "float" ) { s10_r3::check_thresholded< float >(); } + SECTION( "double" ) { s10_r3::check_thresholded< double >(); } + SECTION( "complex" ) { s10_r3::check_thresholded< std::complex< double > >(); } +} + +// S10-R3 (B-042, fft-plans): fft2 builds one plan per length per call (twiddle tables, Bluestein chirp and its +// transform) and reuses it for every row and column; the per-row transform it replaced is +// matrix_details::fft2_reference. The plan's values come from the same expressions, so the bits agree. +namespace s10_r3 +{ + template < typename C > + bool same_complex( feng::matrix< C > const& x, feng::matrix< C > const& y ) + { + if ( x.row() != y.row() || x.col() != y.col() ) return false; + for ( std::size_t i = 0; i != x.size(); ++i ) + if ( !( x.data()[i].real() == y.data()[i].real() && x.data()[i].imag() == y.data()[i].imag() ) ) return false; + return true; + } + + template < typename T > + void check_fft() + { + for ( auto [r, c] : { std::pair< std::size_t, std::size_t >{ 0, 0 }, { 1, 1 }, { 1, 13 }, { 1, 32 }, { 13, 1 }, { 32, 1 }, + { 16, 16 }, { 17, 17 }, { 18, 18 }, { 8, 12 }, { 12, 8 }, { 125, 125 }, { 128, 64 }, { 250, 3 } } ) + { + INFO( "fft " << r << "x" << c ); + auto const x = sample< T >( r, c, 10 ); + auto const f = feng::fft( x ); + auto const fr = feng::matrix_details::fft2_reference( x, -1 ); + REQUIRE( f.row() == r ); + REQUIRE( f.col() == c ); + REQUIRE( same_complex( f, fr ) ); + auto const g = feng::ifft( x ); + REQUIRE( same_complex( g, feng::matrix_details::fft2_reference( x, +1 ) ) ); + } + } +} + +TEST_CASE( "S10-R3 fft with reused plans equals the per-row transform", "[S10][S10-R3]" ) +{ + SECTION( "int" ) { s10_r3::check_fft< int >(); } + SECTION( "float" ) { s10_r3::check_fft< float >(); } + SECTION( "double" ) { s10_r3::check_fft< double >(); } + SECTION( "complex" ) { s10_r3::check_fft< std::complex< double > >(); } +} diff --git a/tests/cases/s1_f16.hpp b/tests/cases/s1_f16.hpp new file mode 100644 index 0000000..88cd3f0 --- /dev/null +++ b/tests/cases/s1_f16.hpp @@ -0,0 +1,121 @@ +// S1-R4 (D-015): the F16 templates (matrix power, valarray products, clone overloads). +#include +#include +#include + +namespace s1_f16 +{ + // A small, well-conditioned matrix whose powers stay bounded (spectral radius below 1). + inline feng::matrix sample_square() + { + feng::matrix a{ 3, 3 }; + double const v[3][3] = { { 0.50, 0.10, -0.20 }, { 0.05, 0.40, 0.10 }, { -0.10, 0.20, 0.30 } }; + for ( std::size_t r = 0; r != 3; ++r ) + for ( std::size_t c = 0; c != 3; ++c ) + a[r][c] = v[r][c]; + return a; + } + + inline feng::matrix sample_rect( std::size_t rows, std::size_t cols ) + { + feng::matrix a{ rows, cols }; + for ( std::size_t r = 0; r != rows; ++r ) + for ( std::size_t c = 0; c != cols; ++c ) + a[r][c] = static_cast( 10 * r + c ) + 0.5; + return a; + } + + inline bool near( feng::matrix const& x, feng::matrix const& y, double tol = 1.0e-12 ) + { + if ( x.row() != y.row() || x.col() != y.col() ) + return false; + for ( std::size_t r = 0; r != x.row(); ++r ) + for ( std::size_t c = 0; c != x.col(); ++c ) + if ( std::abs( x[r][c] - y[r][c] ) > tol * ( 1.0 + std::abs( y[r][c] ) ) ) + return false; + return true; + } + + inline bool equal_block( feng::matrix const& block, feng::matrix const& src, + std::size_t r0, std::size_t r1, std::size_t c0, std::size_t c1 ) + { + if ( block.row() != r1 - r0 || block.col() != c1 - c0 ) + return false; + for ( std::size_t r = r0; r != r1; ++r ) + for ( std::size_t c = c0; c != c1; ++c ) + if ( block[r - r0][c - c0] != src[r][c] ) + return false; + return true; + } +} // namespace s1_f16 + +TEST_CASE( "S1 matrix power equals repeated multiplication", "[S1][S1-R4]" ) +{ + auto const a = s1_f16::sample_square(); + auto const id = feng::eye( 3, 3 ); + + REQUIRE( s1_f16::near( a ^ 0, id ) ); + + feng::matrix expected{ id }; + for ( unsigned n = 1; n <= 13; ++n ) + { + expected = expected * a; + if ( n == 1 || n == 2 || n == 3 || n == 13 ) + { + INFO( "n = " << n ); + REQUIRE( s1_f16::near( a ^ n, expected ) ); + } + } +} + +TEST_CASE( "S1 valarray products match explicit sums", "[S1][S1-R4]" ) +{ + std::size_t const rows = 3, cols = 4; + auto const m = s1_f16::sample_rect( rows, cols ); + + // valarray x matrix: (1 x rows) * (rows x cols) -> 1 x cols + std::valarray const left = { 1.5, -2.0, 0.25 }; + auto const lm = left * m; + REQUIRE( lm.row() == 1 ); + REQUIRE( lm.col() == cols ); + for ( std::size_t c = 0; c != cols; ++c ) + { + double s = 0.0; + for ( std::size_t r = 0; r != rows; ++r ) + s += left[r] * m[r][c]; + REQUIRE( std::abs( lm[0][c] - s ) < 1.0e-12 * ( 1.0 + std::abs( s ) ) ); + } + + // matrix x valarray: (rows x cols) * (cols x 1) -> rows x 1 + std::valarray const right = { 0.5, 1.0, -1.0, 2.0 }; + auto const mr = m * right; + REQUIRE( mr.row() == rows ); + REQUIRE( mr.col() == 1 ); + for ( std::size_t r = 0; r != rows; ++r ) + { + double s = 0.0; + for ( std::size_t c = 0; c != cols; ++c ) + s += m[r][c] * right[c]; + REQUIRE( std::abs( mr[r][0] - s ) < 1.0e-12 * ( 1.0 + std::abs( s ) ) ); + } +} + +TEST_CASE( "S1 every clone overload returns the selected block", "[S1][S1-R4]" ) +{ + auto const src = s1_f16::sample_rect( 5, 6 ); + std::size_t const r0 = 1, r1 = 4, c0 = 2, c1 = 5; + + feng::matrix by_braces; + by_braces.clone( src, { r0, r1 }, { c0, c1 } ); + REQUIRE( s1_f16::equal_block( by_braces, src, r0, r1, c0, c1 ) ); + + feng::matrix by_indices; + by_indices.clone( src, r0, r1, c0, c1 ); + REQUIRE( s1_f16::equal_block( by_indices, src, r0, r1, c0, c1 ) ); + + auto const const_braces = src.clone( { r0, r1 }, { c0, c1 } ); + REQUIRE( s1_f16::equal_block( const_braces, src, r0, r1, c0, c1 ) ); + + auto const const_indices = src.clone( r0, r1, c0, c1 ); + REQUIRE( s1_f16::equal_block( const_indices, src, r0, r1, c0, c1 ) ); +} diff --git a/tests/cases/s2_death.hpp b/tests/cases/s2_death.hpp new file mode 100644 index 0000000..8601caf --- /dev/null +++ b/tests/cases/s2_death.hpp @@ -0,0 +1,124 @@ +// S2-R1: fork-based death tests. The callable runs in a child process whose stderr goes to a pipe; the parent +// collects that text and the wait status. A child whose callable returns exits 0 through _exit, which fails the +// death requirement. The child restores SIG_DFL for SIGABRT, sends stdout to /dev/null and exits 2 if the callable +// throws. +#ifndef S2_DEATH_HPP_INCLUDED +#define S2_DEATH_HPP_INCLUDED + +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include + +namespace s2_death +{ + struct outcome + { + bool signaled = false; // WIFSIGNALED + int signal = 0; // WTERMSIG when signaled + bool exited = false; // WIFEXITED + int exit_code = -1; // WEXITSTATUS when exited + std::string err; // everything the child wrote to stderr + }; + + template < typename Callable > + outcome run( Callable&& fn ) + { + outcome out; + int fds[2]; + if ( ::pipe( fds ) != 0 ) + { + out.err = "s2_death: pipe failed"; + return out; + } + std::fflush( stdout ); + std::fflush( stderr ); + pid_t const pid = ::fork(); + if ( pid < 0 ) + { + ::close( fds[0] ); + ::close( fds[1] ); + out.err = "s2_death: fork failed"; + return out; + } + if ( pid == 0 ) + { + ::close( fds[0] ); + ::dup2( fds[1], STDERR_FILENO ); + ::close( fds[1] ); + // Catch's SIGABRT handler would report a fatal error from the child; the default disposition keeps + // the child's death silent apart from the contract message. + ::signal( SIGABRT, SIG_DFL ); + int const null_fd = ::open( "/dev/null", O_WRONLY ); + if ( null_fd >= 0 ) + { + ::dup2( null_fd, STDOUT_FILENO ); + if ( null_fd != STDOUT_FILENO ) ::close( null_fd ); + } +#if defined( __cpp_exceptions ) + try { std::forward< Callable >( fn )(); } + catch ( ... ) { ::_exit( 2 ); } +#else + std::forward< Callable >( fn )(); +#endif + ::_exit( 0 ); + } + ::close( fds[1] ); + char buf[4096]; + for ( ;; ) + { + ssize_t const n = ::read( fds[0], buf, sizeof( buf ) ); + if ( n > 0 ) { out.err.append( buf, static_cast< std::size_t >( n ) ); continue; } + if ( n < 0 && errno == EINTR ) continue; + break; + } + ::close( fds[0] ); + int status = 0; + while ( ::waitpid( pid, &status, 0 ) < 0 && errno == EINTR ) {} + if ( WIFSIGNALED( status ) ) { out.signaled = true; out.signal = WTERMSIG( status ); } + if ( WIFEXITED( status ) ) { out.exited = true; out.exit_code = WEXITSTATUS( status ); } + return out; + } + + inline bool contains( std::string const& text, std::string const& needle ) + { + return text.find( needle ) != std::string::npos; + } + + inline std::size_t line_count( std::string const& text ) + { + std::size_t n = 0; + for ( char ch : text ) n += ( ch == '\n' ); + if ( !text.empty() && text.back() != '\n' ) ++n; + return n; + } +} + +// Requires that `callable` aborts through the contract-violation path with `expected` in its message and that +// no sanitizer diagnostic appears. A function, so lambdas with commas need no extra parentheses. +namespace s2_death +{ + template < typename Callable > + void require_death( Callable&& fn, std::string const& expected ) + { + outcome const out = run( std::forward< Callable >( fn ) ); + INFO( "child stderr: " << out.err ); + INFO( "expected substring: " << expected ); + REQUIRE( out.signaled ); + REQUIRE( out.signal == SIGABRT ); + REQUIRE( contains( out.err, "contract violation" ) ); + REQUIRE( contains( out.err, expected ) ); + REQUIRE_FALSE( contains( out.err, "AddressSanitizer" ) ); + REQUIRE_FALSE( contains( out.err, "runtime error:" ) ); + } +} +#define S2_REQUIRE_DEATH( ... ) s2_death::require_death( __VA_ARGS__ ) + +#endif // S2_DEATH_HPP_INCLUDED diff --git a/tests/cases/s2_r1.hpp b/tests/cases/s2_r1.hpp new file mode 100644 index 0000000..e3196b9 --- /dev/null +++ b/tests/cases/s2_r1.hpp @@ -0,0 +1,30 @@ +// S2-R1 (PR-2): one always-on violation path; runs in debug and NDEBUG builds. +#include "./s2_death.hpp" + +TEST_CASE( "S2 violation path aborts with one message", "[S2][S2-R1]" ) +{ + SECTION( "m(5, 0) on a 2x2 matrix aborts with exactly one stderr line" ) + { + auto const out = s2_death::run( []{ feng::matrix m{ 2, 2 }; double volatile x = m( 5, 0 ); (void)x; } ); + INFO( "child stderr: " << out.err ); + REQUIRE( out.signaled ); + REQUIRE( out.signal == SIGABRT ); + REQUIRE( out.err.rfind( "feng::matrix: contract violation: ", 0 ) == 0 ); + REQUIRE( s2_death::line_count( out.err ) == 1 ); + REQUIRE( out.err.back() == '\n' ); + REQUIRE_FALSE( s2_death::contains( out.err, "AddressSanitizer" ) ); + REQUIRE_FALSE( s2_death::contains( out.err, "runtime error:" ) ); + } + SECTION( "the message names the site's text" ) + { + S2_REQUIRE_DEATH( []{ feng::matrix m{ 2, 2 }; double volatile x = m( 5, 0 ); (void)x; }, "Row index out of boundary" ); + S2_REQUIRE_DEATH( []{ feng::matrix m{ 2, 2 }; m( 0, 7 ) = 1.0; }, "Column index out of boundary" ); + } + SECTION( "a child whose callable returns exits 0 and is not a death" ) + { + auto const out = s2_death::run( []{} ); + REQUIRE_FALSE( out.signaled ); + REQUIRE( out.exited ); + REQUIRE( out.exit_code == 0 ); + } +} diff --git a/tests/cases/s2_r2.hpp b/tests/cases/s2_r2.hpp new file mode 100644 index 0000000..df7a0e0 --- /dev/null +++ b/tests/cases/s2_r2.hpp @@ -0,0 +1,387 @@ +// S2-R2 (PR-3): sizes are checked before the allocator is called. +#include "./s2_death.hpp" + +#include +#include +#include + +namespace s2_r2 +{ + // Stateless allocator that reports every allocate call on stderr, so a death test can prove none happened. + template < typename T > + struct counting_allocator + { + typedef T value_type; + counting_allocator() noexcept = default; + template < typename U > counting_allocator( counting_allocator const& ) noexcept {} + T* allocate( std::size_t n ) + { + std::fputs( "ALLOCATE\n", stderr ); + return std::allocator{}.allocate( n ); + } + void deallocate( T* p, std::size_t n ) noexcept { std::allocator{}.deallocate( p, n ); } + template < typename U > bool operator==( counting_allocator const& ) const noexcept { return true; } + }; + + // Same as counting_allocator but with a small max_size, so the max_size bound is the first one to fire. + template < typename T > + struct small_allocator : counting_allocator + { + small_allocator() noexcept = default; + template < typename U > small_allocator( small_allocator const& ) noexcept {} + std::size_t max_size() const noexcept { return 1000; } + template < typename U > bool operator==( small_allocator const& ) const noexcept { return true; } + }; + + inline constexpr std::size_t two_62 = std::size_t{ 1 } << 62; +} + +TEST_CASE( "S2 huge extents abort before allocation", "[S2][S2-R2]" ) +{ + using s2_r2::two_62; + using counted = feng::matrix< double, s2_r2::counting_allocator >; + + SECTION( "the counting allocator is observable" ) + { + auto const out = s2_death::run( []{ counted m{ 2, 2 }; (void)m; } ); + REQUIRE( out.exited ); + REQUIRE( s2_death::contains( out.err, "ALLOCATE" ) ); + } + SECTION( "2^62 x 8 overflows rows*cols" ) + { + S2_REQUIRE_DEATH( []{ feng::matrix m{ two_62, 8 }; (void)m; }, "size" ); + auto const out = s2_death::run( []{ counted m{ two_62, 8 }; (void)m; } ); + INFO( "child stderr: " << out.err ); + REQUIRE( out.signaled ); + REQUIRE( out.signal == SIGABRT ); + REQUIRE( s2_death::contains( out.err, "size" ) ); + REQUIRE_FALSE( s2_death::contains( out.err, "ALLOCATE" ) ); + } + SECTION( "2^62 x 1 doubles overflow the byte count" ) + { + S2_REQUIRE_DEATH( []{ feng::matrix m{ two_62, 1 }; (void)m; }, "byte count" ); + auto const out = s2_death::run( []{ counted m{ s2_r2::counting_allocator{}, two_62, 1 }; (void)m; } ); + INFO( "child stderr: " << out.err ); + REQUIRE( out.signaled ); + REQUIRE( out.signal == SIGABRT ); + REQUIRE( s2_death::contains( out.err, "contract violation" ) ); + REQUIRE( s2_death::contains( out.err, "byte count" ) ); + REQUIRE_FALSE( s2_death::contains( out.err, "ALLOCATE" ) ); + } + SECTION( "100 x 20 is above an allocator max_size of 1000" ) + { + using small = feng::matrix< double, s2_r2::small_allocator >; + auto const ok = s2_death::run( []{ small m{ 10, 10 }; (void)m; } ); + REQUIRE( ok.exited ); + REQUIRE( s2_death::contains( ok.err, "ALLOCATE" ) ); + auto const out = s2_death::run( []{ small m{ 100, 20 }; (void)m; } ); + INFO( "child stderr: " << out.err ); + REQUIRE( out.signaled ); + REQUIRE( out.signal == SIGABRT ); + REQUIRE( s2_death::contains( out.err, "contract violation" ) ); + REQUIRE( s2_death::contains( out.err, "max_size" ) ); + REQUIRE( s2_death::contains( out.err, "size" ) ); + REQUIRE_FALSE( s2_death::contains( out.err, "ALLOCATE" ) ); + } + SECTION( "the (row, col, {values}) constructor checks before allocating" ) + { + // 2^32 * 2^32 wraps to 0 in 64 bits; the product must be rejected, not treated as empty. + auto const out = s2_death::run( []{ counted m{ std::size_t{ 1 } << 32, std::size_t{ 1 } << 32, { 1.0, 2.0 } }; (void)m; } ); + INFO( "child stderr: " << out.err ); + REQUIRE( out.signaled ); + REQUIRE( out.signal == SIGABRT ); + REQUIRE( s2_death::contains( out.err, "contract violation" ) ); + REQUIRE( s2_death::contains( out.err, "size" ) ); + REQUIRE_FALSE( s2_death::contains( out.err, "ALLOCATE" ) ); + } +} + +#include +#include +#include +#include +#include + +#include + +namespace s2_r2 +{ + using counted = feng::matrix< double, counting_allocator >; + + // Marks the point after which a death test requires no allocate call. + inline void ready() { std::fputs( "READY\n", stderr ); std::fflush( stderr ); } + + // Child-side checks of a death: SIGABRT, `expected` in the message, no ALLOCATE after READY. + template < typename Callable > + void require_death_without_allocate( Callable&& fn, std::string const& expected ) + { + auto const out = s2_death::run( std::forward< Callable >( fn ) ); + INFO( "child stderr: " << out.err ); + INFO( "expected substring: " << expected ); + REQUIRE( out.signaled ); + REQUIRE( out.signal == SIGABRT ); + REQUIRE( s2_death::contains( out.err, "contract violation" ) ); + REQUIRE( s2_death::contains( out.err, expected ) ); + REQUIRE_FALSE( s2_death::contains( out.err, "AddressSanitizer" ) ); + REQUIRE_FALSE( s2_death::contains( out.err, "runtime error:" ) ); + std::size_t const at = out.err.find( "READY\n" ); + REQUIRE( at != std::string::npos ); + REQUIRE_FALSE( s2_death::contains( out.err.substr( at ), "ALLOCATE" ) ); + } + + inline feng::matrix numbered( std::size_t r, std::size_t c ) + { + feng::matrix m{ r, c }; + for ( std::size_t i = 0; i != m.size(); ++i ) m.data()[i] = static_cast( i ); + return m; + } + + inline counted numbered_counted( std::size_t r, std::size_t c ) + { + counted m{ r, c }; + for ( std::size_t i = 0; i != m.size(); ++i ) m.data()[i] = static_cast( i ); + return m; + } +} + +TEST_CASE( "S2 at and call operator abort out of range", "[S2][S2-R2]" ) +{ + using s2_r2::numbered; + using view_type = feng::matrix_view< double, std::allocator >; + + SECTION( "in-range at and () agree, const and non-const" ) + { + auto m = numbered( 3, 4 ); + auto const& cm = m; + for ( std::size_t r = 0; r != 3; ++r ) + for ( std::size_t c = 0; c != 4; ++c ) + { + REQUIRE( m.at( r, c ) == static_cast( r * 4 + c ) ); + REQUIRE( cm.at( r, c ) == m( r, c ) ); + REQUIRE( &cm.at( r, c ) == &m.at( r, c ) ); + } + m.at( 2, 3 ) = -1.0; + REQUIRE( m( 2, 3 ) == -1.0 ); + view_type const v{ m, { 1, 3 }, { 1, 4 } }; + REQUIRE( v.at( 0, 0 ) == m( 1, 1 ) ); + REQUIRE( v.at( 1, 2 ) == -1.0 ); + REQUIRE( v( 1, 2 ) == v.at( 1, 2 ) ); + } + SECTION( "row or column at the extent aborts with index" ) + { + S2_REQUIRE_DEATH( []{ auto m = s2_r2::numbered( 3, 4 ); m.at( 3, 0 ) = 1.0; }, "index" ); + S2_REQUIRE_DEATH( []{ auto m = s2_r2::numbered( 3, 4 ); m.at( 0, 4 ) = 1.0; }, "index" ); + S2_REQUIRE_DEATH( []{ auto const m = s2_r2::numbered( 3, 4 ); double volatile x = m.at( 3, 0 ); (void)x; }, "index" ); + S2_REQUIRE_DEATH( []{ auto const m = s2_r2::numbered( 3, 4 ); double volatile x = m.at( 0, 4 ); (void)x; }, "index" ); + S2_REQUIRE_DEATH( []{ auto m = s2_r2::numbered( 3, 4 ); m( 3, 0 ) = 1.0; }, "index" ); + S2_REQUIRE_DEATH( []{ auto m = s2_r2::numbered( 3, 4 ); m( 0, 4 ) = 1.0; }, "index" ); + S2_REQUIRE_DEATH( []{ auto const m = s2_r2::numbered( 3, 4 ); double volatile x = m( 3, 0 ); (void)x; }, "index" ); + S2_REQUIRE_DEATH( []{ auto const m = s2_r2::numbered( 3, 4 ); double volatile x = m( 0, 4 ); (void)x; }, "index" ); + S2_REQUIRE_DEATH( []{ auto m = s2_r2::numbered( 3, 4 ); m.at( std::size_t( -1 ), 0 ) = 1.0; }, "index" ); + } + SECTION( "m[r] aborts when r >= row()" ) + { + S2_REQUIRE_DEATH( []{ auto m = s2_r2::numbered( 3, 4 ); double* volatile p = m[3]; (void)p; }, "index" ); + S2_REQUIRE_DEATH( []{ auto const m = s2_r2::numbered( 3, 4 ); double const* volatile p = m[3]; (void)p; }, "index" ); + } + SECTION( "views check against their own extents" ) + { + S2_REQUIRE_DEATH( []{ auto m = s2_r2::numbered( 3, 4 ); view_type const v{ m, { 1, 3 }, { 1, 4 } }; double volatile x = v.at( 2, 0 ); (void)x; }, "index" ); + S2_REQUIRE_DEATH( []{ auto m = s2_r2::numbered( 3, 4 ); view_type const v{ m, { 1, 3 }, { 1, 4 } }; double volatile x = v.at( 0, 3 ); (void)x; }, "index" ); + S2_REQUIRE_DEATH( []{ auto m = s2_r2::numbered( 3, 4 ); view_type const v{ m, { 1, 3 }, { 1, 4 } }; double volatile x = v( 2, 0 ); (void)x; }, "index" ); + S2_REQUIRE_DEATH( []{ auto m = s2_r2::numbered( 3, 4 ); view_type const v{ m, { 1, 3 }, { 1, 4 } }; double volatile x = v( 0, 3 ); (void)x; }, "index" ); + } +} + +TEST_CASE( "S2 empty matrices have no accessible elements", "[S2][S2-R2]" ) +{ + std::pair< std::size_t, std::size_t > const shapes[] = { { 0, 0 }, { 0, 5 }, { 5, 0 } }; + for ( auto const& [r, c] : shapes ) + { + INFO( "shape " << r << "x" << c ); + feng::matrix m{ r, c }; + REQUIRE( m.row() == r ); + REQUIRE( m.col() == c ); + REQUIRE( m.size() == 0 ); + REQUIRE( m.begin() == m.end() ); + auto const& cm = m; + REQUIRE( cm.begin() == cm.end() ); + std::pair< std::size_t, std::size_t > const probes[] = { { 0, 0 }, { 0, 4 }, { 4, 0 }, { 4, 4 }, { 5, 5 } }; + for ( auto const& [pr, pc] : probes ) + { + INFO( "probe (" << pr << ", " << pc << ")" ); + S2_REQUIRE_DEATH( [&]{ feng::matrix e{ r, c }; e.at( pr, pc ) = 1.0; }, "index" ); + S2_REQUIRE_DEATH( [&]{ feng::matrix const e{ r, c }; double volatile x = e.at( pr, pc ); (void)x; }, "index" ); + S2_REQUIRE_DEATH( [&]{ feng::matrix e{ r, c }; e( pr, pc ) = 1.0; }, "index" ); + S2_REQUIRE_DEATH( [&]{ feng::matrix const e{ r, c }; double volatile x = e( pr, pc ); (void)x; }, "index" ); + } + } +} + +TEST_CASE( "S2 negative signed dimensions abort before allocation", "[S2][S2-R2]" ) +{ + using s2_r2::counted; + SECTION( "non-negative signed dimensions build the matrix" ) + { + int const r = 2, c = 3; + feng::matrix m{ r, c, { 1.0, 2.0, 3.0, 4.0, 5.0, 6.0 } }; + REQUIRE( m.row() == 2 ); + REQUIRE( m.col() == 3 ); + REQUIRE( m( 1, 2 ) == 6.0 ); + } + SECTION( "a negative row or column aborts with size" ) + { + s2_r2::require_death_without_allocate( []{ int volatile r = -1; s2_r2::ready(); counted m{ int( r ), 2, { 1.0 } }; (void)m; }, "size" ); + s2_r2::require_death_without_allocate( []{ int volatile c = -3; s2_r2::ready(); counted m{ 2, int( c ), { 1.0 } }; (void)m; }, "size" ); + s2_r2::require_death_without_allocate( []{ long long volatile r = -1; s2_r2::ready(); counted m{ (long long)( r ), 0LL, {} }; (void)m; }, "size" ); + s2_r2::require_death_without_allocate( []{ short volatile c = -1; s2_r2::ready(); counted m{ 0, (short)( c ), {} }; (void)m; }, "size" ); + } +} + +TEST_CASE( "S2 reshape checks the element count and overflow", "[S2][S2-R2]" ) +{ + SECTION( "a matching reshape keeps the data" ) + { + auto m = s2_r2::numbered( 2, 6 ); + double const* const p = m.data(); + m.reshape( 3, 4 ); + REQUIRE( m.row() == 3 ); + REQUIRE( m.col() == 4 ); + REQUIRE( m.data() == p ); + REQUIRE( m( 2, 3 ) == 11.0 ); + } + SECTION( "a different element count aborts with reshape" ) + { + S2_REQUIRE_DEATH( []{ auto m = s2_r2::numbered( 2, 6 ); m.reshape( 4, 4 ); }, "reshape" ); + S2_REQUIRE_DEATH( []{ auto m = s2_r2::numbered( 2, 6 ); m.reshape( 0, 12 ); }, "reshape" ); + } + SECTION( "a wrapping r*c aborts with reshape" ) + { + // 2^32 * 2^32 wraps to 0 and 2^63 * 2 wraps to 0: both equal an empty matrix's size() unless checked. + S2_REQUIRE_DEATH( []{ feng::matrix m; m.reshape( std::size_t{ 1 } << 32, std::size_t{ 1 } << 32 ); }, "reshape" ); + S2_REQUIRE_DEATH( []{ feng::matrix m{ 0, 3 }; m.reshape( std::size_t{ 1 } << 63, 2 ); }, "reshape" ); + // (2^64 - 1) * 13 wraps to 2^64 - 13; 12 + 2^64 wraps to 12 for r*c = 12 via r = 2^62 + 3, c = 4. + S2_REQUIRE_DEATH( []{ auto m = s2_r2::numbered( 2, 6 ); m.reshape( ( std::size_t{ 1 } << 62 ) + 3, 4 ); }, "reshape" ); + } +} + +TEST_CASE( "S2 clone bounds abort before reading the source", "[S2][S2-R2]" ) +{ + using s2_r2::counted; + using s2_r2::numbered_counted; + using s2_r2::require_death_without_allocate; + using s2_r2::ready; + + SECTION( "an in-bounds clone copies the block" ) + { + auto const src = s2_r2::numbered( 4, 5 ); + feng::matrix a; + a.clone( src, 1, 4, 2, 5 ); + REQUIRE( a.row() == 3 ); + REQUIRE( a.col() == 3 ); + REQUIRE( a( 0, 0 ) == src( 1, 2 ) ); + REQUIRE( a( 2, 2 ) == src( 3, 4 ) ); + auto const b = src.clone( { 0, 4 }, { 0, 5 } ); + REQUIRE( b == src ); + feng::matrix const c{ src, { 3, 4 }, { 4, 5 } }; + REQUIRE( c.size() == 1 ); + REQUIRE( c( 0, 0 ) == 19.0 ); + } + SECTION( "member clone( other, r0, r1, c0, c1 )" ) + { + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); counted d; ready(); d.clone( s, 0, 5, 0, 5 ); }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); counted d; ready(); d.clone( s, 0, 4, 0, 6 ); }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); counted d; ready(); d.clone( s, 2, 2, 0, 5 ); }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); counted d; ready(); d.clone( s, 3, 1, 0, 5 ); }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); counted d; ready(); d.clone( s, 0, 4, 5, 5 ); }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); counted d; ready(); d.clone( s, 0, 4, 4, 2 ); }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); counted d; ready(); d.clone( s, 0, std::size_t( -1 ), 0, 5 ); }, "clone" ); + } + SECTION( "member clone with brace lists" ) + { + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); counted d; ready(); d.clone( s, { 0, 5 }, { 0, 5 } ); }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); counted d; ready(); d.clone( s, { 0, 1, 2 }, { 0, 5 } ); }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); counted d; ready(); d.clone( s, { 0, 4 }, { 1 } ); }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); counted d; ready(); d.clone( s, {}, { 0, 5 } ); }, "clone" ); + } + SECTION( "const clone( r0, r1, c0, c1 ) and clone( {r0, r1}, {c0, c1} )" ) + { + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); ready(); auto d = s.clone( 0, 5, 0, 5 ); (void)d; }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); ready(); auto d = s.clone( 1, 1, 0, 5 ); (void)d; }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); ready(); auto d = s.clone( { 0, 4 }, { 0, 6 } ); (void)d; }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); ready(); auto d = s.clone( { 0 }, { 0, 5 } ); (void)d; }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); ready(); auto d = s.clone( { 0, 4 }, { 0, 1, 5 } ); (void)d; }, "clone" ); + } + SECTION( "slicing constructors" ) + { + using range = std::pair< std::size_t, std::size_t >; + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); ready(); counted d{ s, { 0, 5 }, { 0, 5 } }; (void)d; }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); ready(); counted d{ s, { 0, 4, 1 }, { 0, 5 } }; (void)d; }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); ready(); counted d{ s, { 0, 4 }, { 5 } }; (void)d; }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); ready(); counted d{ s, range{ 0, 4 }, range{ 3, 3 } }; (void)d; }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); ready(); counted d{ s, 0, 4, 0, 9 }; (void)d; }, "clone" ); + require_death_without_allocate( []{ auto const s = numbered_counted( 4, 5 ); ready(); counted d{ s, 2, 1, 0, 5 }; (void)d; }, "clone" ); + // the converting (other element type) slicing constructor + S2_REQUIRE_DEATH( []{ feng::matrix const s{ 4, 5 }; feng::matrix d{ s, { 0, 5 }, { 0, 5 } }; (void)d; }, "clone" ); + S2_REQUIRE_DEATH( []{ feng::matrix const s{ 4, 5 }; feng::matrix d{ s, { 0, 4 }, { 0 } }; (void)d; }, "clone" ); + } +} + +TEST_CASE( "S2 byte counts above PTRDIFF_MAX abort before allocation", "[S2][S2-R2]" ) +{ + using s2_r2::counted; + // The counting allocator keeps the default max_size (SIZE_MAX / sizeof(T)), so only the PTRDIFF_MAX bound + // stands between 2^60 doubles (2^63 bytes) and the allocator. + s2_r2::require_death_without_allocate( []{ s2_r2::ready(); counted m{ std::size_t{ 1 } << 60, 1 }; (void)m; }, "PTRDIFF_MAX" ); + s2_r2::require_death_without_allocate( []{ s2_r2::ready(); counted m{ s2_r2::counting_allocator{}, 1, std::size_t{ 1 } << 61 }; (void)m; }, "size" ); + S2_REQUIRE_DEATH( []{ feng::matrix m{ std::size_t{ 1 } << 63, 1 }; (void)m; }, "PTRDIFF_MAX" ); + S2_REQUIRE_DEATH( []{ feng::matrix m{ std::size_t{ 1 } << 63, 1 }; (void)m; }, "size" ); +} + +TEST_CASE( "S2 resize and shrink_to_size check sizes before allocation", "[S2][S2-R2]" ) +{ + using s2_r2::counted; + // 2^32 x 2^32 wraps to 0, which equals an empty matrix's size(); it must not be taken as a reshape. + s2_r2::require_death_without_allocate( []{ counted m; s2_r2::ready(); m.resize( std::size_t{ 1 } << 32, std::size_t{ 1 } << 32 ); }, "size" ); + s2_r2::require_death_without_allocate( []{ counted m{ 2, 2 }; s2_r2::ready(); m.resize( std::size_t{ 1 } << 62, 8 ); }, "size" ); + s2_r2::require_death_without_allocate( []{ counted m{ 2, 2 }; s2_r2::ready(); m.shrink_to_size( std::size_t{ 1 } << 62, 8 ); }, "size" ); + s2_r2::require_death_without_allocate( []{ counted m{ 2, 2 }; s2_r2::ready(); m.shrink_to_size( std::size_t{ 1 } << 60, 1 ); }, "PTRDIFF_MAX" ); +} + +TEST_CASE( "S2 death helper isolates the child", "[S2][S2-R2]" ) +{ + SECTION( "the child runs with SIG_DFL for SIGABRT" ) + { + auto const out = s2_death::run( []{ struct sigaction sa{}; ::sigaction( SIGABRT, nullptr, &sa ); if ( sa.sa_handler != SIG_DFL ) ::_exit( 4 ); } ); + REQUIRE( out.exited ); + REQUIRE( out.exit_code == 0 ); + } + SECTION( "the child's stdout is /dev/null" ) + { + auto const out = s2_death::run( [] + { + struct stat o{}, n{}; + if ( ::fstat( STDOUT_FILENO, &o ) != 0 || ::stat( "/dev/null", &n ) != 0 ) ::_exit( 5 ); + if ( o.st_rdev != n.st_rdev || !S_ISCHR( o.st_mode ) ) ::_exit( 3 ); + } ); + REQUIRE( out.exited ); + REQUIRE( out.exit_code == 0 ); + } + SECTION( "an abort prints only the contract message" ) + { + auto const out = s2_death::run( []{ feng::matrix m{ 1, 1 }; m.at( 1, 1 ) = 0.0; } ); + INFO( "child stderr: " << out.err ); + REQUIRE( out.signaled ); + REQUIRE( out.signal == SIGABRT ); + REQUIRE( s2_death::line_count( out.err ) == 1 ); + REQUIRE( out.err.rfind( "feng::matrix: contract violation: ", 0 ) == 0 ); + } +#if defined( __cpp_exceptions ) + SECTION( "a throwing callable exits with code 2" ) + { + auto const out = s2_death::run( []{ throw std::runtime_error( "s2 death helper" ); } ); + REQUIRE( out.exited ); + REQUIRE( out.exit_code == 2 ); + } +#endif +} diff --git a/tests/cases/s2_r3.hpp b/tests/cases/s2_r3.hpp new file mode 100644 index 0000000..d66f9d3 --- /dev/null +++ b/tests/cases/s2_r3.hpp @@ -0,0 +1,143 @@ +// S2-R3 (PR-4): crop/pad copies the overlapping block exactly; flips are exact (D-004 axes). +#include "./s2_death.hpp" + +#include +#include +#include + +namespace s2_r3 +{ + inline feng::matrix distinct( std::size_t rows, std::size_t cols ) + { + feng::matrix m{ rows, cols }; + for ( std::size_t r = 0; r != rows; ++r ) + for ( std::size_t c = 0; c != cols; ++c ) + m[r][c] = static_cast( r * cols + c + 1 ); + return m; + } + + inline bool same( feng::matrix const& a, feng::matrix const& b ) + { + if ( a.row() != b.row() || a.col() != b.col() ) return false; + for ( std::size_t r = 0; r != a.row(); ++r ) + for ( std::size_t c = 0; c != a.col(); ++c ) + if ( a[r][c] != b[r][c] ) return false; + return true; + } + + inline std::vector< std::pair< std::size_t, std::size_t > > flip_shapes() + { + return { { 3, 5 }, { 5, 3 }, { 4, 4 }, { 1, 1 }, { 0, 0 }, { 0, 3 }, { 3, 0 } }; + } + + inline void check_shrink( std::size_t r0, std::size_t c0, std::size_t r1, std::size_t c1 ) + { + feng::matrix const original = distinct( r0, c0 ); + feng::matrix m = original; + m.shrink_to_size( r1, c1 ); + REQUIRE( m.row() == r1 ); + REQUIRE( m.col() == c1 ); + for ( std::size_t r = 0; r != r1; ++r ) + for ( std::size_t c = 0; c != c1; ++c ) + { + INFO( r0 << "x" << c0 << " -> " << r1 << "x" << c1 << ", r = " << r << ", c = " << c ); + double const expected = ( r < r0 && c < c0 ) ? original[r][c] : 0.0; + REQUIRE( m[r][c] == expected ); + } + } +} + +TEST_CASE( "S2 shrink_to_size 5x5 to 5x3 is exact", "[S2][S2-R3]" ) +{ + feng::matrix m{ 5, 5 }; + for ( std::size_t r = 0; r != 5; ++r ) + for ( std::size_t c = 0; c != 5; ++c ) + m[r][c] = static_cast( r * 5 + c ); + m.shrink_to_size( 5, 3 ); + REQUIRE( m.row() == 5 ); + REQUIRE( m.col() == 3 ); + for ( std::size_t r = 0; r != 5; ++r ) + for ( std::size_t c = 0; c != 3; ++c ) + { + INFO( "r = " << r << ", c = " << c ); + REQUIRE( m[r][c] == static_cast( r * 5 + c ) ); + } +} + +TEST_CASE( "S2 shrink_to_size crops and pads exactly", "[S2][S2-R3]" ) +{ + s2_r3::check_shrink( 5, 5, 5, 3 ); + s2_r3::check_shrink( 3, 10, 5, 2 ); + s2_r3::check_shrink( 2, 3, 4, 5 ); +} + +TEST_CASE( "S2 shrink_to_size aborts on a zero extent", "[S2][S2-R3]" ) +{ + S2_REQUIRE_DEATH( []{ feng::matrix m{ 2, 2 }; m.shrink_to_size( 0, 2 ); }, "shrink_to_size" ); + S2_REQUIRE_DEATH( []{ feng::matrix m{ 2, 2 }; m.shrink_to_size( 2, 0 ); }, "shrink_to_size" ); +} + +TEST_CASE( "S2 flipdim 1 reverses the rows", "[S2][S2-R3]" ) +{ + for ( auto [rows, cols] : s2_r3::flip_shapes() ) + { + INFO( rows << "x" << cols ); + feng::matrix const m = s2_r3::distinct( rows, cols ); + feng::matrix const f = feng::flipdim( m, 1 ); + REQUIRE( f.row() == rows ); + REQUIRE( f.col() == cols ); + for ( std::size_t r = 0; r != rows; ++r ) + for ( std::size_t c = 0; c != cols; ++c ) + REQUIRE( f[r][c] == m[rows - 1 - r][c] ); + } +} + +TEST_CASE( "S2 flipdim 2 reverses the columns", "[S2][S2-R3]" ) +{ + for ( auto [rows, cols] : s2_r3::flip_shapes() ) + { + INFO( rows << "x" << cols ); + feng::matrix const m = s2_r3::distinct( rows, cols ); + feng::matrix const f = feng::flipdim( m, 2 ); + REQUIRE( f.row() == rows ); + REQUIRE( f.col() == cols ); + for ( std::size_t r = 0; r != rows; ++r ) + for ( std::size_t c = 0; c != cols; ++c ) + REQUIRE( f[r][c] == m[r][cols - 1 - c] ); + } +} + +TEST_CASE( "S2 flipdim twice restores the input", "[S2][S2-R3]" ) +{ + for ( auto [rows, cols] : s2_r3::flip_shapes() ) + { + INFO( rows << "x" << cols ); + feng::matrix const m = s2_r3::distinct( rows, cols ); + REQUIRE( s2_r3::same( feng::flipdim( feng::flipdim( m, 1 ), 1 ), m ) ); + REQUIRE( s2_r3::same( feng::flipdim( feng::flipdim( m, 2 ), 2 ), m ) ); + } +} + +TEST_CASE( "S2 fliplr flips columns and flipud flips rows", "[S2][S2-R3]" ) +{ + for ( auto [rows, cols] : s2_r3::flip_shapes() ) + { + INFO( rows << "x" << cols ); + feng::matrix const m = s2_r3::distinct( rows, cols ); + REQUIRE( s2_r3::same( feng::fliplr( m ), feng::flipdim( m, 2 ) ) ); + REQUIRE( s2_r3::same( feng::flipud( m ), feng::flipdim( m, 1 ) ) ); + } + feng::matrix const m = s2_r3::distinct( 2, 3 ); + feng::matrix const lr = feng::fliplr( m ); + feng::matrix const ud = feng::flipud( m ); + REQUIRE( lr[0][0] == 3.0 ); + REQUIRE( lr[1][2] == 4.0 ); + REQUIRE( ud[0][0] == 4.0 ); + REQUIRE( ud[1][2] == 3.0 ); +} + +TEST_CASE( "S2 flipdim aborts on another dim", "[S2][S2-R3]" ) +{ + S2_REQUIRE_DEATH( []{ feng::matrix m{ 2, 2 }; (void)feng::flipdim( m, 0 ); }, "flipdim" ); + S2_REQUIRE_DEATH( []{ feng::matrix m{ 2, 2 }; (void)feng::flipdim( m, 3 ); }, "flipdim" ); +} diff --git a/tests/cases/s2_r4.hpp b/tests/cases/s2_r4.hpp new file mode 100644 index 0000000..2293d98 --- /dev/null +++ b/tests/cases/s2_r4.hpp @@ -0,0 +1,318 @@ +// S2-R4 (PR-3): operand shapes are checked before traversal; pooling modes; self-assignment and overlap policy. +#include "./s2_death.hpp" + +#include +#include + +TEST_CASE( "S2 valarray length 3 times 2x3 aborts", "[S2][S2-R4]" ) +{ + SECTION( "length 2 gives the explicit sums" ) + { + feng::matrix m{ 2, 3 }; + for ( std::size_t r = 0; r != 2; ++r ) + for ( std::size_t c = 0; c != 3; ++c ) + m[r][c] = static_cast( 10 * r + c ) + 0.5; + std::valarray const v{ 2.0, -3.0 }; + auto const p = v * m; + REQUIRE( p.row() == 1 ); + REQUIRE( p.col() == 3 ); + for ( std::size_t c = 0; c != 3; ++c ) + REQUIRE( p[0][c] == 2.0 * m[0][c] - 3.0 * m[1][c] ); + } + SECTION( "length 3 aborts with a shape message" ) + { + S2_REQUIRE_DEATH( []{ + feng::matrix m{ 2, 3 }; + std::valarray const v{ 1.0, 2.0, 3.0 }; + auto const p = v * m; + (void)p; + }, "shape" ); + } +} + +TEST_CASE( "S2 left vector product checks rows", "[S2][S2-R4]" ) +{ + SECTION( "length 2 times 2x3 gives the explicit sums" ) + { + feng::matrix m{ 2, 3 }; + for ( std::size_t r = 0; r != 2; ++r ) + for ( std::size_t c = 0; c != 3; ++c ) + m[r][c] = static_cast( 10 * r + c ) + 0.5; + std::vector const v{ 2.0, -3.0 }; + auto const p = v * m; + REQUIRE( p.row() == 1 ); + REQUIRE( p.col() == 3 ); + for ( std::size_t c = 0; c != 3; ++c ) + REQUIRE( p[0][c] == 2.0 * m[0][c] - 3.0 * m[1][c] ); + } + SECTION( "length 3 times 2x3 aborts with a shape message" ) + { + S2_REQUIRE_DEATH( []{ + feng::matrix m{ 2, 3 }; + std::vector const v{ 1.0, 2.0, 3.0 }; + auto const p = v * m; + (void)p; + }, "shape" ); + } + SECTION( "2x3 times a vector of length 2 aborts with a shape message" ) + { + S2_REQUIRE_DEATH( []{ + feng::matrix m{ 2, 3 }; + std::vector const v{ 1.0, 2.0 }; + auto const p = m * v; + (void)p; + }, "shape" ); + } + SECTION( "2x3 times a valarray of length 2 aborts with a shape message" ) + { + S2_REQUIRE_DEATH( []{ + feng::matrix m{ 2, 3 }; + std::valarray const v{ 1.0, 2.0 }; + auto const p = m * v; + (void)p; + }, "shape" ); + } +} + +TEST_CASE( "S2 binary and ternary element-wise maps check shapes", "[S2][S2-R4]" ) +{ + SECTION( "binary map_impl" ) + { + S2_REQUIRE_DEATH( []{ + feng::matrix a{ 2, 3 }; + feng::matrix b{ 3, 2 }; + auto const p = feng::matrix_details::map( []( double x, double y ){ return x + y; } )( a, b ); + (void)p; + }, "shape" ); + } + SECTION( "ternary map_impl" ) + { + S2_REQUIRE_DEATH( []{ + feng::matrix a{ 2, 3 }; + feng::matrix b{ 2, 3 }; + feng::matrix c{ 2, 2 }; + auto const p = feng::matrix_details::map( []( double x, double y, double z ){ return x + y + z; } )( a, b, c ); + (void)p; + }, "shape" ); + } + SECTION( "hypot" ) + { + S2_REQUIRE_DEATH( []{ + feng::matrix a{ 2, 3 }; + feng::matrix b{ 3, 2 }; + auto const p = feng::hypot( a, b ); + (void)p; + }, "shape" ); + } + SECTION( "fmin" ) + { + S2_REQUIRE_DEATH( []{ + feng::matrix a{ 2, 3 }; + feng::matrix b{ 2, 4 }; + auto const p = feng::fmin( a, b ); + (void)p; + }, "shape" ); + } + SECTION( "fma" ) + { + S2_REQUIRE_DEATH( []{ + feng::matrix a{ 2, 3 }; + feng::matrix b{ 2, 3 }; + feng::matrix c{ 3, 3 }; + auto const p = feng::fma( a, b, c ); + (void)p; + }, "shape" ); + } + SECTION( "matching shapes still work" ) + { + feng::matrix a{ 2, 3 }; + feng::matrix b{ 2, 3 }; + for ( std::size_t i = 0; i != 6; ++i ) { a.data()[i] = 3.0; b.data()[i] = 4.0; } + auto const h = feng::hypot( a, b ); + auto const s = feng::matrix_details::map( []( double x, double y, double z ){ return x + y + z; } )( a, b, a ); + for ( std::size_t i = 0; i != 6; ++i ) + { + REQUIRE( h.data()[i] == 5.0 ); + REQUIRE( s.data()[i] == 10.0 ); + } + } +} + +TEST_CASE( "S2 matrix operators check shapes", "[S2][S2-R4]" ) +{ + SECTION( "+=" ) + { + S2_REQUIRE_DEATH( []{ feng::matrix a{ 2, 3 }; feng::matrix const b{ 3, 2 }; a += b; }, "shape" ); + } + SECTION( "-=" ) + { + S2_REQUIRE_DEATH( []{ feng::matrix a{ 2, 3 }; feng::matrix const b{ 2, 2 }; a -= b; }, "shape" ); + } + SECTION( "/=" ) + { + S2_REQUIRE_DEATH( []{ feng::matrix a{ 2, 2 }; feng::matrix const b{ 3, 3 }; a /= b; }, "shape" ); + } + SECTION( "+" ) + { + S2_REQUIRE_DEATH( []{ feng::matrix const a{ 2, 3 }; feng::matrix const b{ 3, 2 }; auto const p = a + b; (void)p; }, "shape" ); + } + SECTION( "-" ) + { + S2_REQUIRE_DEATH( []{ feng::matrix const a{ 2, 3 }; feng::matrix const b{ 2, 4 }; auto const p = a - b; (void)p; }, "shape" ); + } + SECTION( "/" ) + { + S2_REQUIRE_DEATH( []{ feng::matrix const a{ 2, 2 }; feng::matrix const b{ 3, 3 }; auto const p = a / b; (void)p; }, "shape" ); + } + SECTION( "*= keeps its dims check with a shape message" ) + { + S2_REQUIRE_DEATH( []{ feng::matrix a{ 2, 3 }; feng::matrix const b{ 2, 3 }; a *= b; }, "shape" ); + } +} + +TEST_CASE( "S2 pooling rejects an unknown action", "[S2][S2-R4]" ) +{ + SECTION( "dim 2" ) + { + S2_REQUIRE_DEATH( []{ + feng::matrix m{ 4, 4 }; + auto const p = feng::pooling( m, 2, "median" ); + (void)p; + }, "pooling" ); + } + SECTION( "dim 1, before the identity early return" ) + { + S2_REQUIRE_DEATH( []{ + feng::matrix m{ 4, 4 }; + auto const p = feng::pooling( m, 1, "median" ); + (void)p; + }, "pooling" ); + } + SECTION( "dim 0, before the empty early return" ) + { + S2_REQUIRE_DEATH( []{ + feng::matrix m{ 4, 4 }; + auto const p = feng::pooling( m, 0, "median" ); + (void)p; + }, "pooling" ); + } + SECTION( "dims (1, 1), two-dim form" ) + { + S2_REQUIRE_DEATH( []{ + feng::matrix m{ 4, 4 }; + auto const p = feng::pooling( m, 1, 1, "median" ); + (void)p; + }, "pooling" ); + } +} + +namespace s2_r4_private +{ + inline feng::matrix filled( std::size_t r, std::size_t c ) + { + feng::matrix m{ r, c }; + for ( std::size_t i = 0; i != r; ++i ) + for ( std::size_t j = 0; j != c; ++j ) + m[i][j] = static_cast( ( 7 * i + 3 * j ) % 11 ) - 5.0; + return m; + } + + // A plain copy of the block [r0, r1) x [c0, c1) of m, made element by element. + inline feng::matrix block( feng::matrix const& m, std::size_t r0, std::size_t r1, std::size_t c0, std::size_t c1 ) + { + feng::matrix b{ r1 - r0, c1 - c0 }; + for ( std::size_t i = r0; i != r1; ++i ) + for ( std::size_t j = c0; j != c1; ++j ) + b[i - r0][j - c0] = m[i][j]; + return b; + } + + inline bool same( feng::matrix const& a, feng::matrix const& b ) + { + if ( a.row() != b.row() || a.col() != b.col() ) return false; + for ( std::size_t i = 0; i != a.row(); ++i ) + for ( std::size_t j = 0; j != a.col(); ++j ) + if ( a[i][j] != b[i][j] ) return false; + return true; + } +} + +TEST_CASE( "S2 self-assignment keeps the contents", "[S2][S2-R4]" ) +{ + auto a = s2_r4_private::filled( 5, 7 ); + auto const before = s2_r4_private::filled( 5, 7 ); + double const* const storage = a.data(); + + auto& self = a; + a = self; + REQUIRE( s2_r4_private::same( a, before ) ); + REQUIRE( a.data() == storage ); + + a.operator=< double, std::allocator >( self ); + REQUIRE( s2_r4_private::same( a, before ) ); + REQUIRE( a.data() == storage ); + + a.copy( self ); + REQUIRE( s2_r4_private::same( a, before ) ); + REQUIRE( a.data() == storage ); +} + +TEST_CASE( "S2 self multiply uses the old matrix", "[S2][S2-R4]" ) +{ + for ( std::size_t n : { std::size_t{ 3 }, std::size_t{ 18 }, std::size_t{ 32 } } ) + { + INFO( "n = " << n ); + auto a = s2_r4_private::filled( n, n ); + auto const old = s2_r4_private::filled( n, n ); + feng::matrix expected{ n, n }; + for ( std::size_t i = 0; i != n; ++i ) + for ( std::size_t j = 0; j != n; ++j ) + { + double s = 0.0; + for ( std::size_t k = 0; k != n; ++k ) s += old[i][k] * old[k][j]; + expected[i][j] = s; + } + a *= a; + REQUIRE( s2_r4_private::same( a, expected ) ); + } +} + +TEST_CASE( "S2 overlapping slice copy uses a snapshot", "[S2][S2-R4]" ) +{ + SECTION( "a view of the destination copied into an overlapping block" ) + { + auto m = s2_r4_private::filled( 5, 6 ); + auto const snapshot = s2_r4_private::block( m, 0, 3, 0, 4 ); + auto expected = s2_r4_private::filled( 5, 6 ); + expected.copy( snapshot, { 1, 4 }, { 1, 5 } ); + + m.copy( feng::make_view( m, { 0, 3 }, { 0, 4 } ), { 1, 4 }, { 1, 5 } ); + REQUIRE( s2_r4_private::same( m, expected ) ); + } + SECTION( "a view of the destination copied backwards into an overlapping block" ) + { + auto m = s2_r4_private::filled( 5, 6 ); + auto const snapshot = s2_r4_private::block( m, 1, 4, 2, 6 ); + auto expected = s2_r4_private::filled( 5, 6 ); + expected.copy( snapshot, { 0, 3 }, { 0, 4 } ); + + m.copy( feng::make_view( m, { 1, 4 }, { 2, 6 } ), { 0, 3 }, { 0, 4 } ); + REQUIRE( s2_r4_private::same( m, expected ) ); + } + SECTION( "the whole matrix copied onto itself" ) + { + auto m = s2_r4_private::filled( 4, 4 ); + auto const before = s2_r4_private::filled( 4, 4 ); + m.copy( m, { 0, 4 }, { 0, 4 } ); + REQUIRE( s2_r4_private::same( m, before ) ); + } + SECTION( "a disjoint matrix copies as before" ) + { + auto m = s2_r4_private::filled( 5, 6 ); + auto const src = s2_r4_private::filled( 2, 2 ); + m.copy( src, { 3, 5 }, { 4, 6 } ); + for ( std::size_t i = 0; i != 2; ++i ) + for ( std::size_t j = 0; j != 2; ++j ) + REQUIRE( m[3 + i][4 + j] == src[i][j] ); + } +} diff --git a/tests/cases/s3_alloc.hpp b/tests/cases/s3_alloc.hpp new file mode 100644 index 0000000..47edad8 --- /dev/null +++ b/tests/cases/s3_alloc.hpp @@ -0,0 +1,139 @@ +// S3-R2 (PR-5): test helpers for element lifetimes and stateful allocators (R04). +// - s3_alloc::tracked: a nothrow element type that counts live objects, constructions and destructions. +// - s3_alloc::registry: per resource id, the set of live allocations; a deallocation of a pointer the resource did +// not allocate (or with a different count) is recorded as a failure that the tests check. +// - s3_alloc::tracking_allocator: stateful (a resource id), is_always_equal false, == compares +// ids; the three propagation traits are template parameters so a test can loop over all 8 combinations. +#ifndef S3_ALLOC_HPP_INCLUDED +#define S3_ALLOC_HPP_INCLUDED + +#include +#include +#include +#include +#include +#include + +namespace s3_alloc +{ + struct tracked + { + static inline long live = 0; + static inline long constructed = 0; + static inline long destroyed = 0; + + static void reset() noexcept { live = 0; constructed = 0; destroyed = 0; } + + int value = 0; + + tracked() noexcept { ++live; ++constructed; } + tracked( int v ) noexcept : value{ v } { ++live; ++constructed; } + tracked( tracked const& other ) noexcept : value{ other.value } { ++live; ++constructed; } + tracked( tracked&& other ) noexcept : value{ other.value } { ++live; ++constructed; } + tracked& operator = ( tracked const& other ) noexcept { value = other.value; return *this; } + tracked& operator = ( tracked&& other ) noexcept { value = other.value; return *this; } + ~tracked() noexcept { --live; ++destroyed; } + + friend bool operator == ( tracked const& a, tracked const& b ) noexcept { return a.value == b.value; } + }; + + struct registry + { + // resource id -> (pointer -> element count) of its live allocations + static inline std::map< int, std::map< void const*, std::size_t > > live_allocations; + static inline std::vector< std::string > failures; + static inline long allocations = 0; + + static void reset() { live_allocations.clear(); failures.clear(); allocations = 0; } + + static std::size_t live( int id ) + { + auto const it = live_allocations.find( id ); + return it == live_allocations.end() ? 0 : it->second.size(); + } + + static std::size_t live_total() + { + std::size_t n = 0; + for ( auto const& [id, ptrs] : live_allocations ) n += ptrs.size(); + return n; + } + + static void on_allocate( int id, void const* p, std::size_t n ) + { + ++allocations; + live_allocations[id][p] = n; + } + + // Returns true when the pointer is live on `id` with count `n`; otherwise records a failure and forgets the + // pointer wherever it is live, so the memory is still released exactly once. + static bool on_deallocate( int id, void const* p, std::size_t n ) + { + auto& mine = live_allocations[id]; + auto const it = mine.find( p ); + if ( it != mine.end() && it->second == n ) + { + mine.erase( it ); + return true; + } + failures.push_back( "resource " + std::to_string( id ) + " deallocated a pointer it did not allocate (count " + std::to_string( n ) + ")" ); + for ( auto& [other, ptrs] : live_allocations ) ptrs.erase( p ); + return false; + } + }; + + template < typename T, bool POCCA, bool POCMA, bool POCS > + struct tracking_allocator + { + typedef T value_type; + typedef std::bool_constant< POCCA > propagate_on_container_copy_assignment; + typedef std::bool_constant< POCMA > propagate_on_container_move_assignment; + typedef std::bool_constant< POCS > propagate_on_container_swap; + typedef std::false_type is_always_equal; + + template < typename U > + struct rebind { typedef tracking_allocator< U, POCCA, POCMA, POCS > other; }; + + int id = 0; + + tracking_allocator() noexcept = default; + explicit tracking_allocator( int resource ) noexcept : id{ resource } {} + template < typename U > + tracking_allocator( tracking_allocator< U, POCCA, POCMA, POCS > const& other ) noexcept : id{ other.id } {} + + T* allocate( std::size_t n ) + { + T* const p = std::allocator< T >{}.allocate( n ); + registry::on_allocate( id, p, n ); + return p; + } + + void deallocate( T* p, std::size_t n ) noexcept + { + registry::on_deallocate( id, p, n ); + std::allocator< T >{}.deallocate( p, n ); + } + + friend bool operator == ( tracking_allocator const& a, tracking_allocator const& b ) noexcept { return a.id == b.id; } + }; + + // Resets the registry and the tracked counters; call at the start of each test. + inline void reset() + { + registry::reset(); + tracked::reset(); + } + + // Every allocation released by the resource that made it, every element destroyed exactly once. + // Uses Catch macros, so include this header after catch.hpp. + inline void require_balanced() + { + INFO( "first failure: " << ( registry::failures.empty() ? std::string{ "none" } : registry::failures.front() ) ); + REQUIRE( registry::failures.empty() ); + REQUIRE( registry::live_total() == 0 ); + REQUIRE( tracked::live == 0 ); + REQUIRE( tracked::constructed == tracked::destroyed ); + } +} + +#endif // S3_ALLOC_HPP_INCLUDED diff --git a/tests/cases/s3_r1.hpp b/tests/cases/s3_r1.hpp new file mode 100644 index 0000000..be47fa4 --- /dev/null +++ b/tests/cases/s3_r1.hpp @@ -0,0 +1,340 @@ +// S3-R1 (PR-5): private RAII storage and the size invariant. +#include "./s3_alloc.hpp" + +#include +#include + +namespace s3_r1 +{ + using s3_alloc::require_balanced; + + template< typename M > concept has_public_row_field = requires( M& m ) { m.row_; }; + template< typename M > concept has_public_col_field = requires( M& m ) { m.col_; }; + template< typename M > concept has_public_dat_field = requires( M& m ) { m.dat_; }; + template< typename M > concept has_public_allocator_field = requires( M& m ) { m.allocator_; }; + + static_assert( !has_public_row_field< feng::matrix > ); + static_assert( !has_public_col_field< feng::matrix > ); + static_assert( !has_public_dat_field< feng::matrix > ); + static_assert( !has_public_allocator_field< feng::matrix > ); + static_assert( !has_public_row_field< feng::matrix > && !has_public_dat_field< feng::matrix > ); + + template< typename M > + void require_consistent( M const& m ) + { + REQUIRE( m.size() == m.row() * m.col() ); + REQUIRE( static_cast< std::size_t >( m.end() - m.begin() ) == m.size() ); + std::size_t visited = 0; + for ( auto const& v : m ) { (void)v; ++visited; } + REQUIRE( visited == m.size() ); + } + + // Goes through a reference so the compiler sees no direct self-move. + template< typename M > + void self_move_assign( M& m, M& alias ) { m = std::move( alias ); } + + inline feng::matrix filled( std::size_t r, std::size_t c ) + { + feng::matrix m{ r, c }; + for ( std::size_t i = 0; i != r * c; ++i ) m.data()[i] = static_cast( i ) + 0.25; + return m; + } +} + +TEST_CASE( "S3 self-move-assign keeps a consistent 3x4 matrix", "[S3][S3-R1]" ) +{ + auto m = s3_r1::filled( 3, 4 ); + s3_r1::self_move_assign( m, m ); + REQUIRE( m.row() == 3 ); + REQUIRE( m.col() == 4 ); + REQUIRE( m.size() == 12 ); + REQUIRE( m.end() - m.begin() == 12 ); + std::size_t i = 0; + for ( auto const& v : m ) + { + REQUIRE( v == static_cast( i ) + 0.25 ); + ++i; + } + REQUIRE( i == 12 ); +} + +TEST_CASE( "S3 moved-from matrix is empty and consistent", "[S3][S3-R1]" ) +{ + SECTION( "move construction" ) + { + auto a = s3_r1::filled( 3, 4 ); + feng::matrix b{ std::move( a ) }; + REQUIRE( a.row() == 0 ); + REQUIRE( a.col() == 0 ); + REQUIRE( a.size() == 0 ); + s3_r1::require_consistent( a ); + REQUIRE( b.row() == 3 ); + REQUIRE( b.col() == 4 ); + REQUIRE( b[2][3] == 11.25 ); + s3_r1::require_consistent( b ); + } + SECTION( "move assignment" ) + { + auto a = s3_r1::filled( 3, 4 ); + auto b = s3_r1::filled( 2, 5 ); + b = std::move( a ); + REQUIRE( a.row() == 0 ); + REQUIRE( a.col() == 0 ); + REQUIRE( a.size() == 0 ); + s3_r1::require_consistent( a ); + REQUIRE( b.row() == 3 ); + REQUIRE( b.col() == 4 ); + REQUIRE( b[1][0] == 4.25 ); + s3_r1::require_consistent( b ); + a = s3_r1::filled( 1, 2 ); // a moved-from matrix is usable again + REQUIRE( a.size() == 2 ); + s3_r1::require_consistent( a ); + } +} + +namespace s3_r1 +{ + template < bool POCS > + using tracked_matrix = feng::matrix< s3_alloc::tracked, s3_alloc::tracking_allocator< s3_alloc::tracked, false, false, POCS > >; + template < typename T, bool POCS > + using counted_matrix = feng::matrix< T, s3_alloc::tracking_allocator< T, false, false, POCS > >; + + template < typename M > + void fill_tracked( M& m, int base ) + { + for ( std::size_t i = 0; i != m.size(); ++i ) m.data()[i].value = base + static_cast< int >( i ); + } + + // Element-lifetime and resource balance across the shape-changing operations, the allocator kept throughout. + template < bool POCS > + void tracked_operations() + { + typedef tracked_matrix< POCS > matrix_type; + typedef typename matrix_type::allocator_type allocator_type; + s3_alloc::reset(); + { + matrix_type m{ allocator_type{ 3 }, 3, 4 }; + fill_tracked( m, 0 ); + require_consistent( m ); + + m.resize( 5, 6 ); + require_consistent( m ); + REQUIRE( m.get_allocator().id == 3 ); + m.resize( 6, 5 ); // same size: reshapes + require_consistent( m ); + fill_tracked( m, 0 ); + m.reshape( 5, 6 ); + require_consistent( m ); + REQUIRE( m.row() == 5 ); + + auto const c = m.clone( 1, 4, 2, 5 ); + require_consistent( c ); + REQUIRE( c.get_allocator().id == 3 ); + REQUIRE( c[0][0].value == 1 * 6 + 2 ); + m.clone( c, 0, 2, 0, 3 ); + require_consistent( m ); + REQUIRE( m.get_allocator().id == 3 ); + REQUIRE( m.row() == 2 ); + + m.shrink_to_size( 4, 4 ); + require_consistent( m ); + REQUIRE( m.get_allocator().id == 3 ); + REQUIRE( m[1][2].value == c[1][2].value ); + REQUIRE( m[3][3].value == 0 ); + + auto const t = m.transpose(); + require_consistent( t ); + REQUIRE( t.get_allocator().id == 3 ); + REQUIRE( t[2][1].value == m[1][2].value ); + + auto const same = m.template astype< s3_alloc::tracked >(); + require_consistent( same ); + REQUIRE( same.get_allocator().id == 3 ); + + // copy(rhs) on unequal resources: the destination keeps its own allocator (F04). + matrix_type dst{ allocator_type{ 4 }, 1, 2 }; + dst.copy( t ); + require_consistent( dst ); + REQUIRE( dst.get_allocator().id == 4 ); + REQUIRE( dst.row() == 4 ); + REQUIRE( dst[2][1].value == t[2][1].value ); + dst.copy( m ); // same shape: no reallocation + require_consistent( dst ); + REQUIRE( dst.get_allocator().id == 4 ); + REQUIRE( dst[1][2].value == m[1][2].value ); + REQUIRE( s3_alloc::registry::live( 4 ) == 1 ); + + m.clear(); + require_consistent( m ); + REQUIRE( m.get_allocator().id == 3 ); + } + require_balanced(); + REQUIRE( s3_alloc::registry::live( 3 ) == 0 ); + REQUIRE( s3_alloc::registry::live( 4 ) == 0 ); + } + + // Arithmetic, astype and converting constructors on a stateful allocator: results use the operand's resource. + template < bool POCS > + void counted_arithmetic() + { + typedef counted_matrix< double, POCS > matrix_type; + typedef typename matrix_type::allocator_type allocator_type; + s3_alloc::reset(); + { + matrix_type m{ allocator_type{ 5 }, 2, 3 }; + for ( std::size_t i = 0; i != m.size(); ++i ) m.data()[i] = static_cast( i ) + 0.5; + matrix_type sq{ allocator_type{ 5 }, 2, 2 }; + sq[0][0] = 2.0; sq[0][1] = 1.0; sq[1][0] = 1.0; sq[1][1] = 3.0; + auto const t = m.transpose(); + for ( auto const& r : { m + m, m - m, m + 1.0, 1.0 + m, m - 1.0, 1.0 - m, m * 2.0, 2.0 * m, m / 2.0, sq * m, sq / sq } ) + { + require_consistent( r ); + REQUIRE( r.get_allocator().id == 5 ); + } + auto p = m * t; + require_consistent( p ); + REQUIRE( p.get_allocator().id == 5 ); + p *= sq; + require_consistent( p ); + REQUIRE( p.get_allocator().id == 5 ); + p /= sq; + require_consistent( p ); + REQUIRE( p.get_allocator().id == 5 ); + p += 1.0; p -= p; p *= 2.0; p /= 2.0; + require_consistent( p ); + REQUIRE( p.get_allocator().id == 5 ); + + auto const f = m.template astype< float >(); + require_consistent( f ); + REQUIRE( f.get_allocator().id == 5 ); + REQUIRE( f[1][2] == 5.5f ); + + counted_matrix< float, POCS > const g{ m }; + require_consistent( g ); + REQUIRE( g.get_allocator().id == 5 ); + REQUIRE( g[1][1] == 4.5f ); + + counted_matrix< float, POCS > h{ typename counted_matrix< float, POCS >::allocator_type{ 6 }, 1, 1 }; + h.copy( m ); + require_consistent( h ); + REQUIRE( h.get_allocator().id == 6 ); + REQUIRE( h[0][2] == 2.5f ); + } + require_balanced(); + } +} + +TEST_CASE( "S3 size equals rows times cols after every operation", "[S3][S3-R1]" ) +{ + SECTION( "std::allocator, double" ) + { + auto m = s3_r1::filled( 3, 4 ); + s3_r1::require_consistent( m ); + + feng::matrix copy{ m }; + s3_r1::require_consistent( copy ); + REQUIRE( copy == m ); + + feng::matrix assigned{ 7, 7 }; + assigned = m; + s3_r1::require_consistent( assigned ); + REQUIRE( assigned == m ); + + feng::matrix& alias = assigned; + assigned = alias; + s3_r1::require_consistent( assigned ); + REQUIRE( assigned == m ); + + feng::matrix other{ 2, 2, 1.0 }; + other.swap( copy ); + s3_r1::require_consistent( other ); + s3_r1::require_consistent( copy ); + REQUIRE( copy.row() == 2 ); + REQUIRE( other.row() == 3 ); + + m.resize( 5, 6 ); + s3_r1::require_consistent( m ); + m.reshape( 6, 5 ); + s3_r1::require_consistent( m ); + REQUIRE( m.row() == 6 ); + + auto const c = m.clone( 1, 4, 2, 5 ); + s3_r1::require_consistent( c ); + REQUIRE( c.size() == 9 ); + + m.shrink_to_size( 2, 3 ); + s3_r1::require_consistent( m ); + + auto const f = m.astype(); + s3_r1::require_consistent( f ); + + auto const t = m.transpose(); + s3_r1::require_consistent( t ); + REQUIRE( t.row() == 3 ); + + auto const sum = m + m; + s3_r1::require_consistent( sum ); + auto const prod = m * t; + s3_r1::require_consistent( prod ); + REQUIRE( prod.row() == 2 ); + REQUIRE( prod.col() == 2 ); + + auto const diff = m - m; + s3_r1::require_consistent( diff ); + REQUIRE( diff[1][2] == 0.0 ); + + feng::matrix const sq{ 2, 2, { 2.0, 1.0, 1.0, 3.0 } }; + auto const quot = sq / sq; + s3_r1::require_consistent( quot ); + REQUIRE( quot.row() == 2 ); + + for ( auto const& s : { m + 1.0, 1.0 + m, m - 1.0, 1.0 - m, m * 2.0, 2.0 * m, m / 2.0 } ) + { + s3_r1::require_consistent( s ); + REQUIRE( s.row() == 2 ); + REQUIRE( s.col() == 3 ); + } + REQUIRE( ( m * 2.0 )[1][2] == 2.0 * m[1][2] ); + + auto compound = m; + compound += m; s3_r1::require_consistent( compound ); + compound -= m; s3_r1::require_consistent( compound ); + compound += 1.0; s3_r1::require_consistent( compound ); + compound -= 1.0; s3_r1::require_consistent( compound ); + compound *= 2.0; s3_r1::require_consistent( compound ); + compound /= 2.0; s3_r1::require_consistent( compound ); + REQUIRE( compound == m ); + compound *= t; s3_r1::require_consistent( compound ); + REQUIRE( compound.row() == 2 ); + REQUIRE( compound.col() == 2 ); + compound /= sq; s3_r1::require_consistent( compound ); + REQUIRE( compound.col() == 2 ); + + feng::matrix const converted{ m }; + s3_r1::require_consistent( converted ); + REQUIRE( converted.row() == 2 ); + REQUIRE( converted[1][2] == static_cast( m[1][2] ) ); + feng::matrix const to_int{ m }; + s3_r1::require_consistent( to_int ); + + feng::matrix copied{ 4, 4 }; + copied.copy( m ); + s3_r1::require_consistent( copied ); + REQUIRE( copied.row() == 2 ); + REQUIRE( copied.col() == 3 ); + REQUIRE( copied[0][1] == static_cast( m[0][1] ) ); + feng::matrix assigned_from_double{ 1, 1 }; + assigned_from_double = m; + s3_r1::require_consistent( assigned_from_double ); + REQUIRE( assigned_from_double.size() == 6 ); + + m.clear(); + REQUIRE( m.row() == 0 ); + REQUIRE( m.col() == 0 ); + s3_r1::require_consistent( m ); + } + SECTION( "tracked elements, POCS false" ) { s3_r1::tracked_operations< false >(); } + SECTION( "tracked elements, POCS true" ) { s3_r1::tracked_operations< true >(); } + SECTION( "stateful allocator arithmetic, POCS false" ) { s3_r1::counted_arithmetic< false >(); } + SECTION( "stateful allocator arithmetic, POCS true" ) { s3_r1::counted_arithmetic< true >(); } +} diff --git a/tests/cases/s3_r2.hpp b/tests/cases/s3_r2.hpp new file mode 100644 index 0000000..159c04e --- /dev/null +++ b/tests/cases/s3_r2.hpp @@ -0,0 +1,226 @@ +// S3-R2 (PR-5): element lifetimes and stateful allocators (R04). +#include "./s3_alloc.hpp" + +#include +#include +#include + +namespace s3_r2 +{ + template < bool POCCA, bool POCMA, bool POCS > + using tracked_matrix = feng::matrix< s3_alloc::tracked, s3_alloc::tracking_allocator< s3_alloc::tracked, POCCA, POCMA, POCS > >; + + using s3_alloc::require_balanced; + + // Copy-assign a 3x4 matrix on resource 1 into a 2x5 matrix on resource 2 (unequal allocators), then destroy both. + template < bool POCCA, bool POCMA, bool POCS > + void copy_assign_unequal() + { + typedef tracked_matrix< POCCA, POCMA, POCS > matrix_type; + typedef typename matrix_type::allocator_type allocator_type; + s3_alloc::reset(); + { + matrix_type const src{ allocator_type{ 1 }, 3, 4, s3_alloc::tracked{ 7 } }; + matrix_type dst{ allocator_type{ 2 }, 2, 5, s3_alloc::tracked{ 9 } }; + REQUIRE( src.get_allocator() != dst.get_allocator() ); + dst = src; + REQUIRE( dst.get_allocator().id == ( POCCA ? 1 : 2 ) ); + REQUIRE( src.get_allocator().id == 1 ); + REQUIRE( dst.row() == 3 ); + REQUIRE( dst.col() == 4 ); + REQUIRE( dst.size() == 12 ); + for ( auto const& e : dst ) REQUIRE( e.value == 7 ); + REQUIRE( s3_alloc::registry::live( 1 ) == ( POCCA ? 2 : 1 ) ); + REQUIRE( s3_alloc::registry::live( 2 ) == ( POCCA ? 0 : 1 ) ); + } + REQUIRE( s3_alloc::registry::live( 1 ) == 0 ); + REQUIRE( s3_alloc::registry::live( 2 ) == 0 ); + require_balanced(); + } +} + +TEST_CASE( "S3 unequal non-propagating copy-assign frees each resource's own allocations", "[S3][S3-R2]" ) +{ + SECTION( "POCMA false, POCS false" ) { s3_r2::copy_assign_unequal< false, false, false >(); } + SECTION( "POCMA false, POCS true" ) { s3_r2::copy_assign_unequal< false, false, true >(); } + SECTION( "POCMA true, POCS false" ) { s3_r2::copy_assign_unequal< false, true, false >(); } + SECTION( "POCMA true, POCS true" ) { s3_r2::copy_assign_unequal< false, true, true >(); } +} + +namespace s3_r2 +{ + // Fills m with base, base+1, ... in row order. + template < typename M > + void fill_from( M& m, int base ) + { + for ( std::size_t i = 0; i != m.size(); ++i ) m.data()[i].value = base + static_cast< int >( i ); + } + + // m is r x c with size r*c, end-begin == size, and holds base, base+1, ... in row order. + template < typename M > + void require_holds( M const& m, std::size_t r, std::size_t c, int base ) + { + REQUIRE( m.row() == r ); + REQUIRE( m.col() == c ); + REQUIRE( m.size() == r * c ); + REQUIRE( static_cast< std::size_t >( m.end() - m.begin() ) == m.size() ); + int expected = base; + for ( auto const& e : m ) REQUIRE( e.value == expected++ ); + REQUIRE( expected == base + static_cast< int >( r * c ) ); + } + + template < typename M > + void require_moved_from( M const& m ) + { + REQUIRE( m.row() == 0 ); + REQUIRE( m.col() == 0 ); + REQUIRE( m.size() == 0 ); + REQUIRE( m.begin() == m.end() ); + } + + // Goes through references so the compiler sees no direct self-assignment. + template < typename M > void self_copy_assign( M& m, M const& alias ) { m = alias; } + template < typename M > void self_move_assign( M& m, M& alias ) { m = std::move( alias ); } + + // a: 3x4 on resource 1, holding 0..11; b: 2x5 on resource `other`, holding 100..109. Every scenario runs in + // its own scope and ends with require_balanced(). + template < bool POCCA, bool POCMA, bool POCS > + void every_operation( int const other ) + { + typedef tracked_matrix< POCCA, POCMA, POCS > matrix_type; + typedef typename matrix_type::allocator_type allocator_type; + bool const equal = other == 1; + auto make_a = []{ matrix_type m{ allocator_type{ 1 }, 3, 4 }; fill_from( m, 0 ); return m; }; + auto make_b = [other]{ matrix_type m{ allocator_type{ other }, 2, 5 }; fill_from( m, 100 ); return m; }; + + SECTION( "copy-construct" ) + { + s3_alloc::reset(); + { + auto const a = make_a(); + matrix_type c{ a }; + REQUIRE( c.get_allocator() == std::allocator_traits< allocator_type >::select_on_container_copy_construction( a.get_allocator() ) ); + REQUIRE( c.get_allocator().id == 1 ); + require_holds( c, 3, 4, 0 ); + require_holds( a, 3, 4, 0 ); + } + require_balanced(); + } + SECTION( "copy-assign" ) + { + s3_alloc::reset(); + { + auto const a = make_a(); + auto b = make_b(); + b = a; + REQUIRE( b.get_allocator().id == ( POCCA ? 1 : other ) ); + REQUIRE( a.get_allocator().id == 1 ); + require_holds( b, 3, 4, 0 ); + require_holds( a, 3, 4, 0 ); + } + require_balanced(); + } + SECTION( "move-construct" ) + { + s3_alloc::reset(); + { + auto a = make_a(); + matrix_type c{ std::move( a ) }; + REQUIRE( c.get_allocator().id == 1 ); + require_holds( c, 3, 4, 0 ); + require_moved_from( a ); + } + require_balanced(); + } + SECTION( "move-assign" ) + { + s3_alloc::reset(); + { + auto a = make_a(); + auto b = make_b(); + b = std::move( a ); + REQUIRE( b.get_allocator().id == ( POCMA ? 1 : other ) ); + require_holds( b, 3, 4, 0 ); + require_moved_from( a ); + a = make_a(); // a moved-from matrix is usable again + require_holds( a, 3, 4, 0 ); + } + require_balanced(); + } + SECTION( "member swap" ) + { + s3_alloc::reset(); + { + auto a = make_a(); + auto b = make_b(); + a.swap( b ); + REQUIRE( a.get_allocator().id == ( POCS ? other : 1 ) ); + REQUIRE( b.get_allocator().id == ( POCS ? 1 : other ) ); + require_holds( a, 2, 5, 100 ); + require_holds( b, 3, 4, 0 ); + if ( !equal ) + { + REQUIRE( s3_alloc::registry::live( 1 ) == 1 ); + REQUIRE( s3_alloc::registry::live( other ) == 1 ); + } + } + require_balanced(); + } + SECTION( "free swap" ) + { + s3_alloc::reset(); + { + auto a = make_a(); + auto b = make_b(); + using std::swap; + swap( a, b ); + REQUIRE( a.get_allocator().id == ( POCS ? other : 1 ) ); + REQUIRE( b.get_allocator().id == ( POCS ? 1 : other ) ); + require_holds( a, 2, 5, 100 ); + require_holds( b, 3, 4, 0 ); + } + require_balanced(); + } + SECTION( "self-copy-assign" ) + { + s3_alloc::reset(); + { + auto a = make_a(); + self_copy_assign( a, a ); + REQUIRE( a.get_allocator().id == 1 ); + require_holds( a, 3, 4, 0 ); + } + require_balanced(); + } + SECTION( "self-move-assign" ) + { + s3_alloc::reset(); + { + auto a = make_a(); + self_move_assign( a, a ); + REQUIRE( a.get_allocator().id == 1 ); + require_holds( a, 3, 4, 0 ); + } + require_balanced(); + } + } + + template < bool POCCA, bool POCMA, bool POCS > + void every_resource_pair() + { + SECTION( "equal resources" ) { every_operation< POCCA, POCMA, POCS >( 1 ); } + SECTION( "unequal resources" ) { every_operation< POCCA, POCMA, POCS >( 2 ); } + } +} + +TEST_CASE( "S3 every propagation combination balances each resource", "[S3][S3-R2]" ) +{ + SECTION( "POCCA 0, POCMA 0, POCS 0" ) { s3_r2::every_resource_pair< false, false, false >(); } + SECTION( "POCCA 0, POCMA 0, POCS 1" ) { s3_r2::every_resource_pair< false, false, true >(); } + SECTION( "POCCA 0, POCMA 1, POCS 0" ) { s3_r2::every_resource_pair< false, true, false >(); } + SECTION( "POCCA 0, POCMA 1, POCS 1" ) { s3_r2::every_resource_pair< false, true, true >(); } + SECTION( "POCCA 1, POCMA 0, POCS 0" ) { s3_r2::every_resource_pair< true, false, false >(); } + SECTION( "POCCA 1, POCMA 0, POCS 1" ) { s3_r2::every_resource_pair< true, false, true >(); } + SECTION( "POCCA 1, POCMA 1, POCS 0" ) { s3_r2::every_resource_pair< true, true, false >(); } + SECTION( "POCCA 1, POCMA 1, POCS 1" ) { s3_r2::every_resource_pair< true, true, true >(); } +} diff --git a/tests/cases/s3_r3.hpp b/tests/cases/s3_r3.hpp new file mode 100644 index 0000000..42a96f1 --- /dev/null +++ b/tests/cases/s3_r3.hpp @@ -0,0 +1,78 @@ +// S3-R3 (PR-5): the element-type constraint feng::matrix_element. +#include +#include +#include + +static_assert( feng::matrix_element< int > ); +static_assert( feng::matrix_element< double > ); +static_assert( feng::matrix_element< std::uint8_t > ); +static_assert( feng::matrix_element< std::complex< float > > ); +static_assert( feng::matrix_element< std::complex< double > > ); +static_assert( feng::matrix_element< std::complex< long double > > ); +static_assert( !feng::matrix_element< bool > ); +static_assert( !feng::matrix_element< std::string > ); + +namespace s3_r3 +{ + template < typename... Ts > + inline constexpr bool all_elements = ( feng::matrix_element< Ts > && ... ); + + struct throwing_copy + { + throwing_copy() noexcept = default; + throwing_copy( throwing_copy const& ) noexcept( false ) {} + throwing_copy( throwing_copy&& ) noexcept = default; + throwing_copy& operator=( throwing_copy const& ) noexcept = default; + throwing_copy& operator=( throwing_copy&& ) noexcept = default; + }; + + struct throwing_move + { + throwing_move() noexcept = default; + throwing_move( throwing_move const& ) noexcept = default; + throwing_move( throwing_move&& ) noexcept( false ) {} + throwing_move& operator=( throwing_move const& ) noexcept = default; + throwing_move& operator=( throwing_move&& ) noexcept = default; + }; + + struct throwing_destructor + { + ~throwing_destructor() noexcept( false ) {} + }; + + // D-018: the size constructors default-construct elements. + struct no_default + { + explicit no_default( int ) noexcept {} + }; + + struct throwing_default + { + throwing_default() noexcept( false ) {} + }; +} + +// Every standard arithmetic type except bool, plus the fixed-width aliases. +static_assert( s3_r3::all_elements< char, signed char, unsigned char, short, unsigned short, int, unsigned int, long, + unsigned long, long long, unsigned long long, float, double, long double, + std::size_t, std::int8_t, std::int16_t, std::int32_t, std::int64_t, std::uint8_t, + std::uint16_t, std::uint32_t, std::uint64_t > ); +static_assert( s3_r3::all_elements< std::complex< float >, std::complex< double >, std::complex< long double > > ); +static_assert( !feng::matrix_element< s3_r3::throwing_copy > ); +static_assert( !feng::matrix_element< s3_r3::throwing_move > ); +static_assert( !feng::matrix_element< s3_r3::throwing_destructor > ); +static_assert( !feng::matrix_element< s3_r3::no_default > ); +static_assert( !feng::matrix_element< s3_r3::throwing_default > ); +static_assert( !feng::matrix_element< int const > ); +static_assert( !feng::matrix_element< int& > ); + +TEST_CASE( "S3 matrix_element accepts arithmetic and complex types and rejects bool and std::string", "[S3][S3-R3]" ) +{ + static_assert( feng::matrix_element< int > && feng::matrix_element< double > && feng::matrix_element< std::uint8_t > ); + static_assert( feng::matrix_element< std::complex< float > > && feng::matrix_element< std::complex< double > > && feng::matrix_element< std::complex< long double > > ); + static_assert( !feng::matrix_element< bool > && !feng::matrix_element< std::string > ); + feng::matrix< std::uint8_t > const mask{ 2, 3 }; + feng::matrix< std::complex< long double > > const z{ 3, 2 }; + REQUIRE( mask.size() == 6 ); + REQUIRE( z.size() == 6 ); +} diff --git a/tests/cases/s3_r4.hpp b/tests/cases/s3_r4.hpp new file mode 100644 index 0000000..316f2c9 --- /dev/null +++ b/tests/cases/s3_r4.hpp @@ -0,0 +1,51 @@ +// S3-R4 (PR-5): matrix from a view copies exactly the viewed rectangle (F03). +#include + +namespace s3_r4 +{ + inline feng::matrix distinct( std::size_t r, std::size_t c ) + { + feng::matrix m{ r, c }; + for ( std::size_t i = 0; i != r; ++i ) + for ( std::size_t j = 0; j != c; ++j ) + m[i][j] = static_cast( 100 * i + j ) + 0.5; + return m; + } + + inline void require_rectangle( feng::matrix const& m, std::size_t r0, std::size_t r1, std::size_t c0, std::size_t c1 ) + { + auto const view = feng::make_view( m, { r0, r1 }, { c0, c1 } ); + feng::matrix const out{ view }; + REQUIRE( out.row() == r1 - r0 ); + REQUIRE( out.col() == c1 - c0 ); + REQUIRE( out.size() == out.row() * out.col() ); + REQUIRE( static_cast< std::size_t >( out.end() - out.begin() ) == out.size() ); + auto it = out.begin(); + for ( std::size_t i = r0; i != r1; ++i ) + for ( std::size_t j = c0; j != c1; ++j ) + REQUIRE( *it++ == m[i][j] ); + } +} + +TEST_CASE( "S3 matrix from an offset view copies exactly the rectangle", "[S3][S3-R4]" ) +{ + auto const m = s3_r4::distinct( 5, 6 ); + SECTION( "2x3 view at offset (1, 2)" ) + { + auto const view = feng::make_view( m, { 1UL, 3UL }, { 2UL, 5UL } ); + feng::matrix const out{ view }; + REQUIRE( out.row() == 2 ); + REQUIRE( out.col() == 3 ); + REQUIRE( out.size() == 6 ); + double const expected[] = { 102.5, 103.5, 104.5, 202.5, 203.5, 204.5 }; + std::size_t k = 0; + for ( auto const& v : out ) REQUIRE( v == expected[k++] ); + REQUIRE( k == 6 ); + } + SECTION( "1x1 view" ) { s3_r4::require_rectangle( m, 3, 4, 4, 5 ); } + SECTION( "full-size view" ) { s3_r4::require_rectangle( m, 0, 5, 0, 6 ); } + SECTION( "last-column view" ) { s3_r4::require_rectangle( m, 0, 5, 5, 6 ); } + SECTION( "first-row view" ) { s3_r4::require_rectangle( m, 0, 1, 0, 6 ); } + SECTION( "interior 1xN view" ) { s3_r4::require_rectangle( m, 4, 5, 1, 5 ); } + SECTION( "interior Nx1 view" ) { s3_r4::require_rectangle( m, 1, 4, 0, 1 ); } +} diff --git a/tests/cases/s3_r5.hpp b/tests/cases/s3_r5.hpp new file mode 100644 index 0000000..616570b --- /dev/null +++ b/tests/cases/s3_r5.hpp @@ -0,0 +1,263 @@ +// S3-R5 (PR-2): noexcept special members and the allocation check before allocating. +#include "./s2_death.hpp" + +#include +#include + +namespace s3_r5 +{ + typedef feng::matrix< double > md; + static_assert( std::is_nothrow_default_constructible_v< md > ); + static_assert( std::is_nothrow_copy_constructible_v< md > ); + static_assert( std::is_nothrow_move_constructible_v< md > ); + static_assert( std::is_nothrow_copy_assignable_v< md > ); + static_assert( std::is_nothrow_move_assignable_v< md > ); + static_assert( std::is_nothrow_destructible_v< md > ); + static_assert( noexcept( std::declval< md& >().swap( std::declval< md& >() ) ) ); + static_assert( std::is_nothrow_swappable_v< md > ); +} + +TEST_CASE( "S3 huge resize aborts before allocating", "[S3][S3-R5]" ) +{ + S2_REQUIRE_DEATH( []{ feng::matrix m; m.resize( 1ULL << 61, 8 ); }, "size" ); +} + +// S3-R5 (D-012): every public API family is noexcept; one or more calls per S1 API family (docs/stages/S1/spec.md), +// plus the special members, swap, resize, clone, astype, I/O and callback-taking functions. String arguments are +// passed as lvalues and defaulted string parameters explicitly: building a std::string at the call site can throw, +// which is the caller's expression, not the library's. +namespace s3_r5_api +{ + using std::declval; + typedef feng::matrix< double > md; + typedef feng::matrix< std::complex< double > > mc; + typedef feng::matrix< float, std::allocator< float > > mf; + typedef feng::matrix_view< double, std::allocator< double > > vd; + inline auto const unary = []( double& x ) noexcept { x += 1.0; }; + + // construction + static_assert( noexcept( md{} ) ); + static_assert( noexcept( md{ 3, 4 } ) ); + static_assert( noexcept( md{ 3, 4, 1.0 } ) ); + static_assert( noexcept( md{ declval< md const& >() } ) ); + static_assert( noexcept( md{ declval< md&& >() } ) ); + static_assert( noexcept( feng::zeros< double >( 3, 4 ) ) ); + static_assert( noexcept( feng::ones< double >( 3 ) ) ); + static_assert( noexcept( feng::eye< double >( 3 ) ) ); + static_assert( noexcept( feng::arange< double >( 0, 4 ) ) ); + static_assert( noexcept( feng::linspace( 0.0, 1.0 ) ) ); + static_assert( noexcept( feng::magic( 4 ) ) ); + static_assert( noexcept( feng::hilbert< double >( 3 ) ) ); + static_assert( noexcept( feng::rand< double >( 3 ) ) ); + static_assert( noexcept( feng::zeros_like( declval< md const& >() ) ) ); + static_assert( noexcept( feng::blkdiag( declval< md const& >(), declval< md const& >() ) ) ); + static_assert( noexcept( feng::make_diag( declval< md const& >() ) ) ); + // access + static_assert( noexcept( declval< md& >()[ 0 ] ) ); + static_assert( noexcept( declval< md& >()[ 0 ][ 0 ] ) ); + static_assert( noexcept( declval< md const& >().item() ) ); + static_assert( noexcept( declval< md& >().data() ) ); + static_assert( noexcept( declval< md const& >().shape() ) ); + static_assert( noexcept( declval< md const& >().size() ) ); + static_assert( noexcept( declval< md& >().begin() ) ); + static_assert( noexcept( declval< md& >().row_begin( 0 ) ) ); + static_assert( noexcept( declval< md& >().col_begin( 0 ) ) ); + static_assert( noexcept( declval< md& >().diag_begin() ) ); + // views + static_assert( noexcept( feng::make_view( declval< md const& >(), { 0, 1 }, { 0, 1 } ) ) ); + static_assert( noexcept( declval< md const& >().clone( 0, 1, 0, 1 ) ) ); + static_assert( noexcept( declval< md& >().clone( declval< md const& >(), 0, 1, 0, 1 ) ) ); + static_assert( noexcept( md{ declval< vd const& >() } ) ); + // shape + static_assert( noexcept( declval< md& >().reshape( 2, 2 ) ) ); + static_assert( noexcept( declval< md& >().resize( 2, 2 ) ) ); + static_assert( noexcept( declval< md const& >().transpose() ) ); + static_assert( noexcept( feng::transpose( declval< md const& >() ) ) ); + static_assert( noexcept( feng::fliplr( declval< md const& >() ) ) ); + static_assert( noexcept( feng::tril( declval< md const& >() ) ) ); + static_assert( noexcept( declval< md& >().clear() ) ); + static_assert( noexcept( declval< md& >().shrink_to_size( 1, 1 ) ) ); + // arithmetic + static_assert( noexcept( declval< md const& >() + declval< md const& >() ) ); + static_assert( noexcept( declval< md const& >() - 1.0 ) ); + static_assert( noexcept( 2.0 * declval< md const& >() ) ); + static_assert( noexcept( declval< md const& >() * declval< md const& >() ) ); + static_assert( noexcept( declval< md const& >() / 2.0 ) ); + static_assert( noexcept( declval< md& >() += 1.0 ) ); + static_assert( noexcept( -declval< md const& >() ) ); + static_assert( noexcept( declval< md const& >() ^ 2 ) ); + // elementwise + static_assert( noexcept( feng::sin( declval< md const& >() ) ) ); + static_assert( noexcept( feng::abs( declval< md const& >() ) ) ); + static_assert( noexcept( feng::pow( declval< md const& >(), 2.0 ) ) ); + static_assert( noexcept( feng::real( declval< mc const& >() ) ) ); + static_assert( noexcept( feng::clip( 0.0, 1.0 )( declval< md const& >() ) ) ); + static_assert( noexcept( declval< md const& >().astype< float >() ) ); + // reductions + static_assert( noexcept( feng::sum( declval< md const& >() ) ) ); + static_assert( noexcept( feng::mean( declval< md const& >() ) ) ); + static_assert( noexcept( feng::variance( declval< md const& >() ) ) ); + static_assert( noexcept( feng::max( declval< md const& >() ) ) ); + static_assert( noexcept( declval< md const& >().minmax() ) ); + static_assert( noexcept( feng::norm_1( declval< md const& >() ) ) ); + static_assert( noexcept( feng::tr( declval< md const& >() ) ) ); + static_assert( noexcept( feng::isequal( declval< md const& >(), declval< md const& >() ) ) ); + static_assert( noexcept( feng::is_symmetric( declval< md const& >() ) ) ); + // linalg + static_assert( noexcept( declval< md const& >().det() ) ); + static_assert( noexcept( feng::det( declval< md const& >() ) ) ); + static_assert( noexcept( feng::inverse( declval< md const& >() ) ) ); + static_assert( noexcept( feng::lu_solver( declval< md const& >(), declval< md const& >() ) ) ); + static_assert( noexcept( feng::pinv( declval< md const& >() ) ) ); + static_assert( noexcept( feng::expm( declval< md const& >() ) ) ); + // signal + static_assert( noexcept( feng::conv( declval< md const& >(), declval< md const& >() ) ) ); + static_assert( noexcept( feng::fft( declval< mc const& >() ) ) ); + static_assert( noexcept( feng::fftshift( declval< md const& >() ) ) ); + static_assert( noexcept( feng::pooling( declval< md const& >(), 2, declval< std::string const& >() ) ) ); + // io + static_assert( noexcept( declval< md const& >().save_as_txt( "x" ) ) ); + static_assert( noexcept( declval< md& >().load_txt( "x" ) ) ); + static_assert( noexcept( declval< md const& >().save_as_binary( "x" ) ) ); + static_assert( noexcept( declval< md& >().load_binary( declval< std::string const& >() ) ) ); + static_assert( noexcept( declval< md& >().load_npy( "x" ) ) ); + static_assert( noexcept( declval< md const& >().save_as_npy( "x" ) ) ); + static_assert( noexcept( declval< std::ostream& >() << declval< md const& >() ) ); + static_assert( noexcept( declval< std::istream& >() >> declval< md& >() ) ); + static_assert( noexcept( feng::disp( declval< md const& >() ) ) ); + // image + static_assert( noexcept( declval< md const& >().save_as_bmp( declval< std::string const& >(), declval< std::string const& >() ) ) ); + static_assert( noexcept( declval< md const& >().save_as_png( declval< std::string const& >(), declval< std::string const& >() ) ) ); + static_assert( noexcept( declval< md const& >().save_as_pgm( "x" ) ) ); + static_assert( noexcept( declval< md const& >().plot( declval< std::string const& >(), declval< std::string const& >() ) ) ); + static_assert( noexcept( feng::load_bmp( declval< std::string const& >() ) ) ); + static_assert( noexcept( feng::save_as_bmp( declval< std::string const& >(), declval< md const& >(), declval< std::string const& >() ) ) ); + // allocator + static_assert( noexcept( mf{ std::allocator< float >{}, 2, 2 } ) ); + static_assert( noexcept( declval< md const& >().get_allocator() ) ); + // special members, swap, callbacks + static_assert( noexcept( declval< md& >() = declval< md const& >() ) ); + static_assert( noexcept( declval< md& >() = declval< md&& >() ) ); + static_assert( noexcept( declval< md& >().~md() ) ); + static_assert( noexcept( declval< md& >().swap( declval< md& >() ) ) ); + static_assert( noexcept( swap( declval< md& >(), declval< md& >() ) ) ); + static_assert( noexcept( declval< md& >().apply( unary ) ) ); + static_assert( noexcept( feng::matrix_details::for_each( declval< double* >(), declval< double* >(), unary ) ) ); +} + +TEST_CASE( "S3 every public API family is noexcept", "[S3][S3-R5]" ) +{ + // The namespace-scope static_asserts above are the check; this case records that they compiled. + REQUIRE( noexcept( s3_r5_api::md{ 2, 2 } ) ); + REQUIRE( noexcept( std::declval< s3_r5_api::md& >().resize( 3, 3 ) ) ); +} + +// S3-R5 (D-012): the size check runs on the allocator that will own the storage, before allocate() is called. +// budget_allocator: stateful, max_size() = budget bytes / sizeof(T) (default budget 100 bytes: 12 doubles, 100 +// chars); copy construction selects a default-budget allocator; no propagation; == compares budgets. Once armed, +// allocate() writes a marker to stderr, so a death test can show that the abort came before any allocation. +#include +#include + +namespace s3_r5_budget +{ + inline bool armed = false; + inline char const marker[] = "budget_allocator::allocate called"; + + template < typename T > + struct budget_allocator + { + typedef T value_type; + typedef std::false_type propagate_on_container_copy_assignment; + typedef std::false_type propagate_on_container_move_assignment; + typedef std::false_type propagate_on_container_swap; + typedef std::false_type is_always_equal; + + std::size_t budget = 100; + + budget_allocator() noexcept = default; + explicit budget_allocator( std::size_t b ) noexcept : budget{ b } {} + template < typename U > + budget_allocator( budget_allocator< U > const& other ) noexcept : budget{ other.budget } {} + + T* allocate( std::size_t n ) + { + if ( armed ) { [[maybe_unused]] auto const w = ::write( STDERR_FILENO, marker, sizeof( marker ) - 1 ); } + return std::allocator< T >{}.allocate( n ); + } + void deallocate( T* p, std::size_t n ) noexcept { std::allocator< T >{}.deallocate( p, n ); } + std::size_t max_size() const noexcept { return budget / sizeof( T ); } + budget_allocator select_on_container_copy_construction() const noexcept { return budget_allocator{}; } + + template < typename U > + friend bool operator == ( budget_allocator const& a, budget_allocator< U > const& b ) noexcept { return a.budget == b.budget; } + }; + + typedef budget_allocator< double > ad; + typedef feng::matrix< double, ad > md; + typedef feng::matrix< char, budget_allocator< char > > mc; + static_assert( std::is_same_v< md::allocator_type::value_type, md::value_type > ); // the accepted value_type case + + // Dies with a "size" message, and the dying child never reached an armed allocate(). + template < typename Callable > + void require_size_abort_before_allocate( Callable fn ) + { + S2_REQUIRE_DEATH( fn, "size" ); + s2_death::outcome const out = s2_death::run( fn ); + INFO( "child stderr: " << out.err ); + REQUIRE( out.signaled ); + REQUIRE_FALSE( s2_death::contains( out.err, marker ) ); + } +} + +TEST_CASE( "S3 copy-assign checks the size before allocating", "[S3][S3-R5]" ) +{ + using namespace s3_r5_budget; + + SECTION( "the marker shows when allocate() runs" ) + { + s2_death::outcome const out = s2_death::run( []{ armed = true; md m{ ad{ 1000 }, 2, 2 }; } ); + REQUIRE( out.exited ); + REQUIRE( s2_death::contains( out.err, marker ) ); + } + SECTION( "copy assignment checks against the destination's allocator" ) + { + require_size_abort_before_allocate( []{ md src{ ad{ 1000 }, 5, 5 }; md dst{ ad{ 100 } }; armed = true; dst = src; } ); + } + SECTION( "copy construction checks against the selected allocator" ) + { + require_size_abort_before_allocate( []{ md src{ ad{ 1000 }, 5, 5 }; armed = true; md dst{ src }; (void)dst; } ); + } + SECTION( "converting construction checks against the rebound allocator" ) + { + require_size_abort_before_allocate( []{ mc src{ budget_allocator< char >{ 100 }, 5, 5 }; armed = true; md dst{ src }; (void)dst; } ); + } + SECTION( "astype checks against the rebound allocator" ) + { + require_size_abort_before_allocate( []{ mc src{ budget_allocator< char >{ 100 }, 5, 5 }; armed = true; auto dst = src.astype< double >(); (void)dst; } ); + } + SECTION( "move assignment between unequal allocators checks against the destination's" ) + { + require_size_abort_before_allocate( []{ md src{ ad{ 1000 }, 5, 5 }; md dst{ ad{ 100 } }; armed = true; dst = std::move( src ); } ); + } + SECTION( "element-wise swap checks against each side's allocator" ) + { + require_size_abort_before_allocate( []{ md a{ ad{ 1000 }, 5, 5 }; md b{ ad{ 100 } }; armed = true; a.swap( b ); } ); + } + SECTION( "resize checks against the matrix's allocator" ) + { + require_size_abort_before_allocate( []{ md m{ ad{ 100 } }; armed = true; m.resize( 5, 5 ); } ); + } + SECTION( "within budget every path still works" ) + { + md src{ ad{ 1000 }, 3, 3, 2.0 }; + md dst{ ad{ 100 } }; + dst = src; + REQUIRE( dst.size() == 9 ); + md copy{ src }; + REQUIRE( copy.get_allocator().budget == 100 ); + REQUIRE( copy[2][2] == 2.0 ); + auto const as_char = src.astype< char >(); + REQUIRE( as_char.size() == 9 ); + } +} diff --git a/tests/cases/s4_r1.hpp b/tests/cases/s4_r1.hpp new file mode 100644 index 0000000..6ba125d --- /dev/null +++ b/tests/cases/s4_r1.hpp @@ -0,0 +1,301 @@ +// S4-R1 (PR-6): stride iterators with logical ends (F05); index checks in every build, address checks in the +// checked-iterator build (FENG_MATRIX_CHECKED_ITERATORS, D-020). +#include "./s2_death.hpp" + +#include +#include +#include +#include +#include + +namespace s4_r1 +{ + inline feng::matrix numbered( std::size_t r, std::size_t c ) + { + feng::matrix m{ r, c }; + for ( std::size_t i = 0; i != r; ++i ) + for ( std::size_t j = 0; j != c; ++j ) + m( i, j ) = static_cast( i * 10 + j ); + return m; + } + + template < typename It > + std::vector walk( It first, It last ) + { + std::vector ans; + for ( ; first != last; ++first ) ans.push_back( *first ); + return ans; + } +} + +TEST_CASE( "S4 one-column anti-diagonal traverses forward and backward", "[S4][S4-R1]" ) +{ + SECTION( "5x1 anti-diagonals have length 1, forward and reverse" ) + { + auto m = s4_r1::numbered( 5, 1 ); + for ( std::ptrdiff_t k = -4; k <= 0; ++k ) + { + INFO( "k = " << k ); + REQUIRE( std::distance( m.anti_diag_begin( k ), m.anti_diag_end( k ) ) == 1 ); + REQUIRE( s4_r1::walk( m.anti_diag_begin( k ), m.anti_diag_end( k ) ) == std::vector{ m( static_cast( -k ), 0 ) } ); + REQUIRE( s4_r1::walk( m.anti_diag_rbegin( k ), m.anti_diag_rend( k ) ) == std::vector{ m( static_cast( -k ), 0 ) } ); + REQUIRE( std::distance( m.anti_diag_crbegin( k ), m.anti_diag_crend( k ) ) == 1 ); + } + } + SECTION( "the last column of a 3x4 matrix has distance 3, forward and reverse" ) + { + auto m = s4_r1::numbered( 3, 4 ); + REQUIRE( std::distance( m.col_begin( 3 ), m.col_end( 3 ) ) == 3 ); + REQUIRE( s4_r1::walk( m.col_begin( 3 ), m.col_end( 3 ) ) == std::vector{ 3.0, 13.0, 23.0 } ); + REQUIRE( s4_r1::walk( m.col_rbegin( 3 ), m.col_rend( 3 ) ) == std::vector{ 23.0, 13.0, 3.0 } ); + auto const& cm = m; + REQUIRE( s4_r1::walk( cm.col_cbegin( 3 ), cm.col_cend( 3 ) ) == std::vector{ 3.0, 13.0, 23.0 } ); + } + SECTION( "diagonals of a 3x4 matrix match the index oracle" ) + { + auto m = s4_r1::numbered( 3, 4 ); + for ( std::ptrdiff_t k = -2; k <= 3; ++k ) + { + std::vector diag, anti; + for ( std::ptrdiff_t r = 0; r != 3; ++r ) + { + std::ptrdiff_t const c = r + k; + if ( c >= 0 && c < 4 ) diag.push_back( m( static_cast( r ), static_cast( c ) ) ); + } + // anti-diagonal k: k > 0 starts at (0, 3-k), k <= 0 at (-k, 3); each step goes one row down, one column left + std::ptrdiff_t r = k > 0 ? 0 : -k, c = k > 0 ? 3 - k : 3; + for ( ; r < 3 && c >= 0; ++r, --c ) anti.push_back( m( static_cast( r ), static_cast( c ) ) ); + INFO( "k = " << k ); + REQUIRE( s4_r1::walk( m.diag_begin( k ), m.diag_end( k ) ) == diag ); + REQUIRE( std::distance( m.diag_begin( k ), m.diag_end( k ) ) == static_cast( diag.size() ) ); + REQUIRE( s4_r1::walk( m.diag_rbegin( k ), m.diag_rend( k ) ) == std::vector( diag.rbegin(), diag.rend() ) ); + REQUIRE( s4_r1::walk( m.anti_diag_begin( k ), m.anti_diag_end( k ) ) == anti ); + REQUIRE( s4_r1::walk( m.anti_diag_rbegin( k ), m.anti_diag_rend( k ) ) == std::vector( anti.rbegin(), anti.rend() ) ); + } + } + SECTION( "diagonal 0 of an empty matrix is an empty range" ) + { + feng::matrix m; + REQUIRE( m.diag_begin() == m.diag_end() ); + REQUIRE( m.anti_diag_begin() == m.anti_diag_end() ); + } +} + +TEST_CASE( "S4 column and diagonal indices outside the matrix abort", "[S4][S4-R1]" ) +{ + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); auto it = m.col_begin( 4 ); (void)it; }, "matrix iterator" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); auto it = m.diag_begin( 4 ); (void)it; }, "matrix iterator" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); auto it = m.diag_begin( -3 ); (void)it; }, "matrix iterator" ); +} + +TEST_CASE( "S4 diag of an empty matrix with an offset keeps its result", "[S4][S4-R1]" ) +{ + auto all_zero = []( feng::matrix const& m ) { return std::all_of( m.begin(), m.end(), []( double x ) { return x == 0.0; } ); }; + feng::matrix const empty; + auto const d0 = feng::diag( empty, 0 ); + REQUIRE( d0.row() == 0 ); + REQUIRE( d0.col() == 0 ); + auto const d1 = feng::diag( empty, 1 ); + REQUIRE( d1.row() == 1 ); + REQUIRE( d1.col() == 1 ); + REQUIRE( all_zero( d1 ) ); + auto const dm2 = feng::diag( empty, -2 ); + REQUIRE( dm2.row() == 2 ); + REQUIRE( dm2.col() == 2 ); + REQUIRE( all_zero( dm2 ) ); + auto const v3 = feng::diag( std::vector{}, 3 ); + REQUIRE( v3.row() == 3 ); + REQUIRE( v3.col() == 3 ); + REQUIRE( all_zero( v3 ) ); + + auto const placed = feng::diag( s4_r1::numbered( 2, 2 ), 1 ); + REQUIRE( placed.row() == 3 ); + REQUIRE( placed.col() == 3 ); + for ( std::size_t i = 0; i != 3; ++i ) + for ( std::size_t j = 0; j != 3; ++j ) + { + INFO( "(" << i << ", " << j << ")" ); + double const expected = ( j == i + 1 ) ? static_cast( i * 11 ) : 0.0; + REQUIRE( placed( i, j ) == expected ); + } +} + +#ifdef FENG_MATRIX_CHECKED_ITERATORS +TEST_CASE( "S4 checked iterators abort on a past-the-end dereference", "[S4][S4-R1]" ) +{ + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); double volatile x = *m.col_end( 1 ); (void)x; }, "matrix iterator" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); double volatile x = m.col_begin( 0 )[3]; (void)x; }, "matrix iterator" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); auto it = m.col_begin( 0 ) - 1; (void)it; }, "matrix iterator" ); +} +#endif + +namespace s4_r1 +{ + template < typename It > + std::vector walk_back( It first, It last ) + { + std::vector ans; + while ( last != first ) ans.push_back( *--last ); + return ans; + } + + // the oracle, independent of the library's range code: elements (i, j) of an r x c matrix with + // pred(i, j), in increasing row order (each diagonal and anti-diagonal holds at most one element per row) + template < typename Pred > + std::vector oracle( feng::matrix const& m, Pred pred ) + { + std::vector ans; + for ( std::ptrdiff_t i = 0; i != static_cast( m.row() ); ++i ) + for ( std::ptrdiff_t j = 0; j != static_cast( m.col() ); ++j ) + if ( pred( i, j ) ) ans.push_back( m( static_cast( i ), static_cast( j ) ) ); + return ans; + } +} + +// every variant of range family `fam` at argument `arg` of the matrix `m` (and its const alias `cm`) visits +// `expected` forward and its reverse backward, and every distance equals the oracle length +#define S4_R1_REQUIRE_RANGE( fam, arg, expected ) \ + do { \ + std::vector const ex_ = ( expected ); \ + std::vector const rev_( ex_.rbegin(), ex_.rend() ); \ + std::ptrdiff_t const n_ = static_cast( ex_.size() ); \ + INFO( #fam " " << ( arg ) ); \ + REQUIRE( s4_r1::walk( m.fam##_begin( arg ), m.fam##_end( arg ) ) == ex_ ); \ + REQUIRE( s4_r1::walk_back( m.fam##_begin( arg ), m.fam##_end( arg ) ) == rev_ ); \ + REQUIRE( std::distance( m.fam##_begin( arg ), m.fam##_end( arg ) ) == n_ ); \ + REQUIRE( s4_r1::walk( cm.fam##_begin( arg ), cm.fam##_end( arg ) ) == ex_ ); \ + REQUIRE( std::distance( cm.fam##_begin( arg ), cm.fam##_end( arg ) ) == n_ ); \ + REQUIRE( s4_r1::walk( cm.fam##_cbegin( arg ), cm.fam##_cend( arg ) ) == ex_ ); \ + REQUIRE( s4_r1::walk_back( cm.fam##_cbegin( arg ), cm.fam##_cend( arg ) ) == rev_ ); \ + REQUIRE( std::distance( cm.fam##_cbegin( arg ), cm.fam##_cend( arg ) ) == n_ ); \ + REQUIRE( s4_r1::walk( m.fam##_rbegin( arg ), m.fam##_rend( arg ) ) == rev_ ); \ + REQUIRE( s4_r1::walk_back( m.fam##_rbegin( arg ), m.fam##_rend( arg ) ) == ex_ ); \ + REQUIRE( std::distance( m.fam##_rbegin( arg ), m.fam##_rend( arg ) ) == n_ ); \ + REQUIRE( s4_r1::walk( cm.fam##_rbegin( arg ), cm.fam##_rend( arg ) ) == rev_ ); \ + REQUIRE( std::distance( cm.fam##_rbegin( arg ), cm.fam##_rend( arg ) ) == n_ ); \ + REQUIRE( s4_r1::walk( cm.fam##_crbegin( arg ), cm.fam##_crend( arg ) ) == rev_ ); \ + REQUIRE( std::distance( cm.fam##_crbegin( arg ), cm.fam##_crend( arg ) ) == n_ ); \ + } while ( false ) + +TEST_CASE( "S4 every diagonal of every shape up to 4x5 traverses exactly", "[S4][S4-R1]" ) +{ + for ( std::size_t r = 1; r <= 4; ++r ) + for ( std::size_t c = 1; c <= 5; ++c ) + { + INFO( "shape " << r << "x" << c ); + auto m = s4_r1::numbered( r, c ); + auto const& cm = m; + std::ptrdiff_t const R = static_cast( r ), C = static_cast( c ); + for ( std::size_t j = 0; j != c; ++j ) + { + std::ptrdiff_t const jj = static_cast( j ); + S4_R1_REQUIRE_RANGE( col, j, s4_r1::oracle( m, [=]( std::ptrdiff_t, std::ptrdiff_t y ) { return y == jj; } ) ); + } + for ( std::ptrdiff_t k = -( R - 1 ); k < C; ++k ) + { + auto const diag = s4_r1::oracle( m, [=]( std::ptrdiff_t x, std::ptrdiff_t y ) { return y - x == k; } ); + auto const anti = s4_r1::oracle( m, [=]( std::ptrdiff_t x, std::ptrdiff_t y ) { return x + y == C - 1 - k; } ); + REQUIRE_FALSE( diag.empty() ); + REQUIRE_FALSE( anti.empty() ); + S4_R1_REQUIRE_RANGE( diag, k, diag ); + S4_R1_REQUIRE_RANGE( anti_diag, k, anti ); + std::size_t const u = static_cast( k < 0 ? -k : k ); + if ( k >= 0 ) + { + S4_R1_REQUIRE_RANGE( upper_diag, u, diag ); + S4_R1_REQUIRE_RANGE( upper_anti_diag, u, anti ); + } + if ( k <= 0 ) + { + S4_R1_REQUIRE_RANGE( lower_diag, u, diag ); + S4_R1_REQUIRE_RANGE( lower_anti_diag, u, anti ); + } + } + if ( c == 1 ) + for ( std::ptrdiff_t k = -( R - 1 ); k <= 0; ++k ) + REQUIRE( std::distance( m.anti_diag_begin( k ), m.anti_diag_end( k ) ) == 1 ); + } +} + +TEST_CASE( "S4 every range family rejects a column or diagonal index outside the matrix", "[S4][S4-R1]" ) +{ + auto const run = []( auto f ) { S2_REQUIRE_DEATH( [f]{ auto m = s4_r1::numbered( 3, 4 ); f( m ); }, "matrix iterator" ); }; + SECTION( "column index col()" ) + { + run( []( feng::matrix& m ) { auto it = m.col_begin( 4 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = m.col_end( 4 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = std::as_const( m ).col_cbegin( 4 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = m.col_rbegin( 4 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = std::as_const( m ).col_crend( 4 ); (void)it; } ); + } + SECTION( "diagonal k = col() and k = -row()" ) + { + run( []( feng::matrix& m ) { auto it = m.diag_end( 4 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = std::as_const( m ).diag_cbegin( -3 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = m.diag_rbegin( -3 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = std::as_const( m ).diag_crend( 4 ); (void)it; } ); + } + SECTION( "upper and lower diagonal indices col() and row()" ) + { + run( []( feng::matrix& m ) { auto it = m.upper_diag_begin( 4 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = std::as_const( m ).upper_diag_crbegin( 4 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = m.lower_diag_end( 3 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = std::as_const( m ).lower_diag_cbegin( 3 ); (void)it; } ); + } + SECTION( "anti-diagonal k = col() and k = -row()" ) + { + run( []( feng::matrix& m ) { auto it = m.anti_diag_begin( 4 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = std::as_const( m ).anti_diag_cend( -3 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = m.anti_diag_rend( -3 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = std::as_const( m ).anti_diag_crbegin( 4 ); (void)it; } ); + } + SECTION( "upper and lower anti-diagonal indices col() and row()" ) + { + run( []( feng::matrix& m ) { auto it = m.upper_anti_diag_begin( 4 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = std::as_const( m ).upper_anti_diag_cend( 4 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = m.lower_anti_diag_rbegin( 3 ); (void)it; } ); + run( []( feng::matrix& m ) { auto it = std::as_const( m ).lower_anti_diag_crend( 3 ); (void)it; } ); + } +} + +#ifdef FENG_MATRIX_CHECKED_ITERATORS +TEST_CASE( "S4 checked iterators abort on every position or address outside the owner", "[S4][S4-R1]" ) +{ + using it_t = feng::stride_iterator; + SECTION( "dereference and subscript outside [0, count)" ) + { + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); double volatile x = *m.col_end( 0 ); (void)x; }, "matrix iterator: dereference" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); double volatile x = m.col_begin( 0 )[3]; (void)x; }, "matrix iterator: dereference" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); double volatile x = m.col_end( 0 )[-4]; (void)x; }, "matrix iterator: dereference" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); double* volatile p = m.diag_end().operator->(); (void)p; }, "matrix iterator: dereference" ); + } + SECTION( "index moved outside [0, count]" ) + { + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); auto it = m.col_end( 1 ); ++it; }, "matrix iterator: position" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); auto it = m.col_end( 1 ); it++; }, "matrix iterator: position" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); auto it = m.col_begin( 1 ); --it; }, "matrix iterator: position" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); auto it = m.col_begin( 1 ); it--; }, "matrix iterator: position" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); auto it = m.col_begin( 1 ); it += 4; }, "matrix iterator: position" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); auto it = m.col_end( 1 ); it -= 4; }, "matrix iterator: position" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); auto it = m.anti_diag_begin() + 4; (void)it; }, "matrix iterator: position" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); auto it = 4 + m.anti_diag_begin(); (void)it; }, "matrix iterator: position" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r1::numbered( 3, 4 ); auto it = m.diag_begin() - 1; (void)it; }, "matrix iterator: position" ); + } + SECTION( "first or last address outside the owner, through the public constructor" ) + { + // the owner is the first four elements of an eight-element buffer, so every address below is valid to form + S2_REQUIRE_DEATH( []{ double buf[8] = {}; it_t it( buf + 5, 1, 1, 0, buf, 4 ); (void)it; }, "matrix iterator: first address outside" ); + S2_REQUIRE_DEATH( []{ double buf[8] = {}; it_t it( buf, 1, 1, 0, buf + 1, 3 ); (void)it; }, "matrix iterator: first address outside" ); + S2_REQUIRE_DEATH( []{ double buf[8] = {}; it_t it( buf + 4, 1, 1, 0, buf, 4 ); (void)it; }, "matrix iterator: first element" ); + S2_REQUIRE_DEATH( []{ double buf[8] = {}; it_t it( buf, 2, 3, 0, buf, 4 ); (void)it; }, "matrix iterator: last element" ); + S2_REQUIRE_DEATH( []{ double buf[8] = {}; it_t it( buf + 1, -1, 3, 0, buf, 4 ); (void)it; }, "matrix iterator: last element" ); + S2_REQUIRE_DEATH( []{ double buf[8] = {}; it_t it( buf, 1, 2, 3, buf, 4 ); (void)it; }, "matrix iterator: position" ); + // the legal limits: an empty range at data()+size() and a full range with stride 0 + double buf[8] = {}; + it_t const e( buf + 4, 1, 0, 0, buf, 4 ); + it_t const z( buf + 3, 0, 5, 0, buf, 4 ); + REQUIRE( e - e == 0 ); + REQUIRE( ( z + 5 ) - z == 5 ); + } +} +#endif diff --git a/tests/cases/s4_r2.hpp b/tests/cases/s4_r2.hpp new file mode 100644 index 0000000..3969327 --- /dev/null +++ b/tests/cases/s4_r2.hpp @@ -0,0 +1,440 @@ +// S4-R2 (PR-6): const and mutable views carry a pointer to the first viewed element, the extents and the parent's +// row stride (F06, F05; D-019). +#include +#include +#include + +namespace s4_r2 +{ + inline feng::matrix distinct( std::size_t r, std::size_t c ) + { + feng::matrix m{ r, c }; + for ( std::size_t i = 0; i != r; ++i ) + for ( std::size_t j = 0; j != c; ++j ) + m( i, j ) = static_cast( 100 * i + j ) + 0.5; + return m; + } + + template < typename It > + std::vector walk( It first, It last ) + { + std::vector ans; + for ( ; first != last; ++first ) ans.push_back( *first ); + return ans; + } + + // reads of a 2x3 view at rows [1, 3), columns [2, 5) of distinct( 5, 6 ) + template < typename View > + void require_reads( View const& v, feng::matrix const& m ) + { + REQUIRE( v.row() == 2 ); + REQUIRE( v.col() == 3 ); + REQUIRE( v.size() == 6 ); + REQUIRE( v.row_stride() == 6 ); + auto const [vr, vc] = v.shape(); + REQUIRE( vr == 2 ); + REQUIRE( vc == 3 ); + for ( std::size_t r = 0; r != 2; ++r ) + for ( std::size_t c = 0; c != 3; ++c ) + { + REQUIRE( v( r, c ) == m( r + 1, c + 2 ) ); + REQUIRE( v.at( r, c ) == m( r + 1, c + 2 ) ); + REQUIRE( v[r][c] == m( r + 1, c + 2 ) ); + REQUIRE( &v.at( r, c ) == &m.at( r + 1, c + 2 ) ); + } + REQUIRE( walk( v.row_begin( 0 ), v.row_end( 0 ) ) == std::vector{ 102.5, 103.5, 104.5 } ); + REQUIRE( walk( v.row_cbegin( 1 ), v.row_cend( 1 ) ) == std::vector{ 202.5, 203.5, 204.5 } ); + for ( std::size_t c = 0; c != 3; ++c ) + { + double const top = m( 1, c + 2 ), bottom = m( 2, c + 2 ); + REQUIRE( std::distance( v.col_begin( c ), v.col_end( c ) ) == 2 ); + REQUIRE( walk( v.col_begin( c ), v.col_end( c ) ) == std::vector{ top, bottom } ); + REQUIRE( walk( v.col_cbegin( c ), v.col_cend( c ) ) == std::vector{ top, bottom } ); + REQUIRE( walk( v.col_rbegin( c ), v.col_rend( c ) ) == std::vector{ bottom, top } ); + } + std::vector const all{ 102.5, 103.5, 104.5, 202.5, 203.5, 204.5 }; + REQUIRE( std::distance( v.begin(), v.end() ) == 6 ); + REQUIRE( walk( v.begin(), v.end() ) == all ); + REQUIRE( walk( v.cbegin(), v.cend() ) == all ); + REQUIRE( walk( v.rbegin(), v.rend() ) == std::vector( all.rbegin(), all.rend() ) ); + REQUIRE( v.begin()[4] == 203.5 ); + REQUIRE( v.get_allocator() == m.get_allocator() ); + } + + // every parent element outside rows [1, 3), columns [2, 5) still holds its distinct value + inline void require_outside_untouched( feng::matrix const& m ) + { + auto const fresh = distinct( 5, 6 ); + for ( std::size_t i = 0; i != 5; ++i ) + for ( std::size_t j = 0; j != 6; ++j ) + if ( !( i >= 1 && i < 3 && j >= 2 && j < 5 ) ) + REQUIRE( m( i, j ) == fresh( i, j ) ); + } + + inline void require_inside( feng::matrix const& m, std::vector const& expected ) + { + std::size_t k = 0; + for ( std::size_t i = 1; i != 3; ++i ) + for ( std::size_t j = 2; j != 5; ++j ) + REQUIRE( m( i, j ) == expected[k++] ); + } + + inline void require_six( feng::matrix const& n ) + { + REQUIRE( n.row() == 2 ); + REQUIRE( n.col() == 3 ); + REQUIRE( n.size() == 6 ); + REQUIRE( std::vector( n.begin(), n.end() ) == std::vector{ 102.5, 103.5, 104.5, 202.5, 203.5, 204.5 } ); + } +} + +TEST_CASE( "S4 offset 2x3 view of a 5x6 matrix reads writes copies and iterates its elements", "[S4][S4-R2]" ) +{ + auto m = s4_r2::distinct( 5, 6 ); + auto const& cm = m; + + SECTION( "the const view reads exactly the six elements, forward and reverse" ) + { + auto const v = feng::make_view( cm, { 1, 3 }, { 2, 5 } ); + s4_r2::require_reads( v, m ); + feng::matrix_view> const w{ cm, { 1, 3 }, { 2, 5 } }; + s4_r2::require_reads( w, m ); + } + SECTION( "the mutable view reads exactly the six elements and converts to the const view" ) + { + auto v = feng::make_mutable_view( m, { 1, 3 }, { 2, 5 } ); + s4_r2::require_reads( v, m ); + feng::matrix_view> const w = v; + s4_r2::require_reads( w, m ); + feng::mutable_matrix_view> u{ m, { 1, 3 }, { 2, 5 } }; + s4_r2::require_reads( u, m ); + } + SECTION( "writes through at, (), [][] change exactly the viewed elements" ) + { + auto v = feng::make_mutable_view( m, { 1, 3 }, { 2, 5 } ); + v.at( 0, 0 ) = -1.0; + v( 0, 2 ) = -2.0; + v[1][1] = -3.0; + *v.row_begin( 1 ) = -4.0; + s4_r2::require_inside( m, { -1.0, 103.5, -2.0, -4.0, -3.0, 204.5 } ); + s4_r2::require_outside_untouched( m ); + } + SECTION( "std::fill over begin..end changes exactly the viewed elements" ) + { + auto v = feng::make_mutable_view( m, { 1, 3 }, { 2, 5 } ); + std::fill( v.begin(), v.end(), 7.0 ); + s4_r2::require_inside( m, std::vector( 6, 7.0 ) ); + s4_r2::require_outside_untouched( m ); + } + SECTION( "writes through a column iterator change exactly that view column" ) + { + auto v = feng::make_mutable_view( m, { 1, 3 }, { 2, 5 } ); + std::fill( v.col_begin( 1 ), v.col_end( 1 ), 9.0 ); + s4_r2::require_inside( m, { 102.5, 9.0, 104.5, 202.5, 9.0, 204.5 } ); + s4_r2::require_outside_untouched( m ); + } + SECTION( "a matrix from either view and copy from a view hold exactly the six values" ) + { + auto const cv = feng::make_view( cm, { 1, 3 }, { 2, 5 } ); + auto const mv = feng::make_mutable_view( m, { 1, 3 }, { 2, 5 } ); + feng::matrix const a{ cv }; + feng::matrix const b{ mv }; + s4_r2::require_six( a ); + s4_r2::require_six( b ); + REQUIRE( a.get_allocator() == m.get_allocator() ); + feng::matrix n{ 4, 4 }; + n.copy( cv ); + s4_r2::require_six( n ); + feng::matrix p; + p.copy( mv ); + s4_r2::require_six( p ); + } +} + +// S4-T8 (S4-R2): every listed view shape (1x1, full-size, first-row, last-column, 1xN, Nx1) on 5x6, 1x6 and 6x1 +// owners of distinct values, through make_view and make_mutable_view, on the stateful test allocator. +#include "./s3_alloc.hpp" +#include + +namespace s4_r2_shapes +{ + using alloc_type = s3_alloc::tracking_allocator< double, false, false, false >; + using owner_type = feng::matrix< double, alloc_type >; + + struct shape_case + { + char const* name; + std::size_t rows, cols; // owner + std::size_t r0, r1, c0, c1; // view + }; + + inline std::vector< shape_case > const& cases() + { + static std::vector< shape_case > const all{ + { "1x1 at (2, 3) of 5x6", 5, 6, 2, 3, 3, 4 }, + { "full-size 5x6", 5, 6, 0, 5, 0, 6 }, + { "first row of 5x6", 5, 6, 0, 1, 0, 6 }, + { "last column of 5x6", 5, 6, 0, 5, 5, 6 }, + { "1x4 at (3, 1) of 5x6", 5, 6, 3, 4, 1, 5 }, + { "4x1 at (1, 2) of 5x6", 5, 6, 1, 5, 2, 3 }, + { "1x4 at (0, 1) of 1x6", 1, 6, 0, 1, 1, 5 }, + { "full-size 1x6", 1, 6, 0, 1, 0, 6 }, + { "3x1 at (2, 0) of 6x1", 6, 1, 2, 5, 0, 1 }, + { "full-size 6x1", 6, 1, 0, 6, 0, 1 }, + }; + return all; + } + + // the owner's element (i, j) before any write: distinct over the whole owner + inline double original( std::size_t i, std::size_t j ) { return 1000.0 * static_cast< double >( i ) + static_cast< double >( j ) + 0.25; } + + inline owner_type make_owner( shape_case const& s, int id ) + { + owner_type m{ alloc_type{ id }, s.rows, s.cols }; + double* const p = m.data(); + for ( std::size_t i = 0; i != s.rows; ++i ) + for ( std::size_t j = 0; j != s.cols; ++j ) + p[ i * s.cols + j ] = original( i, j ); + return m; + } + + inline bool inside( shape_case const& s, std::size_t i, std::size_t j ) { return i >= s.r0 && i < s.r1 && j >= s.c0 && j < s.c1; } + + // oracle: the viewed values in row order, by index arithmetic on the owner's storage + inline std::vector< double > viewed( shape_case const& s, owner_type const& m ) + { + std::vector< double > ans; + double const* const p = m.data(); + for ( std::size_t i = s.r0; i != s.r1; ++i ) + for ( std::size_t j = s.c0; j != s.c1; ++j ) + ans.push_back( p[ i * s.cols + j ] ); + return ans; + } + + template < typename It > + std::vector< double > walk( It first, It last ) + { + std::vector< double > ans; + for ( ; first != last; ++first ) ans.push_back( *first ); + return ans; + } + + template < typename View > + void require_reads( shape_case const& s, View const& v, owner_type const& m ) + { + std::size_t const vr = s.r1 - s.r0, vc = s.c1 - s.c0; + double const* const p = m.data(); + auto const at = [&]( std::size_t r, std::size_t c ) { return p[ ( s.r0 + r ) * s.cols + ( s.c0 + c ) ]; }; + REQUIRE( v.row() == vr ); + REQUIRE( v.col() == vc ); + REQUIRE( v.size() == vr * vc ); + REQUIRE( v.row_stride() == s.cols ); + auto const [sr, sc] = v.shape(); + REQUIRE( sr == vr ); + REQUIRE( sc == vc ); + for ( std::size_t r = 0; r != vr; ++r ) + for ( std::size_t c = 0; c != vc; ++c ) + { + REQUIRE( v( r, c ) == at( r, c ) ); + REQUIRE( v.at( r, c ) == at( r, c ) ); + REQUIRE( v[r][c] == at( r, c ) ); + REQUIRE( &v.at( r, c ) == p + ( s.r0 + r ) * s.cols + ( s.c0 + c ) ); + } + for ( std::size_t r = 0; r != vr; ++r ) + { + std::vector< double > row; + for ( std::size_t c = 0; c != vc; ++c ) row.push_back( at( r, c ) ); + REQUIRE( std::distance( v.row_begin( r ), v.row_end( r ) ) == static_cast< std::ptrdiff_t >( vc ) ); + REQUIRE( walk( v.row_begin( r ), v.row_end( r ) ) == row ); + REQUIRE( walk( v.row_cbegin( r ), v.row_cend( r ) ) == row ); + } + for ( std::size_t c = 0; c != vc; ++c ) + { + std::vector< double > col; + for ( std::size_t r = 0; r != vr; ++r ) col.push_back( at( r, c ) ); + std::vector< double > const rcol( col.rbegin(), col.rend() ); + REQUIRE( std::distance( v.col_begin( c ), v.col_end( c ) ) == static_cast< std::ptrdiff_t >( vr ) ); + REQUIRE( std::distance( v.col_rbegin( c ), v.col_rend( c ) ) == static_cast< std::ptrdiff_t >( vr ) ); + REQUIRE( walk( v.col_begin( c ), v.col_end( c ) ) == col ); + REQUIRE( walk( v.col_cbegin( c ), v.col_cend( c ) ) == col ); + REQUIRE( walk( v.col_rbegin( c ), v.col_rend( c ) ) == rcol ); + REQUIRE( walk( v.col_crbegin( c ), v.col_crend( c ) ) == rcol ); + } + std::vector< double > const all = viewed( s, m ); + std::vector< double > const rall( all.rbegin(), all.rend() ); + REQUIRE( std::distance( v.begin(), v.end() ) == static_cast< std::ptrdiff_t >( all.size() ) ); + REQUIRE( std::distance( v.rbegin(), v.rend() ) == static_cast< std::ptrdiff_t >( all.size() ) ); + REQUIRE( walk( v.begin(), v.end() ) == all ); + REQUIRE( walk( v.cbegin(), v.cend() ) == all ); + REQUIRE( walk( v.rbegin(), v.rend() ) == rall ); + REQUIRE( walk( v.crbegin(), v.crend() ) == rall ); + REQUIRE( v.get_allocator() == m.get_allocator() ); + } + + // after a write of `written` (row order) through a view, the viewed elements hold it and the rest are original + inline void require_written( shape_case const& s, owner_type const& m, std::vector< double > const& written ) + { + REQUIRE( viewed( s, m ) == written ); + double const* const p = m.data(); + for ( std::size_t i = 0; i != s.rows; ++i ) + for ( std::size_t j = 0; j != s.cols; ++j ) + if ( !inside( s, i, j ) ) + REQUIRE( p[ i * s.cols + j ] == original( i, j ) ); + } + + // values -1, -2, ... in row order of the view + inline std::vector< double > marks( shape_case const& s ) + { + std::vector< double > ans( ( s.r1 - s.r0 ) * ( s.c1 - s.c0 ) ); + for ( std::size_t k = 0; k != ans.size(); ++k ) ans[k] = -1.0 - static_cast< double >( k ); + return ans; + } + + template < typename Write > + void check_write( shape_case const& s, Write write ) + { + auto m = make_owner( s, 3 ); + auto v = feng::make_mutable_view( m, { s.r0, s.r1 }, { s.c0, s.c1 } ); + write( v, s.c1 - s.c0 ); + require_written( s, m, marks( s ) ); + } + + inline void require_matrix( owner_type const& n, std::size_t r, std::size_t c, std::vector< double > const& values ) + { + REQUIRE( n.row() == r ); + REQUIRE( n.col() == c ); + REQUIRE( n.size() == values.size() ); + REQUIRE( std::vector< double >( n.begin(), n.end() ) == values ); + } +} + +TEST_CASE( "S4 every listed view shape reads writes and copies exactly", "[S4][S4-R2]" ) +{ + using namespace s4_r2_shapes; + s3_alloc::reset(); + for ( auto const& s : cases() ) + { + INFO( "view " << s.name ); + std::size_t const vr = s.r1 - s.r0, vc = s.c1 - s.c0; + { + auto m = make_owner( s, 1 ); + auto const& cm = m; + auto const cv = feng::make_view( cm, { s.r0, s.r1 }, { s.c0, s.c1 } ); + require_reads( s, cv, m ); + auto const mv = feng::make_mutable_view( m, { s.r0, s.r1 }, { s.c0, s.c1 } ); + require_reads( s, mv, m ); + feng::matrix_view< double, alloc_type > const conv = mv; + require_reads( s, conv, m ); + } + // writes through the mutable view change exactly the viewed elements + check_write( s, []( auto& v, std::size_t ) { for ( std::size_t r = 0; r != v.row(); ++r ) for ( std::size_t c = 0; c != v.col(); ++c ) v.at( r, c ) = -1.0 - static_cast< double >( r * v.col() + c ); } ); + check_write( s, []( auto& v, std::size_t ) { for ( std::size_t r = 0; r != v.row(); ++r ) for ( std::size_t c = 0; c != v.col(); ++c ) v( r, c ) = -1.0 - static_cast< double >( r * v.col() + c ); } ); + check_write( s, []( auto& v, std::size_t ) { for ( std::size_t r = 0; r != v.row(); ++r ) for ( std::size_t c = 0; c != v.col(); ++c ) v[r][c] = -1.0 - static_cast< double >( r * v.col() + c ); } ); + check_write( s, []( auto& v, std::size_t ) { double x = -1.0; for ( auto it = v.begin(); it != v.end(); ++it ) *it = x--; } ); + check_write( s, []( auto& v, std::size_t ) { + for ( std::size_t c = 0; c != v.col(); ++c ) + { + std::size_t r = 0; + for ( auto it = v.col_begin( c ); it != v.col_end( c ); ++it, ++r ) *it = -1.0 - static_cast< double >( r * v.col() + c ); + } + } ); + check_write( s, []( auto& v, std::size_t ) { + for ( std::size_t c = 0; c != v.col(); ++c ) + { + std::size_t r = v.row(); + for ( auto it = v.col_rbegin( c ); it != v.col_rend( c ); ++it ) { --r; *it = -1.0 - static_cast< double >( r * v.col() + c ); } + } + } ); + check_write( s, []( auto& v, std::size_t ) { std::size_t k = v.size(); for ( auto it = v.rbegin(); it != v.rend(); ++it ) { --k; *it = -1.0 - static_cast< double >( k ); } } ); + { + auto m = make_owner( s, 3 ); + auto v = feng::make_mutable_view( m, { s.r0, s.r1 }, { s.c0, s.c1 } ); + std::fill( v.begin(), v.end(), -7.0 ); + require_written( s, m, std::vector< double >( vr * vc, -7.0 ) ); + } + // copies from either view hold exactly the viewed values in row order + { + auto m = make_owner( s, 4 ); + auto const& cm = m; + auto const expected = viewed( s, m ); + auto const cv = feng::make_view( cm, { s.r0, s.r1 }, { s.c0, s.c1 } ); + auto const mv = feng::make_mutable_view( m, { s.r0, s.r1 }, { s.c0, s.c1 } ); + owner_type const a{ cv }; + owner_type const b{ mv }; + require_matrix( a, vr, vc, expected ); + require_matrix( b, vr, vc, expected ); + REQUIRE( a.get_allocator().id == 4 ); + REQUIRE( b.get_allocator().id == 4 ); + owner_type n{ alloc_type{ 5 }, 2, 2 }; + n.copy( cv ); + require_matrix( n, vr, vc, expected ); + owner_type p{ alloc_type{ 5 } }; + p.copy( mv ); + require_matrix( p, vr, vc, expected ); + // S3 (F04): the destination of copy keeps its own allocator + REQUIRE( n.get_allocator().id == 5 ); + REQUIRE( p.get_allocator().id == 5 ); + require_written( s, m, expected ); + } + } + s3_alloc::require_balanced(); +} + +TEST_CASE( "S4 copy from a view of the destination snapshots first", "[S4][S4-R2]" ) +{ + using namespace s4_r2_shapes; + s3_alloc::reset(); + for ( auto const& s : cases() ) + { + INFO( "view " << s.name ); + std::size_t const vr = s.r1 - s.r0, vc = s.c1 - s.c0; + { + auto m = make_owner( s, 6 ); + auto const expected = viewed( s, m ); + auto const& cm = m; + m.copy( feng::make_view( cm, { s.r0, s.r1 }, { s.c0, s.c1 } ) ); + require_matrix( m, vr, vc, expected ); + REQUIRE( m.get_allocator().id == 6 ); + } + { + auto m = make_owner( s, 6 ); + auto const expected = viewed( s, m ); + m.copy( feng::make_mutable_view( m, { s.r0, s.r1 }, { s.c0, s.c1 } ) ); + require_matrix( m, vr, vc, expected ); + REQUIRE( m.get_allocator().id == 6 ); + } + } + s3_alloc::require_balanced(); +} + +// S4-R2, S4-R1 (D-020): in the checked-iterator build the view element iterator aborts on a dereference or +// subscript outside [0, count) and on a position outside [0, count], for both view types. The unchecked build +// keeps the case with in-range checks only, so the name matches in every lane. +#include "./s2_death.hpp" + +TEST_CASE( "S4 checked view iterators abort outside the view", "[S4][S4-R2]" ) +{ +#ifdef FENG_MATRIX_CHECKED_ITERATORS + // const view: rows [1, 3), columns [2, 5) of a 5x6 owner, six elements + S2_REQUIRE_DEATH( []{ auto m = s4_r2::distinct( 5, 6 ); auto const& cm = m; auto v = feng::make_view( cm, { 1, 3 }, { 2, 5 } ); double volatile x = *v.end(); (void)x; }, "matrix iterator" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r2::distinct( 5, 6 ); auto const& cm = m; auto v = feng::make_view( cm, { 1, 3 }, { 2, 5 } ); double volatile x = v.begin()[6]; (void)x; }, "matrix iterator" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r2::distinct( 5, 6 ); auto const& cm = m; auto v = feng::make_view( cm, { 1, 3 }, { 2, 5 } ); double volatile x = v.end()[-7]; (void)x; }, "matrix iterator" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r2::distinct( 5, 6 ); auto const& cm = m; auto v = feng::make_view( cm, { 1, 3 }, { 2, 5 } ); auto it = v.begin(); --it; (void)it; }, "matrix iterator" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r2::distinct( 5, 6 ); auto const& cm = m; auto v = feng::make_view( cm, { 1, 3 }, { 2, 5 } ); auto it = v.end() + 1; (void)it; }, "matrix iterator" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r2::distinct( 5, 6 ); auto const& cm = m; auto v = feng::make_view( cm, { 1, 3 }, { 2, 5 } ); auto it = v.cbegin() - 1; (void)it; }, "matrix iterator" ); + // mutable view of the same elements + S2_REQUIRE_DEATH( []{ auto m = s4_r2::distinct( 5, 6 ); auto v = feng::make_mutable_view( m, { 1, 3 }, { 2, 5 } ); *v.end() = 1.0; }, "matrix iterator" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r2::distinct( 5, 6 ); auto v = feng::make_mutable_view( m, { 1, 3 }, { 2, 5 } ); v.begin()[6] = 1.0; }, "matrix iterator" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r2::distinct( 5, 6 ); auto v = feng::make_mutable_view( m, { 1, 3 }, { 2, 5 } ); v.begin()[-1] = 1.0; }, "matrix iterator" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r2::distinct( 5, 6 ); auto v = feng::make_mutable_view( m, { 1, 3 }, { 2, 5 } ); auto it = v.begin(); it -= 1; (void)it; }, "matrix iterator" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r2::distinct( 5, 6 ); auto v = feng::make_mutable_view( m, { 1, 3 }, { 2, 5 } ); auto it = v.end(); ++it; (void)it; }, "matrix iterator" ); + S2_REQUIRE_DEATH( []{ auto m = s4_r2::distinct( 5, 6 ); auto v = feng::make_mutable_view( m, { 1, 3 }, { 2, 5 } ); auto it = v.begin() + 7; (void)it; }, "matrix iterator" ); +#endif + // the boundary positions themselves are legal in every build + auto m = s4_r2::distinct( 5, 6 ); + auto v = feng::make_mutable_view( m, { 1, 3 }, { 2, 5 } ); + REQUIRE( v.end() - v.begin() == 6 ); + REQUIRE( v.begin()[5] == m( 2, 4 ) ); + REQUIRE( *( v.end() - 1 ) == m( 2, 4 ) ); + REQUIRE( v.end()[-6] == m( 1, 2 ) ); +} diff --git a/tests/cases/s4_r3.hpp b/tests/cases/s4_r3.hpp new file mode 100644 index 0000000..c508635 --- /dev/null +++ b/tests/cases/s4_r3.hpp @@ -0,0 +1,127 @@ +// S4-R3 (PR-6): view ranges are validated in every build, never normalized (F06). +#include "./s2_death.hpp" + +#include +#include +#include +#include + +TEST_CASE( "S4 view ranges outside the owner abort", "[S4][S4-R3]" ) +{ + feng::matrix m{ 4, 5 }; + auto const& cm = m; + + SECTION( "make_view aborts on a range outside the owner" ) + { + S2_REQUIRE_DEATH( [&] { auto v = feng::make_view( cm, { 0, 5 }, { 0, 1 } ); (void)v; }, "matrix view:" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_view( cm, { 3, 2 }, { 0, 1 } ); (void)v; }, "matrix view:" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_view( cm, { 0, 1 }, { 0, 6 } ); (void)v; }, "matrix view:" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_view( cm, { 0, 1, 2 }, { 0, 1 } ); (void)v; }, "matrix view:" ); + } + SECTION( "make_mutable_view aborts on a range outside the owner" ) + { + S2_REQUIRE_DEATH( [&] { auto v = feng::make_mutable_view( m, { 0, 5 }, { 0, 1 } ); (void)v; }, "matrix view:" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_mutable_view( m, { 3, 2 }, { 0, 1 } ); (void)v; }, "matrix view:" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_mutable_view( m, { 0, 1 }, { 0, 6 } ); (void)v; }, "matrix view:" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_mutable_view( m, { 0, 1 }, { 0, 1, 2 } ); (void)v; }, "matrix view:" ); + } + SECTION( "an empty row range gives an empty view" ) + { + auto const v = feng::make_view( cm, { 2, 2 }, { 0, 5 } ); + REQUIRE( v.row() == 0 ); + REQUIRE( v.col() == 5 ); + REQUIRE( v.size() == 0 ); + REQUIRE( v.begin() == v.end() ); + auto const w = feng::make_mutable_view( m, { 2, 2 }, { 0, 5 } ); + REQUIRE( w.row() == 0 ); + REQUIRE( w.col() == 5 ); + REQUIRE( w.begin() == w.end() ); + feng::matrix const n{ v }; + REQUIRE( n.size() == 0 ); + } +} + +TEST_CASE( "S4 every view factory and constructor rejects ranges outside the owner", "[S4][S4-R3]" ) +{ + // R = 4 rows, C = 5 columns + using range = std::pair; + using cview = feng::matrix_view>; + using mview = feng::mutable_matrix_view>; + feng::matrix m{ 4, 5 }; + auto const& cm = m; + + SECTION( "make_view: r0 > r1, r1 > R, c0 > c1, c1 > C" ) + { + S2_REQUIRE_DEATH( [&] { auto v = feng::make_view( cm, { 3, 2 }, { 0, 5 } ); (void)v; }, "matrix view: row range" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_view( cm, { 0, 5 }, { 0, 5 } ); (void)v; }, "matrix view: row range" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_view( cm, { 0, 4 }, { 3, 2 } ); (void)v; }, "matrix view: column range" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_view( cm, { 0, 4 }, { 0, 6 } ); (void)v; }, "matrix view: column range" ); + } + SECTION( "make_mutable_view: r0 > r1, r1 > R, c0 > c1, c1 > C" ) + { + S2_REQUIRE_DEATH( [&] { auto v = feng::make_mutable_view( m, { 3, 2 }, { 0, 5 } ); (void)v; }, "matrix view: row range" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_mutable_view( m, { 0, 5 }, { 0, 5 } ); (void)v; }, "matrix view: row range" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_mutable_view( m, { 0, 4 }, { 3, 2 } ); (void)v; }, "matrix view: column range" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_mutable_view( m, { 0, 4 }, { 0, 6 } ); (void)v; }, "matrix view: column range" ); + } + SECTION( "matrix_view constructor: r0 > r1, r1 > R, c0 > c1, c1 > C" ) + { + S2_REQUIRE_DEATH( [&] { cview v( cm, range{ 3, 2 }, range{ 0, 5 } ); (void)v; }, "matrix view: row range" ); + S2_REQUIRE_DEATH( [&] { cview v( cm, range{ 0, 5 }, range{ 0, 5 } ); (void)v; }, "matrix view: row range" ); + S2_REQUIRE_DEATH( [&] { cview v( cm, range{ 0, 4 }, range{ 3, 2 } ); (void)v; }, "matrix view: column range" ); + S2_REQUIRE_DEATH( [&] { cview v( cm, range{ 0, 4 }, range{ 0, 6 } ); (void)v; }, "matrix view: column range" ); + } + SECTION( "mutable_matrix_view constructor: r0 > r1, r1 > R, c0 > c1, c1 > C" ) + { + S2_REQUIRE_DEATH( [&] { mview v( m, range{ 3, 2 }, range{ 0, 5 } ); (void)v; }, "matrix view: row range" ); + S2_REQUIRE_DEATH( [&] { mview v( m, range{ 0, 5 }, range{ 0, 5 } ); (void)v; }, "matrix view: row range" ); + S2_REQUIRE_DEATH( [&] { mview v( m, range{ 0, 4 }, range{ 3, 2 } ); (void)v; }, "matrix view: column range" ); + S2_REQUIRE_DEATH( [&] { mview v( m, range{ 0, 4 }, range{ 0, 6 } ); (void)v; }, "matrix view: column range" ); + } + SECTION( "factory lists of size 0, 1 and 3" ) + { + S2_REQUIRE_DEATH( [&] { auto v = feng::make_view( cm, {}, { 0, 1 } ); (void)v; }, "matrix view: the row range needs exactly two values, got 0" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_view( cm, { 1 }, { 0, 1 } ); (void)v; }, "matrix view: the row range needs exactly two values, got 1" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_view( cm, { 0, 1, 2 }, { 0, 1 } ); (void)v; }, "matrix view: the row range needs exactly two values, got 3" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_view( cm, { 0, 1 }, {} ); (void)v; }, "matrix view: the column range needs exactly two values, got 0" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_view( cm, { 0, 1 }, { 1 } ); (void)v; }, "matrix view: the column range needs exactly two values, got 1" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_view( cm, { 0, 1 }, { 0, 1, 2 } ); (void)v; }, "matrix view: the column range needs exactly two values, got 3" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_mutable_view( m, {}, { 0, 1 } ); (void)v; }, "matrix view: the row range needs exactly two values, got 0" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_mutable_view( m, { 1 }, { 0, 1 } ); (void)v; }, "matrix view: the row range needs exactly two values, got 1" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_mutable_view( m, { 0, 1, 2 }, { 0, 1 } ); (void)v; }, "matrix view: the row range needs exactly two values, got 3" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_mutable_view( m, { 0, 1 }, {} ); (void)v; }, "matrix view: the column range needs exactly two values, got 0" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_mutable_view( m, { 0, 1 }, { 1 } ); (void)v; }, "matrix view: the column range needs exactly two values, got 1" ); + S2_REQUIRE_DEATH( [&] { auto v = feng::make_mutable_view( m, { 0, 1 }, { 0, 1, 2 } ); (void)v; }, "matrix view: the column range needs exactly two values, got 3" ); + } + SECTION( "empty ranges r0 == r1 and c0 == c1 give empty views; the owner's limits are accepted" ) + { + auto const require_empty = []( auto const& v, std::size_t rows, std::size_t cols ) + { + REQUIRE( v.row() == rows ); + REQUIRE( v.col() == cols ); + REQUIRE( v.size() == 0 ); + REQUIRE( v.begin() == v.end() ); + REQUIRE( std::distance( v.begin(), v.end() ) == 0 ); + }; + for ( std::size_t i = 0; i <= 4; ++i ) + { + INFO( "row range [" << i << ", " << i << ")" ); + int const ii = static_cast( i ); + require_empty( feng::make_view( cm, { ii, ii }, { 0, 5 } ), 0, 5 ); + require_empty( feng::make_mutable_view( m, { ii, ii }, { 0, 5 } ), 0, 5 ); + require_empty( cview( cm, range{ i, i }, range{ 0, 5 } ), 0, 5 ); + require_empty( mview( m, range{ i, i }, range{ 0, 5 } ), 0, 5 ); + } + for ( std::size_t j = 0; j <= 5; ++j ) + { + INFO( "column range [" << j << ", " << j << ")" ); + int const jj = static_cast( j ); + require_empty( feng::make_view( cm, { 0, 4 }, { jj, jj } ), 4, 0 ); + require_empty( feng::make_mutable_view( m, { 0, 4 }, { jj, jj } ), 4, 0 ); + require_empty( cview( cm, range{ 0, 4 }, range{ j, j } ), 4, 0 ); + require_empty( mview( m, range{ 0, 4 }, range{ j, j } ), 4, 0 ); + } + REQUIRE( feng::make_view( cm, { 0, 4 }, { 0, 5 } ).size() == 20 ); + REQUIRE( mview( m, range{ 3, 4 }, range{ 4, 5 } ).size() == 1 ); + } +} diff --git a/tests/cases/s4_r4.hpp b/tests/cases/s4_r4.hpp new file mode 100644 index 0000000..943afae --- /dev/null +++ b/tests/cases/s4_r4.hpp @@ -0,0 +1,85 @@ +// S4-R4 (PR-6): the curried map and reduce helpers hold copies of their arguments (F12). +#include +#include +#include +#include + +TEST_CASE( "S4 stored curried reduce outlives its init", "[S4][S4-R4]" ) +{ + feng::matrix m{ 3, 5 }; + std::iota( m.begin(), m.end(), 1.0 ); + + SECTION( "reduce( func, init ) keeps copies of func and init" ) + { + std::vector weights_copy; + double init_copy = 0.0; + auto f = [&] + { + std::vector weights( 4, 1.0 ); + double init = 7.0; + weights_copy = weights; + init_copy = init; + auto func = [weights]( double a, double b ) noexcept { return a + b * weights[0]; }; + return feng::matrix_details::reduce( func, init ); + }(); + auto const plus = []( double a, double b ) { return a + b; }; + REQUIRE( m.size() < 32 ); + REQUIRE( f( m ) == std::accumulate( m.begin(), m.end(), init_copy, plus ) ); + REQUIRE( weights_copy.size() == 4 ); + } + SECTION( "map( func ) keeps a copy of func" ) + { + feng::matrix b{ 3, 5 }; + std::iota( b.begin(), b.end(), 100.0 ); + auto g = [] + { + std::vector scale( 3, 2.0 ); + auto func = [scale]( double x, double y ) noexcept { return x * scale[1] + y; }; + return feng::matrix_details::map( func ); + }(); + auto const r = g( m, b ); + REQUIRE( r.row() == 3 ); + REQUIRE( r.col() == 5 ); + for ( std::size_t i = 0; i != r.size(); ++i ) + REQUIRE( r.data()[i] == m.data()[i] * 2.0 + b.data()[i] ); + } +} + +// S4-R4 (F12): reduce( func, init ) folds init in exactly once, serial or parallel, and never forms a pointer past +// m.end(); the values are small integers so every sum is exact in double whatever the grouping. +TEST_CASE( "S4 curried reduce includes init once at every size", "[S4][S4-R4]" ) +{ + auto const plus = []( double a, double b ) noexcept { return a + b; }; + for ( std::size_t n : { std::size_t{ 0 }, std::size_t{ 1 }, std::size_t{ 31 }, std::size_t{ 32 }, std::size_t{ 33 }, std::size_t{ 1000 } } ) + { + INFO( "size " << n ); + feng::matrix m{ 1, n }; + std::iota( m.begin(), m.end(), 1.0 ); + double const init = 1000003.0; + auto const f = feng::matrix_details::reduce( plus, init ); + REQUIRE( f( m ) == std::accumulate( m.begin(), m.end(), init, plus ) ); + auto const f0 = feng::matrix_details::reduce( plus, 0.0 ); + REQUIRE( f0( m ) == std::accumulate( m.begin(), m.end(), 0.0, plus ) ); + } + + SECTION( "a stored map whose callable owned a heap value runs after that scope ends" ) + { + auto g = [] + { + auto offset = std::make_unique( 3.0 ); + auto func = [offset = std::shared_ptr( std::move( offset ) )]( double x, double y ) noexcept { return x - y + *offset; }; + return feng::matrix_details::map( func ); + }(); + for ( std::size_t n : { std::size_t{ 1 }, std::size_t{ 33 }, std::size_t{ 1000 } } ) + { + INFO( "size " << n ); + feng::matrix a{ 1, n }, b{ 1, n }; + std::iota( a.begin(), a.end(), 10.0 ); + std::iota( b.begin(), b.end(), 1.0 ); + auto const r = g( a, b ); + REQUIRE( r.size() == n ); + for ( std::size_t i = 0; i != n; ++i ) + REQUIRE( r.data()[i] == 12.0 ); + } + } +} diff --git a/tests/cases/s4_r5.hpp b/tests/cases/s4_r5.hpp new file mode 100644 index 0000000..ccde5ab --- /dev/null +++ b/tests/cases/s4_r5.hpp @@ -0,0 +1,82 @@ +// S4-R5 (PR-6, PR-14): the stride iterators model the standard iterator concepts. +#include +#include +#include +#include +#include + +TEST_CASE( "S4 stride iterators satisfy the standard iterator concepts", "[S4][S4-R5]" ) +{ + using it = feng::stride_iterator; + using cit = feng::stride_iterator; + static_assert( std::random_access_iterator ); + static_assert( std::random_access_iterator ); + static_assert( std::random_access_iterator> ); + static_assert( std::random_access_iterator> ); + static_assert( std::sized_sentinel_for ); + static_assert( std::sized_sentinel_for ); + static_assert( std::sized_sentinel_for, std::reverse_iterator> ); + static_assert( std::sized_sentinel_for, std::reverse_iterator> ); + static_assert( std::output_iterator ); + static_assert( std::is_convertible_v ); + + feng::matrix m{ 4, 3 }; + std::iota( m.begin(), m.end(), 0.0 ); + auto col = std::ranges::subrange( m.col_begin( 1 ), m.col_end( 1 ) ); + REQUIRE( std::ranges::size( col ) == 4 ); + REQUIRE( std::ranges::equal( col, std::vector{ 1.0, 4.0, 7.0, 10.0 } ) ); + std::ranges::fill( col, -1.0 ); + REQUIRE( std::ranges::count( m, -1.0 ) == 4 ); + REQUIRE( m[2][1] == -1.0 ); + auto const& cm = m; + auto ccol = std::ranges::subrange( cm.col_begin( 1 ), cm.col_end( 1 ) ); + REQUIRE( std::ranges::all_of( ccol, []( double x ) { return x == -1.0; } ) ); + REQUIRE( *std::ranges::max_element( std::ranges::subrange( cm.col_begin( 2 ), cm.col_end( 2 ) ) ) == 11.0 ); +} + +// S4-R5 (PR-6, PR-14): the view element iterator models the standard iterator concepts and both view types are +// borrowed random-access ranges; the owner is not borrowed. +TEST_CASE( "S4 view types are borrowed random-access ranges", "[S4][S4-R5]" ) +{ + using A = feng::matrix::allocator_type; + using vit = feng::view_iterator; + using cvit = feng::view_iterator; + using view = feng::matrix_view; + using mview = feng::mutable_matrix_view; + static_assert( std::random_access_iterator ); + static_assert( std::random_access_iterator ); + static_assert( std::random_access_iterator> ); + static_assert( std::random_access_iterator> ); + static_assert( std::sized_sentinel_for ); + static_assert( std::sized_sentinel_for ); + static_assert( std::sized_sentinel_for, std::reverse_iterator> ); + static_assert( std::sized_sentinel_for, std::reverse_iterator> ); + static_assert( std::output_iterator ); + static_assert( std::is_convertible_v ); + static_assert( std::ranges::random_access_range ); + static_assert( std::ranges::random_access_range ); + static_assert( std::ranges::borrowed_range ); + static_assert( std::ranges::borrowed_range ); + static_assert( !std::ranges::borrowed_range> ); + static_assert( std::is_convertible_v ); + + feng::matrix m{ 4, 5 }; + std::iota( m.begin(), m.end(), 0.0 ); + auto mv = feng::make_mutable_view( m, { 1, 3 }, { 1, 4 } ); + REQUIRE( std::ranges::size( mv ) == 6 ); + REQUIRE( std::ranges::equal( mv, std::vector{ 6.0, 7.0, 8.0, 11.0, 12.0, 13.0 } ) ); + std::ranges::fill( mv, -1.0 ); + REQUIRE( std::ranges::count( m, -1.0 ) == 6 ); + REQUIRE( m( 2, 3 ) == -1.0 ); + REQUIRE( m( 2, 4 ) == 14.0 ); + view const v = mv; + REQUIRE( std::ranges::all_of( v, []( double x ) { return x == -1.0; } ) ); + auto const& cm = m; + auto const cv = feng::make_view( cm, { 0, 4 }, { 4, 5 } ); + REQUIRE( *std::ranges::max_element( cv ) == 19.0 ); + // a borrowed range: the iterator from a temporary view stays usable while the owner lives + auto const it = std::ranges::find( feng::make_view( cm, { 3, 4 }, { 0, 5 } ), 17.0 ); + REQUIRE( *it == 17.0 ); + cvit const cit = mv.begin(); + REQUIRE( *cit == -1.0 ); +} diff --git a/tests/cases/s5_r1.hpp b/tests/cases/s5_r1.hpp new file mode 100644 index 0000000..86f85e5 --- /dev/null +++ b/tests/cases/s5_r1.hpp @@ -0,0 +1,587 @@ +// S5-R1 (PR-7, F07): the NPY loader validates every header field; S5-R4 (PR-2, D-011): it is transactional. +// numpy-made fixtures live in ./tests/fixtures/s5/ (tests/fixtures/s5/make_npy.py); the suite runs from the repo root. +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include // getpid: per-process temp names + +namespace s5_r1 +{ + inline std::string fixture( char const* name ) { return std::string{ "./tests/fixtures/s5/" } + name; } + + // The process id keeps two suite binaries run by hand at once from sharing a temp path. + inline std::string temp_path( std::string const& name ) + { + return ( std::filesystem::temp_directory_path() / ( "feng_s5_r1_" + std::to_string( ::getpid() ) + "_" + name ) ).string(); + } + + inline std::string temp_file( std::string const& name, std::string const& bytes ) + { + auto const path = s5_r1::temp_path( name ); + std::ofstream ofs( path, std::ios::binary | std::ios::trunc ); + ofs.write( bytes.data(), static_cast( bytes.size() ) ); + ofs.close(); + return path; + } + + // An NPY file with the given header text (a newline is appended) and payload; version 1 uses a 2-byte length. + inline std::string npy_bytes( std::string header, std::string const& payload, int version = 1 ) + { + header += '\n'; + std::string out{ "\x93NUMPY", 6 }; + out += static_cast( version ); + out += '\0'; + std::size_t const n = header.size(); + out += static_cast( n & 0xff ); + out += static_cast( ( n >> 8 ) & 0xff ); + if ( version != 1 ) + { + out += static_cast( ( n >> 16 ) & 0xff ); + out += static_cast( ( n >> 24 ) & 0xff ); + } + return out + header + payload; + } + + template < typename T > + std::string payload_of( std::vector const& v ) + { + std::string s( v.size() * sizeof( T ), '\0' ); + if ( !v.empty() ) std::memcpy( s.data(), v.data(), s.size() ); + return s; + } + + inline feng::matrix prefilled() + { + feng::matrix m{ 2, 2 }; + m[0][0] = 1.0; m[0][1] = 2.0; m[1][0] = 3.0; m[1][1] = 4.0; + return m; + } + + inline bool is_prefilled( feng::matrix const& m ) + { + return m.row() == 2 && m.col() == 2 && m[0][0] == 1.0 && m[0][1] == 2.0 && m[1][0] == 3.0 && m[1][1] == 4.0; + } + + // Loads the synthetic file into a pre-filled 2x2 matrix; true when the load failed and kept it unchanged. + inline bool rejected( std::string const& name, std::string const& bytes ) + { + auto m = prefilled(); + bool const ok = m.load_npy( temp_file( name, bytes ) ); + return !ok && is_prefilled( m ); + } + + inline std::string const six_doubles = payload_of( std::vector{ 1.0, 2.0, 3.0, 4.0, 5.0, 6.0 } ); +} + +TEST_CASE( "S5 load_npy rejects a 3-byte file and keeps the destination", "[S5][S5-R4]" ) +{ + auto m = s5_r1::prefilled(); + std::ostringstream captured; + auto* const old = std::cerr.rdbuf( captured.rdbuf() ); + bool const ok = m.load_npy( s5_r1::temp_file( "three_bytes.npy", std::string{ "\x93NU", 3 } ) ); + std::cerr.rdbuf( old ); + REQUIRE( !ok ); + REQUIRE( s5_r1::is_prefilled( m ) ); + std::string const text = captured.str(); + REQUIRE( text.find( "load_npy" ) != std::string::npos ); + REQUIRE( std::count( text.begin(), text.end(), '\n' ) == 1 ); +} + +TEST_CASE( "S5 load_npy fails on a missing file and keeps the destination", "[S5][S5-R4]" ) +{ + auto m = s5_r1::prefilled(); + auto const path = s5_r1::temp_path( "does_not_exist.npy" ); + std::filesystem::remove( path ); + REQUIRE( !m.load_npy( path ) ); + REQUIRE( s5_r1::is_prefilled( m ) ); + REQUIRE( !m.load_npy( std::filesystem::temp_directory_path().string() ) ); // a directory + REQUIRE( s5_r1::is_prefilled( m ) ); +} + +TEST_CASE( "S5 load_npy rejects a shape whose byte size overflows", "[S5][S5-R1]" ) +{ + auto const bytes = s5_r1::npy_bytes( "{'descr': '{ 1.0 } ) ); + REQUIRE( s5_r1::rejected( "huge_shape.npy", bytes ) ); + REQUIRE( s5_r1::rejected( "huge_dim.npy", s5_r1::npy_bytes( "{'descr': ' f; + REQUIRE( f.load_npy( s5_r1::fixture( "f4_le.npy" ) ) ); + REQUIRE( f.row() == 2 ); REQUIRE( f.col() == 3 ); + REQUIRE( f[0][0] == 1.5f ); REQUIRE( f[0][1] == -2.25f ); REQUIRE( f[0][2] == 3.0f ); + REQUIRE( f[1][0] == 4.125f ); REQUIRE( f[1][1] == 5.0f ); REQUIRE( f[1][2] == -6.5f ); + + feng::matrix i8; + REQUIRE( !i8.load_npy( s5_r1::fixture( "f8_le.npy" ) ) ); + feng::matrix u4; + REQUIRE( !u4.load_npy( s5_r1::fixture( "i4_be.npy" ) ) ); + feng::matrix i1; + REQUIRE( !i1.load_npy( s5_r1::fixture( "u1.npy" ) ) ); +} + +TEST_CASE( "S5 load_npy loads little- and big-endian numpy files to numpy's values", "[S5][S5-R1]" ) +{ + for ( char const* name : { "f8_le.npy", "f8_be.npy" } ) + { + auto m = s5_r1::prefilled(); + REQUIRE( m.load_npy( s5_r1::fixture( name ) ) ); + REQUIRE( m.row() == 2 ); REQUIRE( m.col() == 3 ); + REQUIRE( m[0][0] == 1.5 ); REQUIRE( m[0][1] == -2.25 ); REQUIRE( m[0][2] == 3.0 ); + REQUIRE( m[1][0] == 4.125 ); REQUIRE( m[1][1] == 5.0 ); REQUIRE( m[1][2] == -6.5 ); + } + feng::matrix i; + REQUIRE( i.load_npy( s5_r1::fixture( "i4_be.npy" ) ) ); + REQUIRE( i.row() == 2 ); REQUIRE( i.col() == 3 ); + REQUIRE( i[0][0] == 1 ); REQUIRE( i[0][1] == -2 ); REQUIRE( i[0][2] == 3 ); + REQUIRE( i[1][0] == 70000 ); REQUIRE( i[1][1] == -80000 ); REQUIRE( i[1][2] == 2147483647 ); + for ( char const* name : { "c16_le.npy", "c16_be.npy" } ) + { + feng::matrix> c; + REQUIRE( c.load_npy( s5_r1::fixture( name ) ) ); + REQUIRE( c.row() == 2 ); REQUIRE( c.col() == 2 ); + REQUIRE( c[0][0] == std::complex{ 1.0, 2.0 } ); REQUIRE( c[0][1] == std::complex{ -3.5, 0.25 } ); + REQUIRE( c[1][0] == std::complex{ 0.0, -1.0 } ); REQUIRE( c[1][1] == std::complex{ 7.0, 0.0 } ); + } + feng::matrix u; + REQUIRE( u.load_npy( s5_r1::fixture( "u1.npy" ) ) ); + REQUIRE( u.row() == 2 ); REQUIRE( u.col() == 3 ); + REQUIRE( u[0][0] == 0 ); REQUIRE( u[0][1] == 1 ); REQUIRE( u[0][2] == 255 ); + REQUIRE( u[1][0] == 128 ); REQUIRE( u[1][1] == 7 ); REQUIRE( u[1][2] == 42 ); +} + +TEST_CASE( "S5 load_npy loads a Fortran-order 2x3 as the logical matrix", "[S5][S5-R1]" ) +{ + feng::matrix m; + REQUIRE( m.load_npy( s5_r1::fixture( "fortran_2x3.npy" ) ) ); + REQUIRE( m.row() == 2 ); REQUIRE( m.col() == 3 ); + REQUIRE( m[0][0] == 1.5 ); REQUIRE( m[0][1] == -2.25 ); REQUIRE( m[0][2] == 3.0 ); + REQUIRE( m[1][0] == 4.125 ); REQUIRE( m[1][1] == 5.0 ); REQUIRE( m[1][2] == -6.5 ); +} + +TEST_CASE( "S5 load_npy loads a 1-D (3,) array as 1x3 and rejects rank 0 and 3", "[S5][S5-R1]" ) +{ + feng::matrix m; + REQUIRE( m.load_npy( s5_r1::fixture( "one_d.npy" ) ) ); + REQUIRE( m.row() == 1 ); REQUIRE( m.col() == 3 ); + REQUIRE( m[0][0] == 10.0 ); REQUIRE( m[0][1] == 20.0 ); REQUIRE( m[0][2] == 30.0 ); + + auto p = s5_r1::prefilled(); + REQUIRE( !p.load_npy( s5_r1::fixture( "three_d.npy" ) ) ); + REQUIRE( s5_r1::is_prefilled( p ) ); + REQUIRE( s5_r1::rejected( "rank0.npy", s5_r1::npy_bytes( "{'descr': '{ 1.0 } ) ) ) ); +} + +TEST_CASE( "S5 load_npy rejects object and structured dtypes", "[S5][S5-R1]" ) +{ + for ( char const* name : { "object.npy", "structured.npy" } ) + { + auto m = s5_r1::prefilled(); + REQUIRE( !m.load_npy( s5_r1::fixture( name ) ) ); + REQUIRE( s5_r1::is_prefilled( m ) ); + } +} + +TEST_CASE( "S5 load_npy accepts the dict literal variants numpy may write", "[S5][S5-R1]" ) +{ + using s5_r1::npy_bytes; + using s5_r1::six_doubles; + for ( std::string const& header : { + std::string{ "{\"descr\": \" m; + REQUIRE( m.load_npy( s5_r1::temp_file( "variant.npy", npy_bytes( header, six_doubles, version ) ) ) ); + REQUIRE( m.row() == 2 ); REQUIRE( m.col() == 3 ); + REQUIRE( m[0][0] == 1.0 ); REQUIRE( m[1][2] == 6.0 ); + } + } + feng::matrix e; + REQUIRE( e.load_npy( s5_r1::temp_file( "empty.npy", npy_bytes( "{'descr': ' i1; + REQUIRE( i1.load_npy( s5_r1::temp_file( "i1.npy", npy_bytes( "{'descr': '|i1', 'fortran_order': False, 'shape': (1, 2), }", std::string{ "\x01\xff", 2 } ) ) ) ); + REQUIRE( i1.row() == 1 ); REQUIRE( i1.col() == 2 ); + REQUIRE( i1[0][0] == 1 ); REQUIRE( i1[0][1] == -1 ); +} + +// ---- S5-T7 (S5-R1): every fixture, every truncation, the dtype x element-type grid, malformed numbers ---- +namespace s5_r1 +{ + using all_element_types = std::tuple< std::int8_t, std::int16_t, std::int32_t, std::int64_t, std::uint8_t, std::uint16_t, + std::uint32_t, std::uint64_t, float, double, std::complex< float >, + std::complex< double >, long double >; + + template < typename T > + feng::matrix< T > prefilled_of() + { + feng::matrix< T > m{ 2, 2 }; + for ( std::size_t k = 0; k != 4; ++k ) m.data()[k] = static_cast< T >( k + 1 ); + return m; + } + + template < typename T > + bool is_prefilled_of( feng::matrix< T > const& m ) + { + if ( m.row() != 2 || m.col() != 2 ) return false; + for ( std::size_t k = 0; k != 4; ++k ) + if ( !( m.data()[k] == static_cast< T >( k + 1 ) ) ) return false; + return true; + } + + inline std::vector< std::uint8_t > file_bytes( std::string const& path ) + { + std::ifstream ifs( path, std::ios::binary ); + return std::vector< std::uint8_t >( std::istreambuf_iterator< char >( ifs ), std::istreambuf_iterator< char >() ); + } + + // In-memory parse of `bytes` into a pre-filled matrix: 1 parsed (`loaded` receives the result), 0 failed with + // the destination unchanged, -1 failed but changed the destination. + template < typename T > + int parses( std::string const& bytes, feng::matrix< T >& loaded ) + { + std::vector< std::uint8_t > const buffer( bytes.begin(), bytes.end() ); // exact size, so ASan sees overreads + auto m = prefilled_of< T >(); + if ( !feng::matrix_details::parse_npy< T >( buffer.data(), buffer.size(), m ) ) return is_prefilled_of( m ) ? 0 : -1; + loaded = m; + return 1; + } + + // Loads fixture `name` into matrix and compares it with numpy's values (row-major). + template < typename T > + bool loads_to( char const* name, std::size_t r, std::size_t c, std::vector< T > const& expected ) + { + auto m = prefilled_of< T >(); + if ( !m.load_npy( fixture( name ) ) ) return false; + if ( m.row() != r || m.col() != c || expected.size() != r * c ) return false; + for ( std::size_t k = 0; k != expected.size(); ++k ) + if ( !( m.data()[k] == expected[k] ) ) return false; + return true; + } + + // True when fixture `name` fails, with the destination unchanged, for every element type except `Except`. + template < typename Except = void > + bool rejected_by_all_but( char const* name ) + { + bool all = true; + std::apply( [&]( auto... tag ) + { + ( [&]( auto t ) + { + using U = decltype( t ); + if constexpr ( !std::is_same_v< U, Except > ) + { + auto m = prefilled_of< U >(); + if ( m.load_npy( fixture( name ) ) || !is_prefilled_of( m ) ) all = false; + } + }( tag ), ... ); + }, + all_element_types{} ); + return all; + } + + template < typename T > + constexpr char kind_of() noexcept + { + if constexpr ( std::is_same_v< T, float > || std::is_same_v< T, double > ) return 'f'; + else if constexpr ( std::is_same_v< T, std::complex< float > > || std::is_same_v< T, std::complex< double > > ) return 'c'; + else if constexpr ( std::is_integral_v< T > && std::is_signed_v< T > ) return 'i'; + else if constexpr ( std::is_integral_v< T > ) return 'u'; + else return '?'; // long double: no NPY dtype (D-021) + } +} + +TEST_CASE( "S5 load_npy loads every numpy fixture to numpy's values", "[S5][S5-R1]" ) +{ + using namespace s5_r1; + using cf = std::complex< float >; + using cd = std::complex< double >; + std::vector< double > const base{ 1.5, -2.25, 3.0, 4.125, 5.0, -6.5 }; + std::vector< float > const basef{ 1.5f, -2.25f, 3.0f, 4.125f, 5.0f, -6.5f }; + std::vector< std::int8_t > const i1{ -128, -1, 0, 1, 64, 127 }; + std::vector< std::int16_t > const i2{ -32768, -2, 0, 1, 300, 32767 }; + std::vector< std::int32_t > const i4{ 1, -2, 3, 70000, -80000, 2147483647 }; + std::vector< std::int64_t > const i8{ std::numeric_limits< std::int64_t >::min(), -1, 0, 1, 1099511627776LL, std::numeric_limits< std::int64_t >::max() }; + std::vector< std::uint8_t > const u1{ 0, 1, 255, 128, 7, 42 }; + std::vector< std::uint16_t > const u2{ 0, 1, 65535, 256, 4660, 65280 }; + std::vector< std::uint32_t > const u4{ 0, 1, 4294967295u, 65536, 305419896u, 4278190080u }; + std::vector< std::uint64_t > const u8{ 0, 1, 18446744073709551615ULL, 4294967296ULL, 0x0123456789ABCDEFULL, 0xFF00000000000000ULL }; + std::vector< cf > const c8{ { 1.0f, 2.0f }, { -3.5f, 0.25f }, { 0.0f, -1.0f }, { 7.0f, 0.0f } }; + std::vector< cd > const c16{ { 1.0, 2.0 }, { -3.5, 0.25 }, { 0.0, -1.0 }, { 7.0, 0.0 } }; + + // Each loadable fixture loads to numpy's values into its element type and into no other. + for ( char const* name : { "f8_le.npy", "f8_be.npy", "f8_v2.npy", "f8_v3.npy", "fortran_2x3.npy" } ) + { + REQUIRE( loads_to< double >( name, 2, 3, base ) ); + REQUIRE( rejected_by_all_but< double >( name ) ); + } + REQUIRE( loads_to< double >( "one_d.npy", 1, 3, { 10.0, 20.0, 30.0 } ) ); + REQUIRE( rejected_by_all_but< double >( "one_d.npy" ) ); + for ( char const* name : { "f4_le.npy", "f4_be.npy" } ) + { + REQUIRE( loads_to< float >( name, 2, 3, basef ) ); + REQUIRE( rejected_by_all_but< float >( name ) ); + } + REQUIRE( loads_to< std::int8_t >( "i1.npy", 2, 3, i1 ) ); + REQUIRE( rejected_by_all_but< std::int8_t >( "i1.npy" ) ); + for ( char const* name : { "i2_le.npy", "i2_be.npy" } ) + { + REQUIRE( loads_to< std::int16_t >( name, 2, 3, i2 ) ); + REQUIRE( rejected_by_all_but< std::int16_t >( name ) ); + } + for ( char const* name : { "i4_le.npy", "i4_be.npy" } ) + { + REQUIRE( loads_to< std::int32_t >( name, 2, 3, i4 ) ); + REQUIRE( rejected_by_all_but< std::int32_t >( name ) ); + } + for ( char const* name : { "i8_le.npy", "i8_be.npy" } ) + { + REQUIRE( loads_to< std::int64_t >( name, 2, 3, i8 ) ); + REQUIRE( rejected_by_all_but< std::int64_t >( name ) ); + } + REQUIRE( loads_to< std::uint8_t >( "u1.npy", 2, 3, u1 ) ); + REQUIRE( rejected_by_all_but< std::uint8_t >( "u1.npy" ) ); + for ( char const* name : { "u2_le.npy", "u2_be.npy" } ) + { + REQUIRE( loads_to< std::uint16_t >( name, 2, 3, u2 ) ); + REQUIRE( rejected_by_all_but< std::uint16_t >( name ) ); + } + for ( char const* name : { "u4_le.npy", "u4_be.npy" } ) + { + REQUIRE( loads_to< std::uint32_t >( name, 2, 3, u4 ) ); + REQUIRE( rejected_by_all_but< std::uint32_t >( name ) ); + } + for ( char const* name : { "u8_le.npy", "u8_be.npy" } ) + { + REQUIRE( loads_to< std::uint64_t >( name, 2, 3, u8 ) ); + REQUIRE( rejected_by_all_but< std::uint64_t >( name ) ); + } + for ( char const* name : { "c8_le.npy", "c8_be.npy" } ) + { + REQUIRE( loads_to< cf >( name, 2, 2, c8 ) ); + REQUIRE( rejected_by_all_but< cf >( name ) ); + } + for ( char const* name : { "c16_le.npy", "c16_be.npy" } ) + { + REQUIRE( loads_to< cd >( name, 2, 2, c16 ) ); + REQUIRE( rejected_by_all_but< cd >( name ) ); + } + // 3-D, object, structured, bool and f2 fixtures fail for every element type. + for ( char const* name : { "three_d.npy", "object.npy", "structured.npy", "bool.npy", "f2.npy" } ) + REQUIRE( rejected_by_all_but( name ) ); +} + +TEST_CASE( "S5 load_npy fails at every truncation of a valid v1.0 and v2.0 file", "[S5][S5-R1]" ) +{ + using namespace s5_r1; + for ( char const* name : { "f8_le.npy", "f8_v2.npy" } ) + { + auto const bytes = file_bytes( fixture( name ) ); + REQUIRE( bytes.size() == 176 ); + { + auto whole = prefilled_of< double >(); + REQUIRE( feng::matrix_details::parse_npy< double >( bytes.data(), bytes.size(), whole ) ); + } + for ( std::size_t n = 0; n != bytes.size(); ++n ) + { + std::vector< std::uint8_t > const prefix( bytes.begin(), bytes.begin() + static_cast< std::ptrdiff_t >( n ) ); + auto m = prefilled_of< double >(); + INFO( name << " cut at " << n ); + REQUIRE( !feng::matrix_details::parse_npy< double >( prefix.data(), prefix.size(), m ) ); + REQUIRE( is_prefilled_of( m ) ); + } + } + // The file loader agrees for a few cuts (empty file, inside the magic, the header and the payload). + for ( std::size_t n : { std::size_t{ 0 }, std::size_t{ 5 }, std::size_t{ 60 }, std::size_t{ 175 } } ) + { + auto const bytes = file_bytes( fixture( "f8_v2.npy" ) ); + REQUIRE( rejected( "cut_v2.npy", std::string( bytes.begin(), bytes.begin() + static_cast< std::ptrdiff_t >( n ) ) ) ); + } +} + +TEST_CASE( "S5 load_npy loads a dtype only into the element type of the same kind and size", "[S5][S5-R1]" ) +{ + using namespace s5_r1; + struct dtype { char kind; std::size_t size; }; + dtype const dtypes[] = { { 'i', 1 }, { 'i', 2 }, { 'i', 4 }, { 'i', 8 }, { 'u', 1 }, { 'u', 2 }, { 'u', 4 }, { 'u', 8 }, + { 'f', 4 }, { 'f', 8 }, { 'c', 8 }, { 'c', 16 } }; + std::size_t accepted = 0; + for ( char const order : { '<', '>', '=', '|' } ) + for ( dtype const d : dtypes ) + { + std::string const descr = std::string{ order, d.kind } + std::to_string( d.size ); + std::string payload( 2 * d.size, '\0' ); + for ( std::size_t k = 0; k != payload.size(); ++k ) payload[k] = static_cast< char >( k + 1 ); + auto const bytes = npy_bytes( "{'descr': '" + descr + "', 'fortran_order': False, 'shape': (1, 2), }", payload ); + bool const foreign = ( order == '>' && std::endian::native == std::endian::little ) + || ( order == '<' && std::endian::native == std::endian::big ); + std::apply( [&]( auto... tag ) + { + ( [&]( auto t ) + { + using T = decltype( t ); + bool const expect = kind_of< T >() == d.kind && sizeof( T ) == d.size && ( order != '|' || d.size == 1 ); + feng::matrix< T > loaded; + int const result = parses< T >( bytes, loaded ); + INFO( descr << " into an element of " << sizeof( T ) << " bytes, kind " << kind_of< T >() ); + REQUIRE( result == ( expect ? 1 : 0 ) ); + if ( result != 1 ) return; + ++accepted; + REQUIRE( loaded.row() == 1 ); + REQUIRE( loaded.col() == 2 ); + // Foreign order reverses each component (each half of a complex element). + std::string want = payload; + std::size_t const unit = d.kind == 'c' ? d.size / 2 : d.size; + if ( foreign ) + for ( std::size_t u = 0; u != want.size(); u += unit ) + std::reverse( want.begin() + static_cast< std::ptrdiff_t >( u ), + want.begin() + static_cast< std::ptrdiff_t >( u + unit ) ); + REQUIRE( std::memcmp( loaded.data(), want.data(), want.size() ) == 0 ); + }( tag ), ... ); + }, + all_element_types{} ); + } + REQUIRE( accepted == 12 * 3 + 2 ); // '<', '>' and '=' for every dtype; '|' only for i1 and u1 +} + +TEST_CASE( "S5 load_npy rejects malformed dicts and numbers with leading zeros, signs or overflow", "[S5][S5-R1]" ) +{ + using s5_r1::npy_bytes; + using s5_r1::rejected; + using s5_r1::six_doubles; + auto const header = []( std::string const& descr, std::string const& shape ) + { return "{'descr': '" + descr + "', 'fortran_order': False, 'shape': " + shape + ", }"; }; + REQUIRE( !rejected( "control.npy", npy_bytes( header( " i; + REQUIRE( !i.load_npy( s5_r1::temp_file( "i04.npy", npy_bytes( header( "( ifs ), std::istreambuf_iterator< char >() }; + REQUIRE( text.size() > 100000 ); + std::string hits; + for ( std::size_t p = text.find( "std::sto" ); p != std::string::npos; p = text.find( "std::sto", p + 1 ) ) + { + std::size_t e = p + 8; + while ( e != text.size() && ( std::isalnum( static_cast< unsigned char >( text[e] ) ) || text[e] == '_' ) ) ++e; + std::string const name = text.substr( p + 5, e - p - 5 ); + for ( char const* banned : { "stoi", "stol", "stoll", "stoul", "stoull", "stof", "stod", "stold" } ) + if ( name == banned ) hits += "std::" + name + " at offset " + std::to_string( p ) + "; "; + } + INFO( hits ); + REQUIRE( hits.empty() ); +} diff --git a/tests/cases/s5_r2.hpp b/tests/cases/s5_r2.hpp new file mode 100644 index 0000000..38d7f5f --- /dev/null +++ b/tests/cases/s5_r2.hpp @@ -0,0 +1,553 @@ +// S5-R2 (PR-7, F08): text and native binary inputs are bounded and typed; S5-R4 (PR-2, D-011): their loaders are +// transactional. Temporary files go under std::filesystem::temp_directory_path(). +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include // getpid: per-process temp names + +namespace s5_r2 +{ + // The process id keeps two suite binaries run by hand at once from sharing a temp path. + inline std::string temp_path( std::string const& name ) + { + return ( std::filesystem::temp_directory_path() / ( "feng_s5_r2_" + std::to_string( ::getpid() ) + "_" + name ) ).string(); + } + + inline std::string temp_file( std::string const& name, std::string const& bytes ) + { + auto const path = temp_path( name ); + std::ofstream ofs( path, std::ios::binary | std::ios::trunc ); + ofs.write( bytes.data(), static_cast( bytes.size() ) ); + ofs.close(); + return path; + } + + template < typename T = double > + feng::matrix prefilled() + { + feng::matrix m{ 2, 2 }; + m[0][0] = T{ 1 }; m[0][1] = T{ 2 }; m[1][0] = T{ 3 }; m[1][1] = T{ 4 }; + return m; + } + + template < typename T = double > + bool is_prefilled( feng::matrix const& m ) + { + return m.row() == 2 && m.col() == 2 && m[0][0] == T{ 1 } && m[0][1] == T{ 2 } && m[1][0] == T{ 3 } && m[1][1] == T{ 4 }; + } + + // Loads the text into a pre-filled 2x2 matrix; true when the load failed and left it unchanged. + template < typename T = double > + bool txt_rejected( std::string const& name, std::string const& text ) + { + auto m = prefilled(); + bool const ok = m.load_txt( temp_file( name, text ) ); + return !ok && is_prefilled( m ); + } + + template < typename T = double > + bool bin_rejected( std::string const& name, std::string const& bytes ) + { + auto m = prefilled(); + bool const ok = m.load_binary( temp_file( name, bytes ) ); + return !ok && is_prefilled( m ); + } + + // A native binary image: two size_type counts, then the raw elements. + template < typename T > + std::string binary_bytes( std::uint64_t r, std::uint64_t c, std::vector const& v ) + { + using size_type = typename feng::matrix::size_type; + size_type const rr = static_cast( r ); + size_type const cc = static_cast( c ); + std::string s( 2 * sizeof( size_type ) + v.size() * sizeof( T ), '\0' ); + std::memcpy( s.data(), &rr, sizeof( rr ) ); + std::memcpy( s.data() + sizeof( rr ), &cc, sizeof( cc ) ); + if ( !v.empty() ) std::memcpy( s.data() + 2 * sizeof( size_type ), v.data(), v.size() * sizeof( T ) ); + return s; + } + + template < typename T > + bool same( feng::matrix const& a, feng::matrix const& b ) + { + return a.row() == b.row() && a.col() == b.col() && std::equal( a.begin(), a.end(), b.begin() ); + } +} + +TEST_CASE( "S5 load_txt rejects empty and ragged text and keeps the destination", "[S5][S5-R2][S5-R4]" ) +{ + REQUIRE( s5_r2::txt_rejected( "empty.txt", "" ) ); + REQUIRE( s5_r2::txt_rejected( "blank.txt", " \n\t\r\n,;\n\n" ) ); + REQUIRE( s5_r2::txt_rejected( "ragged.txt", "1 2 3\n4 5\n" ) ); + REQUIRE( s5_r2::txt_rejected( "ragged2.txt", "1\n2 3\n" ) ); + + auto m = s5_r2::prefilled(); + std::ostringstream captured; + auto* const old = std::cerr.rdbuf( captured.rdbuf() ); + bool const ok = m.load_txt( s5_r2::temp_file( "ragged3.txt", "1,2\n3\n" ) ); + std::cerr.rdbuf( old ); + REQUIRE( !ok ); + REQUIRE( s5_r2::is_prefilled( m ) ); + std::string const text = captured.str(); + REQUIRE( text.find( "load_txt" ) != std::string::npos ); + REQUIRE( std::count( text.begin(), text.end(), '\n' ) == 1 ); + + auto const missing = s5_r2::temp_path( "does_not_exist.txt" ); + std::filesystem::remove( missing ); + REQUIRE( !m.load_txt( missing ) ); + REQUIRE( !m.load_txt( std::filesystem::temp_directory_path().string() ) ); + REQUIRE( s5_r2::is_prefilled( m ) ); +} + +TEST_CASE( "S5 load_txt rejects truncated and oversized tokens", "[S5][S5-R2][S5-R4]" ) +{ + REQUIRE( s5_r2::txt_rejected( "trunc_exp.txt", "1 2\n3 1.5e\n" ) ); + REQUIRE( s5_r2::txt_rejected( "junk.txt", "1 2\n3 4x\n" ) ); + REQUIRE( s5_r2::txt_rejected( "double_plus.txt", "1 ++2\n" ) ); + REQUIRE( s5_r2::txt_rejected( "plus_minus.txt", "1 +-2\n" ) ); + REQUIRE( s5_r2::txt_rejected( "huge_double.txt", "1 1e999\n" ) ); + REQUIRE( s5_r2::txt_rejected( "u8_300.txt", "1 300\n" ) ); + REQUIRE( s5_r2::txt_rejected( "u8_neg.txt", "1 -1\n" ) ); + REQUIRE( s5_r2::txt_rejected( "int_fraction.txt", "1 2.5\n" ) ); + REQUIRE( s5_r2::txt_rejected( "int_huge.txt", "1 99999999999999999999\n" ) ); + REQUIRE( s5_r2::txt_rejected( "i8_128.txt", "128\n" ) ); + REQUIRE( s5_r2::txt_rejected>( "cx_open.txt", "(1,2 3\n" ) ); + REQUIRE( s5_r2::txt_rejected>( "cx_one.txt", "(1)\n" ) ); + REQUIRE( s5_r2::txt_rejected>( "cx_three.txt", "(1,2,3)\n" ) ); +} + +TEST_CASE( "S5 load_txt accepts separators, blank lines, CRLF, a plus sign and complex pairs", "[S5][S5-R2]" ) +{ + feng::matrix m; + REQUIRE( m.load_txt( s5_r2::temp_file( "seps.txt", "\n1,2;3\r\n \n+4\t5 -6.25e1\r\n\n" ) ) ); + REQUIRE( m.row() == 2 ); REQUIRE( m.col() == 3 ); + REQUIRE( m[0][0] == 1.0 ); REQUIRE( m[0][1] == 2.0 ); REQUIRE( m[0][2] == 3.0 ); + REQUIRE( m[1][0] == 4.0 ); REQUIRE( m[1][1] == 5.0 ); REQUIRE( m[1][2] == -62.5 ); + + feng::matrix u; + REQUIRE( u.load_txt( s5_r2::temp_file( "u8.txt", "0 255\n+7 32" ) ) ); + REQUIRE( u.row() == 2 ); REQUIRE( u.col() == 2 ); + REQUIRE( u[0][0] == 0 ); REQUIRE( u[0][1] == 255 ); REQUIRE( u[1][0] == 7 ); REQUIRE( u[1][1] == 32 ); + + feng::matrix> z; + REQUIRE( z.load_txt( s5_r2::temp_file( "cx.txt", "(1,2) 3\n(-1.5,+0.5),(4, 0)\n" ) ) ); + REQUIRE( z.row() == 2 ); REQUIRE( z.col() == 2 ); + REQUIRE( z[0][0] == std::complex( 1, 2 ) ); REQUIRE( z[0][1] == std::complex( 3, 0 ) ); + REQUIRE( z[1][0] == std::complex( -1.5, 0.5 ) ); REQUIRE( z[1][1] == std::complex( 4, 0 ) ); +} + +TEST_CASE( "S5 save_as_txt output loads back for double, int and uint8_t", "[S5][S5-R2]" ) +{ + { + feng::matrix a{ 2, 3 }; + a[0][0] = 0.1; a[0][1] = 1.0 / 3.0; a[0][2] = -2.5e-300; + a[1][0] = 1.0e300; a[1][1] = std::numeric_limits::min(); a[1][2] = -0.0; + auto const path = s5_r2::temp_path( "rt_double.txt" ); + REQUIRE( a.save_as_txt( path ) ); + feng::matrix b; + REQUIRE( b.load_txt( path ) ); + REQUIRE( s5_r2::same( a, b ) ); + } + { + feng::matrix a{ 2, 2 }; + a[0][0] = std::numeric_limits::min(); a[0][1] = std::numeric_limits::max(); a[1][0] = 0; a[1][1] = -7; + auto const path = s5_r2::temp_path( "rt_int.txt" ); + REQUIRE( a.save_as_txt( path ) ); + feng::matrix b; + REQUIRE( b.load_txt( path ) ); + REQUIRE( s5_r2::same( a, b ) ); + } + { + feng::matrix a{ 2, 3 }; + a[0][0] = 0; a[0][1] = 9; a[0][2] = 10; a[1][0] = 32; a[1][1] = 44; a[1][2] = 255; + auto const path = s5_r2::temp_path( "rt_u8.txt" ); + REQUIRE( a.save_as_txt( path ) ); + feng::matrix b; + REQUIRE( b.load_txt( path ) ); + REQUIRE( s5_r2::same( a, b ) ); + } + { + feng::matrix a{ 1, 3 }; + a[0][0] = -128; a[0][1] = 0; a[0][2] = 127; + auto const path = s5_r2::temp_path( "rt_i8.txt" ); + REQUIRE( a.save_as_txt( path ) ); + feng::matrix b; + REQUIRE( b.load_txt( path ) ); + REQUIRE( s5_r2::same( a, b ) ); + } +} + +TEST_CASE( "S5 operator>> parses text and sets failbit without touching the matrix on bad input", "[S5][S5-R2][S5-R4]" ) +{ + feng::matrix m; + std::istringstream good{ "1\t2\t\n3\t4\t\n" }; + good >> m; + REQUIRE( !good.fail() ); + REQUIRE( s5_r2::is_prefilled( m ) ); + + std::istringstream ragged{ "1 2\n3\n" }; + ragged >> m; + REQUIRE( ragged.fail() ); + REQUIRE( s5_r2::is_prefilled( m ) ); + + std::istringstream empty{ "" }; + empty >> m; + REQUIRE( empty.fail() ); + REQUIRE( s5_r2::is_prefilled( m ) ); +} + +TEST_CASE( "S5 load_binary rejects short, long and overflowing-count files into a pre-filled matrix", "[S5][S5-R2][S5-R4]" ) +{ + REQUIRE( s5_r2::bin_rejected( "empty.bin", "" ) ); + REQUIRE( s5_r2::bin_rejected( "three.bin", "abc" ) ); + auto const header_only = s5_r2::binary_bytes( 2, 2, {} ); + REQUIRE( s5_r2::bin_rejected( "one_count.bin", header_only.substr( 0, 8 ) ) ); + REQUIRE( s5_r2::bin_rejected( "short.bin", s5_r2::binary_bytes( 2, 2, { 1, 2, 3 } ) ) ); + REQUIRE( s5_r2::bin_rejected( "long.bin", s5_r2::binary_bytes( 2, 2, { 1, 2, 3, 4, 5 } ) ) ); + auto trailing = s5_r2::binary_bytes( 2, 2, { 1, 2, 3, 4 } ); + trailing += '\0'; + REQUIRE( s5_r2::bin_rejected( "trailing.bin", trailing ) ); + REQUIRE( s5_r2::bin_rejected( "count_overflow.bin", s5_r2::binary_bytes( 1ull << 32, 1ull << 32, { 1 } ) ) ); + REQUIRE( s5_r2::bin_rejected( "bytes_overflow.bin", s5_r2::binary_bytes( 1ull << 62, 1, { 1 } ) ) ); + REQUIRE( s5_r2::bin_rejected( "max_counts.bin", s5_r2::binary_bytes( ~0ull, ~0ull, { 1 } ) ) ); + REQUIRE( s5_r2::bin_rejected( "huge_rows.bin", s5_r2::binary_bytes( 1ull << 40, 1, { 1 } ) ) ); + + auto m = s5_r2::prefilled(); + std::ostringstream captured; + auto* const old = std::cerr.rdbuf( captured.rdbuf() ); + bool const ok = m.load_binary( s5_r2::temp_file( "short2.bin", s5_r2::binary_bytes( 2, 2, { 1 } ) ) ); + std::cerr.rdbuf( old ); + REQUIRE( !ok ); + REQUIRE( s5_r2::is_prefilled( m ) ); + std::string const text = captured.str(); + REQUIRE( text.find( "load_binary" ) != std::string::npos ); + REQUIRE( std::count( text.begin(), text.end(), '\n' ) == 1 ); + + auto const missing = s5_r2::temp_path( "does_not_exist.bin" ); + std::filesystem::remove( missing ); + REQUIRE( !m.load_binary( missing ) ); + REQUIRE( s5_r2::is_prefilled( m ) ); +} + +TEST_CASE( "S5 load_binary round-trips 0x3, 3x0 and filled matrices", "[S5][S5-R2]" ) +{ + for ( auto const& [r, c] : { std::pair{ 0, 3 }, { 3, 0 }, { 2, 3 } } ) + { + feng::matrix a{ r, c }; + for ( std::size_t k = 0; k != a.size(); ++k ) a.data()[k] = 0.5 * static_cast( k ) - 1.0; + auto const path = s5_r2::temp_path( "rt_" + std::to_string( r ) + "x" + std::to_string( c ) + ".bin" ); + REQUIRE( a.save_as_binary( path ) ); + auto b = s5_r2::prefilled(); + REQUIRE( b.load_binary( path ) ); + REQUIRE( b.row() == r ); + REQUIRE( b.col() == c ); + REQUIRE( s5_r2::same( a, b ) ); + } + feng::matrix> z{ 1, 2 }; + z[0][0] = { 1.0f, -2.0f }; z[0][1] = { 0.25f, 8.0f }; + auto const path = s5_r2::temp_path( "rt_cf.bin" ); + REQUIRE( z.save_as_binary( path ) ); + feng::matrix> w; + REQUIRE( w.load_binary( path ) ); + REQUIRE( s5_r2::same( z, w ) ); +} + +namespace s5_r2 +{ + // Records the largest single request (in bytes) any destination matrix made through this allocator. + inline std::size_t largest_request = 0; + + template < typename T > + struct max_request_allocator + { + using value_type = T; + max_request_allocator() noexcept = default; + template < typename U > max_request_allocator( max_request_allocator const& ) noexcept {} + T* allocate( std::size_t n ) + { + largest_request = std::max( largest_request, n * sizeof( T ) ); + return std::allocator{}.allocate( n ); + } + void deallocate( T* p, std::size_t n ) noexcept { std::allocator{}.deallocate( p, n ); } + template < typename U > friend bool operator == ( max_request_allocator const&, max_request_allocator const& ) noexcept { return true; } + }; + + template < typename T > + using tracked_matrix = feng::matrix< T, max_request_allocator >; + + // A version 1.0 NPY image with the given header dict and payload; the header is padded to a multiple of 64. + inline std::string npy_bytes( std::string const& dict, std::string const& payload ) + { + std::string header = dict; + while ( ( 10 + header.size() + 1 ) % 64 != 0 ) header += ' '; + header += '\n'; + std::string s = std::string( "\x93NUMPY\x01\x00", 8 ); + s += static_cast( header.size() & 0xff ); + s += static_cast( header.size() >> 8 ); + return s + header + payload; + } + + template < typename T > + std::string raw( std::vector const& v ) + { + std::string s( v.size() * sizeof( T ), '\0' ); + if ( !v.empty() ) std::memcpy( s.data(), v.data(), s.size() ); + return s; + } + + // Runs `load` on the bytes into a pre-filled tracked matrix; true when the largest destination request is at + // most sizeof(T) * (file size + 1). + template < typename T, typename Load > + bool bounded( std::string const& name, std::string const& bytes, Load load ) + { + tracked_matrix m{ 2, 2, T{ 1 } }; + auto const path = temp_file( name, bytes ); + largest_request = 0; + (void)load( m, path ); + return largest_request <= sizeof( T ) * ( bytes.size() + 1 ); + } + + template < typename T > + std::vector extremes() + { + if constexpr ( std::is_integral_v ) + return { std::numeric_limits::lowest(), std::numeric_limits::max(), T{ 0 }, T{ 1 }, static_cast( std::numeric_limits::max() - 1 ), static_cast( std::numeric_limits::lowest() + 1 ) }; + else + return { std::numeric_limits::lowest(), std::numeric_limits::max(), std::numeric_limits::min(), std::numeric_limits::denorm_min(), + -std::numeric_limits::denorm_min(), std::numeric_limits::epsilon(), T{ 1 } / T{ 3 }, T{ 0 } }; + } + + template < typename T > + std::vector extreme_elements() + { + if constexpr ( std::is_floating_point_v || std::is_integral_v ) + return extremes(); + else + { + using F = typename T::value_type; + auto const e = extremes(); + std::vector v; + for ( std::size_t k = 0; k != e.size(); ++k ) v.emplace_back( e[k], e[e.size() - 1 - k] ); + return v; + } + } + + // A 2 x (n/2) matrix holding the type's extreme values. + template < typename T > + feng::matrix extreme_matrix() + { + auto const v = extreme_elements(); + feng::matrix a{ 2, static_cast( v.size() / 2 ) }; + std::copy( v.begin(), v.begin() + static_cast( a.size() ), a.begin() ); + return a; + } + + template < typename T > + bool txt_round_trips( std::string const& name ) + { + auto const a = extreme_matrix(); + auto const path = temp_path( name ); + if ( !a.save_as_txt( path ) ) return false; + feng::matrix b; + return b.load_txt( path ) && same( a, b ); + } + + template < typename T > + bool bin_round_trips( std::string const& name ) + { + auto const a = extreme_matrix(); + auto const path = temp_path( name ); + if ( !a.save_as_binary( path ) ) return false; + auto b = prefilled(); + return b.load_binary( path ) && same( a, b ); + } + + // Integer type T rejects max+1 and min-1 (given as text). + template < typename T > + bool int_out_of_range_rejected( std::string const& above, std::string const& below ) + { + return txt_rejected( "above.txt", "1 " + above + "\n" ) && txt_rejected( "below.txt", "1 " + below + "\n" ); + } +} + +TEST_CASE( "S5 every loader's largest allocation is bounded by the file size", "[S5][S5-R2]" ) +{ + auto txt = []( auto& m, std::string const& p ) { return m.load_txt( p ); }; + auto bin = []( auto& m, std::string const& p ) { return m.load_binary( p ); }; + auto npy = []( auto& m, std::string const& p ) { return m.load_npy( p ); }; + + // load_txt: valid, ragged, bad token, blank and empty inputs. + for ( std::string const& text : std::vector{ "1 2 3\n4 5 6\n", "1\n", "1,2;3 4\t5\r\n", "1 2\n3\n", "1 2\n3 x\n", "\n\n", "", + std::string( 4096, '1' ) + "\n", "(1,2) (3,4)\n" } ) + { + REQUIRE( s5_r2::bounded( "bnd.txt", text, txt ) ); + REQUIRE( s5_r2::bounded( "bnd.txt", text, txt ) ); + REQUIRE( s5_r2::bounded>( "bnd.txt", text, txt ) ); + } + // The densest text: one-character tokens with one-character separators. + std::string dense; + for ( int k = 0; k != 1000; ++k ) dense += "1 "; + REQUIRE( s5_r2::bounded( "dense.txt", dense, txt ) ); + REQUIRE( s5_r2::largest_request == 1000 ); + REQUIRE( s5_r2::bounded( "dense.txt", dense + "\n" + dense, txt ) ); + + // load_binary: valid, empty-shape, short, long and huge declared shapes. + REQUIRE( s5_r2::bounded( "bnd.bin", s5_r2::binary_bytes( 2, 3, { 1, 2, 3, 4, 5, 6 } ), bin ) ); + REQUIRE( s5_r2::largest_request == 6 * sizeof( double ) ); // the destination allocator is the one tracked + REQUIRE( s5_r2::bounded( "bnd.bin", s5_r2::binary_bytes( 0, 3, {} ), bin ) ); + REQUIRE( s5_r2::bounded( "bnd.bin", s5_r2::binary_bytes( 2, 3, { 1 } ), bin ) ); + REQUIRE( s5_r2::bounded( "bnd.bin", s5_r2::binary_bytes( 1, 1, { 1, 2 } ), bin ) ); + REQUIRE( s5_r2::bounded( "bnd.bin", s5_r2::binary_bytes( 1ull << 40, 1ull << 20, { 1 } ), bin ) ); + REQUIRE( s5_r2::bounded( "bnd.bin", s5_r2::binary_bytes( 1ull << 28, 1, { 1 } ), bin ) ); + REQUIRE( s5_r2::bounded( "bnd.bin", s5_r2::binary_bytes( ~0ull, ~0ull, {} ), bin ) ); + REQUIRE( s5_r2::bounded( "bnd.bin", "abc", bin ) ); + REQUIRE( s5_r2::bounded( "bnd.bin", s5_r2::binary_bytes( 1ull << 30, 1ull << 30, { 1, 2, 3 } ), bin ) ); + + // load_npy: valid C and Fortran order, short payloads and headers declaring huge shapes. + auto const six = s5_r2::raw( { 1, 2, 3, 4, 5, 6 } ); + REQUIRE( s5_r2::bounded( "bnd.npy", s5_r2::npy_bytes( "{'descr': '( "bnd.npy", s5_r2::npy_bytes( "{'descr': '( "bnd.npy", s5_r2::npy_bytes( "{'descr': '( "bnd.npy", s5_r2::npy_bytes( "{'descr': '( "bnd.npy", s5_r2::npy_bytes( "{'descr': '( "bnd.npy", s5_r2::npy_bytes( "{'descr': '( "bnd.npy", s5_r2::npy_bytes( "{'descr': '( "bnd.npy", s5_r2::npy_bytes( "{'descr': '( "bnd.npy", "\x93NUMPY", npy ) ); + REQUIRE( s5_r2::bounded( "bnd.npy", s5_r2::npy_bytes( "{'descr': '|u1', 'fortran_order': False, 'shape': (65536, 65536), }", "abc" ), npy ) ); + REQUIRE( s5_r2::bounded( "bnd.npy", s5_r2::npy_bytes( "{'descr': '|u1', 'fortran_order': False, 'shape': (1, 3), }", "abc" ), npy ) ); +} + +TEST_CASE( "S5 load_txt delimiters, out-of-range values per type, underflow and complex forms", "[S5][S5-R2][S5-R4]" ) +{ + // Each delimiter alone separates tokens; CRLF and blank lines are accepted. + for ( std::string const& sep : std::vector{ ",", ";", " ", "\t", "\r" } ) + { + feng::matrix m; + REQUIRE( m.load_txt( s5_r2::temp_file( "sep.txt", "1" + sep + "2\r\n\r\n\n+3" + sep + "-4\r\n" ) ) ); + REQUIRE( m.row() == 2 ); REQUIRE( m.col() == 2 ); + REQUIRE( m[0][0] == 1.0 ); REQUIRE( m[0][1] == 2.0 ); REQUIRE( m[1][0] == 3.0 ); REQUIRE( m[1][1] == -4.0 ); + } + + REQUIRE( s5_r2::int_out_of_range_rejected( "128", "-129" ) ); + REQUIRE( s5_r2::int_out_of_range_rejected( "32768", "-32769" ) ); + REQUIRE( s5_r2::int_out_of_range_rejected( "2147483648", "-2147483649" ) ); + REQUIRE( s5_r2::int_out_of_range_rejected( "9223372036854775808", "-9223372036854775809" ) ); + REQUIRE( s5_r2::int_out_of_range_rejected( "256", "-1" ) ); + REQUIRE( s5_r2::int_out_of_range_rejected( "65536", "-1" ) ); + REQUIRE( s5_r2::int_out_of_range_rejected( "4294967296", "-1" ) ); + REQUIRE( s5_r2::int_out_of_range_rejected( "18446744073709551616", "-1" ) ); + REQUIRE( s5_r2::txt_rejected( "f_1e39.txt", "1 1e39\n" ) ); + REQUIRE( s5_r2::txt_rejected( "f_m1e39.txt", "1 -1e39\n" ) ); + REQUIRE( s5_r2::txt_rejected( "d_1e999.txt", "1 1e999\n" ) ); + // Underflow: the implementation rejects 1e-400 into double as out of range (from_chars reports it). + REQUIRE( s5_r2::txt_rejected( "d_1em400.txt", "1 1e-400\n" ) ); + REQUIRE( s5_r2::txt_rejected>( "cx_1e999.txt", "(1,1e999)\n" ) ); + + // The largest in-range integers load. + feng::matrix i; + REQUIRE( i.load_txt( s5_r2::temp_file( "i64.txt", "9223372036854775807 -9223372036854775808\n" ) ) ); + REQUIRE( i[0][0] == std::numeric_limits::max() ); + REQUIRE( i[0][1] == std::numeric_limits::min() ); + + feng::matrix> z; + REQUIRE( z.load_txt( s5_r2::temp_file( "cxf.txt", "(+1.5,-2)\t7\r\n-3;( 0 , 1 )\r\n" ) ) ); + REQUIRE( z.row() == 2 ); REQUIRE( z.col() == 2 ); + REQUIRE( z[0][0] == std::complex( 1.5f, -2.0f ) ); REQUIRE( z[0][1] == std::complex( 7.0f, 0.0f ) ); + REQUIRE( z[1][0] == std::complex( -3.0f, 0.0f ) ); REQUIRE( z[1][1] == std::complex( 0.0f, 1.0f ) ); +} + +TEST_CASE( "S5 save_as_txt output loads back for every arithmetic and complex type at its extremes", "[S5][S5-R2]" ) +{ + REQUIRE( s5_r2::txt_round_trips( "rtx_i8.txt" ) ); + REQUIRE( s5_r2::txt_round_trips( "rtx_i16.txt" ) ); + REQUIRE( s5_r2::txt_round_trips( "rtx_i32.txt" ) ); + REQUIRE( s5_r2::txt_round_trips( "rtx_i64.txt" ) ); + REQUIRE( s5_r2::txt_round_trips( "rtx_u8.txt" ) ); + REQUIRE( s5_r2::txt_round_trips( "rtx_u16.txt" ) ); + REQUIRE( s5_r2::txt_round_trips( "rtx_u32.txt" ) ); + REQUIRE( s5_r2::txt_round_trips( "rtx_u64.txt" ) ); + REQUIRE( s5_r2::txt_round_trips( "rtx_f.txt" ) ); + REQUIRE( s5_r2::txt_round_trips( "rtx_d.txt" ) ); + //REQUIRE( s5_r2::txt_round_trips( "rtx_ld.txt" ) ); + REQUIRE( s5_r2::txt_round_trips>( "rtx_cf.txt" ) ); + REQUIRE( s5_r2::txt_round_trips>( "rtx_cd.txt" ) ); + //REQUIRE( s5_r2::txt_round_trips>( "rtx_cld.txt" ) ); +} + +TEST_CASE( "S5 operator>> reads every element type and leaves the matrix unchanged on a bad token", "[S5][S5-R2][S5-R4]" ) +{ + feng::matrix> z; + std::istringstream good{ "(1,2);3\r\n\n4 (5,-6)\n" }; + good >> z; + REQUIRE( !good.fail() ); + REQUIRE( z.row() == 2 ); REQUIRE( z.col() == 2 ); + REQUIRE( z[0][0] == std::complex( 1, 2 ) ); REQUIRE( z[1][1] == std::complex( 5, -6 ) ); + + auto m = s5_r2::prefilled(); + std::istringstream bad_token{ "1 2\n3 x\n" }; + bad_token >> m; + REQUIRE( bad_token.fail() ); + REQUIRE( s5_r2::is_prefilled( m ) ); + + std::istringstream out_of_range{ "1 2\n3 256\n" }; + out_of_range >> m; + REQUIRE( out_of_range.fail() ); + REQUIRE( s5_r2::is_prefilled( m ) ); + + std::istringstream ok{ "4 3\n2 1\n" }; + ok >> m; + REQUIRE( !ok.fail() ); + REQUIRE( m[0][0] == 4 ); REQUIRE( m[1][1] == 1 ); +} + +TEST_CASE( "S5 load_binary fails at every truncation and round-trips every element type", "[S5][S5-R2][S5-R4]" ) +{ + auto const valid = s5_r2::binary_bytes( 2, 2, { 5, 6, 7, 8 } ); + for ( std::size_t k = 0; k != valid.size(); ++k ) + REQUIRE( s5_r2::bin_rejected( "trunc.bin", valid.substr( 0, k ) ) ); + { + auto m = s5_r2::prefilled(); + REQUIRE( m.load_binary( s5_r2::temp_file( "full.bin", valid ) ) ); + REQUIRE( m[0][0] == 5.0 ); REQUIRE( m[1][1] == 8.0 ); + } + REQUIRE( s5_r2::bin_rejected( "trail.bin", valid + "x" ) ); + REQUIRE( s5_r2::bin_rejected( "rc_overflow.bin", s5_r2::binary_bytes( 1ull << 33, 1ull << 31, {} ) ) ); + REQUIRE( s5_r2::bin_rejected( "byte_overflow.bin", s5_r2::binary_bytes( 1ull << 61, 1, {} ) ) ); + REQUIRE( s5_r2::bin_rejected( "u8_overflow.bin", s5_r2::binary_bytes( ~0ull, 2, {} ) ) ); + + REQUIRE( s5_r2::bin_round_trips( "rtb_i8.bin" ) ); + REQUIRE( s5_r2::bin_round_trips( "rtb_i16.bin" ) ); + REQUIRE( s5_r2::bin_round_trips( "rtb_i32.bin" ) ); + REQUIRE( s5_r2::bin_round_trips( "rtb_i64.bin" ) ); + REQUIRE( s5_r2::bin_round_trips( "rtb_u8.bin" ) ); + REQUIRE( s5_r2::bin_round_trips( "rtb_u16.bin" ) ); + REQUIRE( s5_r2::bin_round_trips( "rtb_u32.bin" ) ); + REQUIRE( s5_r2::bin_round_trips( "rtb_u64.bin" ) ); + REQUIRE( s5_r2::bin_round_trips( "rtb_f.bin" ) ); + REQUIRE( s5_r2::bin_round_trips( "rtb_d.bin" ) ); + REQUIRE( s5_r2::bin_round_trips( "rtb_ld.bin" ) ); + REQUIRE( s5_r2::bin_round_trips>( "rtb_cf.bin" ) ); + REQUIRE( s5_r2::bin_round_trips>( "rtb_cd.bin" ) ); + REQUIRE( s5_r2::bin_round_trips>( "rtb_cld.bin" ) ); + + for ( auto const& [r, c] : { std::pair{ 0, 5 }, { 5, 0 }, { 0, 0 } } ) + { + feng::matrix> a{ r, c }; + auto const path = s5_r2::temp_path( "rtb_empty.bin" ); + REQUIRE( a.save_as_binary( path ) ); + auto b = s5_r2::prefilled>(); + REQUIRE( b.load_binary( path ) ); + REQUIRE( b.row() == r ); REQUIRE( b.col() == c ); + } +} diff --git a/tests/cases/s5_r3.hpp b/tests/cases/s5_r3.hpp new file mode 100644 index 0000000..6a9c50f --- /dev/null +++ b/tests/cases/s5_r3.hpp @@ -0,0 +1,284 @@ +// S5-R3 (PR-7, F08): the BMP loader validates the headers with checked arithmetic; S5-R4 (PR-2, D-011): it fails +// with an empty optional and one stderr line instead of aborting. BMP bytes are built here; temporary files go under +// std::filesystem::temp_directory_path(). +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include // getpid: per-process temp names + +namespace s5_r3 +{ + // The process id keeps two suite binaries run by hand at once from sharing a temp path. + inline std::string temp_path( std::string const& name ) + { + return ( std::filesystem::temp_directory_path() / ( "feng_s5_r3_" + std::to_string( ::getpid() ) + "_" + name ) ).string(); + } + + inline std::string temp_file( std::string const& name, std::vector const& bytes ) + { + auto const path = temp_path( name ); + std::ofstream ofs( path, std::ios::binary | std::ios::trunc ); + ofs.write( reinterpret_cast( bytes.data() ), static_cast( bytes.size() ) ); + ofs.close(); + return path; + } + + inline void put16( std::vector& b, std::size_t at, std::uint32_t v ) + { + b[at] = static_cast( v & 0xff ); + b[at + 1] = static_cast( ( v >> 8 ) & 0xff ); + } + + inline void put32( std::vector& b, std::size_t at, std::uint32_t v ) + { + put16( b, at, v & 0xffff ); + put16( b, at + 2, v >> 16 ); + } + + struct header + { + std::int32_t width = 3; + std::int32_t height = 2; + std::uint32_t info_size = 40; + std::uint32_t bpp = 24; + std::uint32_t compression = 0; + std::uint32_t planes = 1; + std::uint32_t extra_offset = 0; // gap between the info header and the pixels + std::size_t trailing = 0; + }; + + // The colour of logical (top-down) pixel (r, c): {red, green, blue}. + inline std::array colour( std::size_t r, std::size_t c ) + { + return { static_cast( 10 * r + c ), static_cast( 100 + 10 * r + c ), static_cast( 200 + 10 * r + c ) }; + } + + // A BMP whose logical image is colour(r, c); a positive height stores rows bottom-up, a negative one top-down. + inline std::vector bmp( header const& h ) + { + std::uint32_t const offset = 14 + h.info_size + h.extra_offset; + std::size_t const rows = static_cast( h.height < 0 ? -static_cast( h.height ) : h.height ); + std::size_t const cols = static_cast( h.width ); + std::size_t const bytes_per_pixel = h.bpp / 8; + std::size_t const stride = ( cols * h.bpp + 31 ) / 32 * 4; + std::vector b( offset + stride * rows + h.trailing, 0xEE ); + std::fill( b.begin(), b.begin() + offset, std::uint8_t{ 0 } ); + b[0] = 'B'; b[1] = 'M'; + put32( b, 2, static_cast( b.size() ) ); + put32( b, 10, offset ); + put32( b, 14, h.info_size ); + put32( b, 18, static_cast( h.width ) ); + put32( b, 22, static_cast( h.height ) ); + put16( b, 26, h.planes ); + put16( b, 28, h.bpp ); + put32( b, 30, h.compression ); + for ( std::size_t k = 0; bytes_per_pixel >= 3 && k != rows; ++k ) + { + std::size_t const r = h.height > 0 ? rows - 1 - k : k; // the logical row stored k-th + for ( std::size_t c = 0; c != cols; ++c ) + { + auto const [red, green, blue] = colour( r, c ); + std::size_t const at = offset + k * stride + c * bytes_per_pixel; + b[at] = blue; b[at + 1] = green; b[at + 2] = red; + if ( bytes_per_pixel == 4 ) b[at + 3] = 0x7F; + } + } + return b; + } + + inline bool matches( std::optional, 3>> const& img, std::size_t rows, std::size_t cols ) + { + if ( !img ) return false; + for ( auto const& m : *img ) + if ( m.row() != rows || m.col() != cols ) return false; + for ( std::size_t r = 0; r != rows; ++r ) + for ( std::size_t c = 0; c != cols; ++c ) + { + auto const [red, green, blue] = colour( r, c ); + if ( ( *img )[0][r][c] != red || ( *img )[1][r][c] != green || ( *img )[2][r][c] != blue ) return false; + } + return true; + } + + inline bool rejected( std::string const& name, std::vector const& bytes ) + { + std::ostringstream captured; + auto* const old = std::cerr.rdbuf( captured.rdbuf() ); + auto const img = feng::load_bmp( temp_file( name, bytes ) ); + std::cerr.rdbuf( old ); + std::string const text = captured.str(); + return !img && text.find( "load_bmp" ) != std::string::npos && std::count( text.begin(), text.end(), '\n' ) == 1; + } +} + +TEST_CASE( "S5 load_bmp rejects width 2147483647 and height -2147483648", "[S5][S5-R3][S5-R4]" ) +{ + auto b = s5_r3::bmp( {} ); + s5_r3::put32( b, 18, 2147483647u ); + s5_r3::put32( b, 22, 0x80000000u ); + REQUIRE( s5_r3::rejected( "huge.bmp", b ) ); + s5_r3::put32( b, 22, 0x80000001u ); // -2147483647: the pixel array would not fit in the file + REQUIRE( s5_r3::rejected( "huge2.bmp", b ) ); + s5_r3::put32( b, 22, 2147483647u ); + REQUIRE( s5_r3::rejected( "huge3.bmp", b ) ); +} + +TEST_CASE( "S5 bottom-up and top-down files load to the same channels", "[S5][S5-R3]" ) +{ + // 24- and 32-bit (alpha ignored), bottom-up and top-down, widths whose rows need 0 to 3 padding bytes + for ( std::uint32_t bpp : { 24u, 32u } ) + for ( std::int32_t width : { 1, 2, 3, 4, 5 } ) + for ( std::int32_t height : { 3, -3, 1, -1 } ) + { + INFO( "bpp " << bpp << ", width " << width << ", height " << height ); + s5_r3::header h; + h.bpp = bpp; h.width = width; h.height = height; + auto const rows = static_cast( height < 0 ? -height : height ); + REQUIRE( s5_r3::matches( feng::load_bmp( s5_r3::temp_file( "orient.bmp", s5_r3::bmp( h ) ) ), rows, static_cast( width ) ) ); + } + + // every supported info header size, alone, with a gap before the pixels and with trailing bytes + for ( std::uint32_t info : { 40u, 52u, 56u, 108u, 124u } ) + for ( std::uint32_t bpp : { 24u, 32u } ) + for ( std::uint32_t gap : { 0u, 6u } ) + for ( std::size_t trailing : { std::size_t{ 0 }, std::size_t{ 3 } } ) + { + INFO( "info " << info << ", bpp " << bpp << ", gap " << gap << ", trailing " << trailing ); + s5_r3::header g; + g.width = 5; g.height = info % 2 ? 3 : -3; g.info_size = info; g.bpp = bpp; + g.extra_offset = gap; g.trailing = trailing; + REQUIRE( s5_r3::matches( feng::load_bmp( s5_r3::temp_file( "var.bmp", s5_r3::bmp( g ) ) ), 3, 5 ) ); + } +} + +TEST_CASE( "S5 load_bmp rejects truncations and unsupported headers", "[S5][S5-R3][S5-R4]" ) +{ + auto const good = s5_r3::bmp( {} ); + s5_r3::header h32; h32.bpp = 32; h32.info_size = 124; + for ( auto const& valid : { good, s5_r3::bmp( h32 ) } ) + for ( std::size_t n = 0; n != valid.size(); ++n ) + { + INFO( "truncated to " << n << " of " << valid.size() << " bytes" ); + REQUIRE( s5_r3::rejected( "trunc.bmp", std::vector( valid.begin(), valid.begin() + static_cast( n ) ) ) ); + } + + auto sig = good; sig[1] = 'A'; + REQUIRE( s5_r3::rejected( "sig.bmp", sig ) ); + for ( std::uint32_t info : { 0u, 12u, 39u, 41u, 64u, 125u, 100000u } ) + { + auto b = good; s5_r3::put32( b, 14, info ); + REQUIRE( s5_r3::rejected( "info.bmp", b ) ); + } + for ( std::uint32_t comp : { 1u, 2u, 3u, 6u } ) + { + s5_r3::header h; h.compression = comp; + REQUIRE( s5_r3::rejected( "comp.bmp", s5_r3::bmp( h ) ) ); + } + for ( std::uint32_t bpp : { 1u, 4u, 8u, 16u } ) + { + s5_r3::header h; h.bpp = bpp; + auto b = s5_r3::bmp( h ); + b.resize( b.size() + 64, 0 ); + REQUIRE( s5_r3::rejected( "bpp.bmp", b ) ); + } + { s5_r3::header h; h.planes = 2; REQUIRE( s5_r3::rejected( "planes.bmp", s5_r3::bmp( h ) ) ); } + { auto b = good; s5_r3::put32( b, 18, 0 ); REQUIRE( s5_r3::rejected( "w0.bmp", b ) ); } + { auto b = good; s5_r3::put32( b, 18, 0xFFFFFFFFu ); REQUIRE( s5_r3::rejected( "wneg.bmp", b ) ); } + { auto b = good; s5_r3::put32( b, 18, 0x80000000u ); REQUIRE( s5_r3::rejected( "wmin.bmp", b ) ); } + { auto b = good; s5_r3::put32( b, 22, 0 ); REQUIRE( s5_r3::rejected( "h0.bmp", b ) ); } + { auto b = good; s5_r3::put32( b, 22, 0x80000000u ); REQUIRE( s5_r3::rejected( "hmin.bmp", b ) ); } + // the pixel-data offset one byte before the end of each info header + for ( std::uint32_t info : { 40u, 124u } ) + { + s5_r3::header h; h.info_size = info; + auto b = s5_r3::bmp( h ); + s5_r3::put32( b, 10, 14 + info - 1 ); + REQUIRE( s5_r3::rejected( "offset_in_info.bmp", b ) ); + } + { auto b = good; s5_r3::put32( b, 10, 50 ); REQUIRE( s5_r3::rejected( "offset_low.bmp", b ) ); } + { auto b = good; s5_r3::put32( b, 10, 0xFFFFFFF0u ); REQUIRE( s5_r3::rejected( "offset_high.bmp", b ) ); } + + // width x height whose stride or pixel-array arithmetic overflows, or merely exceeds a 54-byte image: parse_bmp + // must reject them from the header alone, before allocating any channel + for ( auto const [w, hgt] : std::vector>{ + { 0x7FFFFFFFu, 0x80000001u }, { 0x7FFFFFFFu, 0x7FFFFFFFu }, { 65536u, 65536u }, { 0x7FFFFFFFu, 1u }, { 1u, 0x7FFFFFFFu } } ) + for ( std::uint32_t bpp : { 24u, 32u } ) + { + INFO( "width " << w << ", height " << hgt << ", bpp " << bpp ); + auto b = good; + b.resize( 54 ); + s5_r3::put32( b, 18, w ); s5_r3::put32( b, 22, hgt ); s5_r3::put16( b, 28, bpp ); + REQUIRE( !feng::matrix_details::parse_bmp( b.data(), b.size() ) ); + REQUIRE( s5_r3::rejected( "overflow.bmp", b ) ); + } + + auto const missing = s5_r3::temp_path( "does_not_exist.bmp" ); + std::filesystem::remove( missing ); + REQUIRE( !feng::load_bmp( missing ) ); + REQUIRE( !feng::load_bmp( std::filesystem::temp_directory_path().string() ) ); +} + +TEST_CASE( "S5 a save_as_bmp file loads back to its channels", "[S5][S5-R3]" ) +{ + auto const& map = feng::matrix_details::bmp_details::color_maps.at( "parula" ); + for ( unsigned cols : { 1u, 2u, 3u, 5u } ) + { + INFO( "member save_as_bmp, 3 x " << cols ); + feng::matrix m{ 3, cols }; + for ( std::size_t k = 0; k != m.size(); ++k ) m.data()[k] = static_cast( ( k * 7 ) % 11 ); + auto const path = s5_r3::temp_path( "saved.bmp" ); + REQUIRE( m.save_as_bmp( path, "parula" ) ); + auto const img = feng::load_bmp( path ); + REQUIRE( img ); + auto const [mn, mx] = m.minmax(); + for ( auto const& channel : *img ) { REQUIRE( channel.row() == 3 ); REQUIRE( channel.col() == cols ); } + for ( std::size_t r = 0; r != 3; ++r ) + for ( std::size_t c = 0; c != cols; ++c ) + { + auto const [red, green, blue] = map( ( m[r][c] - mn ) / ( mx - mn + 1.0e-10 ) ); + REQUIRE( ( *img )[0][r][c] == red ); + REQUIRE( ( *img )[1][r][c] == green ); + REQUIRE( ( *img )[2][r][c] == blue ); + } + std::filesystem::remove( path ); + } + + // channels with minimum 0 and maximum 255 are written unscaled, so load_bmp returns them exactly + for ( unsigned cols : { 1u, 2u, 3u, 5u } ) + { + INFO( "free save_as_bmp( name, r, g, b ), 3 x " << cols ); + feng::matrix r{ 3, cols }, g{ 3, cols }, b{ 3, cols }; + for ( std::size_t k = 0; k != r.size(); ++k ) + { + r.data()[k] = static_cast( ( k * 37 ) % 256 ); + g.data()[k] = static_cast( ( k * 91 + 5 ) % 256 ); + b.data()[k] = static_cast( ( k * 53 + 11 ) % 256 ); + } + for ( auto* m : { &r, &g, &b } ) + { + ( *m )[0][0] = 0.0; + ( *m )[2][cols - 1] = 255.0; + } + auto const path = s5_r3::temp_path( "saved_rgb.bmp" ); + REQUIRE( feng::save_as_bmp( path, r, g, b ) ); + auto const img = feng::load_bmp( path ); + REQUIRE( img ); + for ( auto const& channel : *img ) { REQUIRE( channel.row() == 3 ); REQUIRE( channel.col() == cols ); } + for ( std::size_t y = 0; y != 3; ++y ) + for ( std::size_t x = 0; x != cols; ++x ) + { + INFO( "pixel " << y << ", " << x ); + REQUIRE( ( *img )[0][y][x] == static_cast( r[y][x] ) ); + REQUIRE( ( *img )[1][y][x] == static_cast( g[y][x] ) ); + REQUIRE( ( *img )[2][y][x] == static_cast( b[y][x] ) ); + } + std::filesystem::remove( path ); + } +} diff --git a/tests/cases/s5_r4.hpp b/tests/cases/s5_r4.hpp new file mode 100644 index 0000000..9b5f624 --- /dev/null +++ b/tests/cases/s5_r4.hpp @@ -0,0 +1,365 @@ +// S5-R4 (PR-2, PR-7, D-011): every file writer returns false with one stderr line, and never aborts, on a directory, +// open, write or close failure, and true on success; D-022: save_as_npy writes a v1.0 header numpy reads and that +// load_npy loads back. S5-R3: the free three-channel save_as_bmp loads back to its channels. Temporary files go under +// std::filesystem::temp_directory_path(). +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include // getpid: per-process temp names + +namespace s5_r4 +{ + // The process id keeps two suite binaries run by hand at once from sharing a temp path. + inline std::string temp_path( std::string const& name ) + { + return ( std::filesystem::temp_directory_path() / ( "feng_s5_r4_" + std::to_string( ::getpid() ) + "_" + name ) ).string(); + } + + inline std::vector file_bytes( std::string const& path ) + { + std::ifstream ifs( path, std::ios::binary ); + return std::vector{ std::istreambuf_iterator( ifs ), std::istreambuf_iterator() }; + } + + // Redirects std::cerr into a string for the lifetime of the object. + struct capture_cerr + { + std::ostringstream text; + std::streambuf* old; + capture_cerr() : old( std::cerr.rdbuf( text.rdbuf() ) ) {} + ~capture_cerr() { std::cerr.rdbuf( old ); } + std::string str() const { return text.str(); } + }; + + struct writer + { + char const* name; // what the stderr line must name + char const* extension; // the writer's file extension + std::function call; + }; + + inline feng::matrix const& sample() + { + static feng::matrix const m = [] + { + feng::matrix x{ 3, 4 }; + for ( std::size_t k = 0; k != x.size(); ++k ) x.data()[k] = static_cast( ( k * 5 ) % 7 ) - 1.5; + return x; + }(); + return m; + } + + // Every writer overload: member txt, binary, bmp, png, pgm, npy (string and char const*), and the free bmp ones. + inline std::vector writers() + { + auto const& m = sample(); + return { + { "save_as_txt", ".txt", [&m]( std::string const& p ) { return m.save_as_txt( p ); } }, + { "save_as_txt", ".txt", [&m]( std::string const& p ) { return m.save_as_txt( p.c_str() ); } }, + { "save_as_binary", ".bin", [&m]( std::string const& p ) { return m.save_as_binary( p ); } }, + { "save_as_binary", ".bin", [&m]( std::string const& p ) { return m.save_as_binary( p.c_str() ); } }, + { "save_as_bmp", ".bmp", [&m]( std::string const& p ) { return m.save_as_bmp( p ); } }, + { "save_as_bmp", ".bmp", [&m]( std::string const& p ) { return m.save_as_bmp( p.c_str() ); } }, + { "save_as_png", ".png", [&m]( std::string const& p ) { return m.save_as_png( p ); } }, + { "save_as_pgm", ".pgm", [&m]( std::string const& p ) { return m.save_as_pgm( p ); } }, + { "save_as_pgm", ".pgm", [&m]( std::string const& p ) { return m.save_as_pgm( p.c_str() ); } }, + { "save_as_npy", ".npy", [&m]( std::string const& p ) { return m.save_as_npy( p ); } }, + { "save_as_npy", ".npy", [&m]( std::string const& p ) { return m.save_as_npy( p.c_str() ); } }, + { "save_as_bmp", ".bmp", [&m]( std::string const& p ) { return feng::save_as_bmp( p, m, m, m ); } }, + { "save_as_bmp", ".bmp", [&m]( std::string const& p ) { return feng::save_as_bmp( p, m, std::string{ "jet" } ); } }, + }; + } + + inline void require_failure( writer const& w, std::string const& path ) + { + INFO( w.name << " into " << path ); + capture_cerr cap; + bool const ok = w.call( path ); + std::string const text = cap.str(); + REQUIRE( !ok ); + REQUIRE( text.find( w.name ) != std::string::npos ); + REQUIRE( text.find( path ) != std::string::npos ); + REQUIRE( text.find( '\n' ) == text.size() - 1 ); // one line + } +} // namespace s5_r4 + +TEST_CASE( "S5 save_as_png into a missing directory returns false", "[S5][S5-R4]" ) +{ + s5_r4::capture_cerr cap; + bool const ok = s5_r4::sample().save_as_png( "/nonexistent-dir/x.png" ); + std::string const text = cap.str(); + REQUIRE( !ok ); + REQUIRE( text.find( "save_as_png" ) != std::string::npos ); + REQUIRE( text.find( "/nonexistent-dir/x.png" ) != std::string::npos ); +} + +TEST_CASE( "S5 every writer reports a write failure on /dev/full", "[S5][S5-R4]" ) +{ + // write or close failure: a symlink to /dev/full opens but every write fails with ENOSPC + // the write-failure path must be exercised: a host without /dev/full fails this test rather than skipping it + REQUIRE( std::filesystem::exists( "/dev/full" ) ); + for ( auto const& w : s5_r4::writers() ) + { + auto const link = s5_r4::temp_path( std::string{ "full" } + w.extension ); + std::filesystem::remove( link ); + std::filesystem::create_symlink( "/dev/full", link ); + s5_r4::require_failure( w, link ); + std::filesystem::remove( link ); + } + + // open and directory failures: the parent path is a regular file + auto const plain = s5_r4::temp_path( "plain_file" ); + std::filesystem::remove_all( plain ); + { + std::ofstream ofs( plain ); + ofs << "not a directory\n"; + } + for ( auto const& w : s5_r4::writers() ) + { + s5_r4::require_failure( w, plain + "/x" + w.extension ); // open failure + s5_r4::require_failure( w, plain + "/sub/x" + w.extension ); // directory failure + } + std::filesystem::remove( plain ); + + // directory failure: the parent directory does not exist and cannot be created + for ( auto const& w : s5_r4::writers() ) + s5_r4::require_failure( w, std::string{ "/nonexistent-dir/x" } + w.extension ); +} + +TEST_CASE( "S5 every writer returns true on success", "[S5][S5-R4]" ) +{ + int k = 0; + for ( auto const& w : s5_r4::writers() ) + { + auto const path = s5_r4::temp_path( "ok_" + std::to_string( k++ ) + w.extension ); + std::filesystem::remove( path ); + INFO( w.name << " into " << path ); + REQUIRE( w.call( path ) ); + REQUIRE( std::filesystem::file_size( path ) > 0 ); + std::filesystem::remove( path ); + } +} + +namespace s5_r4 +{ + template < typename T > + void npy_round_trip( feng::matrix const& a, std::string const& name ) + { + INFO( name ); + auto const path = temp_path( name ); + REQUIRE( a.save_as_npy( path ) ); + auto const bytes = file_bytes( path ); + REQUIRE( bytes.size() >= 10 ); + std::size_t const header_len = static_cast( bytes[8] ) | ( static_cast( static_cast( bytes[9] ) ) << 8 ); + REQUIRE( ( 10 + header_len ) % 64 == 0 ); + REQUIRE( bytes[9 + header_len] == '\n' ); + REQUIRE( bytes.size() == 10 + header_len + sizeof( T ) * a.size() ); + feng::matrix b{ 1, 1 }; + REQUIRE( b.load_npy( path ) ); + REQUIRE( b.row() == a.row() ); + REQUIRE( b.col() == a.col() ); + for ( std::size_t k = 0; k != a.size(); ++k ) REQUIRE( b.data()[k] == a.data()[k] ); + std::filesystem::remove( path ); + } +} // namespace s5_r4 + +TEST_CASE( "S5 save_as_npy output loads back through load_npy", "[S5][S5-R4]" ) +{ + feng::matrix d{ 2, 3 }; + feng::matrix f{ 3, 2 }; + feng::matrix i{ 2, 4 }; + feng::matrix> c{ 2, 2 }; + for ( std::size_t k = 0; k != d.size(); ++k ) d.data()[k] = 1.0 / ( 1.0 + static_cast( k ) ) - 0.3; + for ( std::size_t k = 0; k != f.size(); ++k ) f.data()[k] = static_cast( k ) * -1.25f + 0.1f; + for ( std::size_t k = 0; k != i.size(); ++k ) i.data()[k] = static_cast( k * 70001 ) - 2147483647; + for ( std::size_t k = 0; k != c.size(); ++k ) c.data()[k] = std::complex{ 0.5 * static_cast( k ), -1.0 / ( 1.0 + static_cast( k ) ) }; + s5_r4::npy_round_trip( d, "rt_f8.npy" ); + s5_r4::npy_round_trip( f, "rt_f4.npy" ); + s5_r4::npy_round_trip( i, "rt_i4.npy" ); + s5_r4::npy_round_trip( c, "rt_c16.npy" ); + feng::matrix empty{ 0, 3 }; + s5_r4::npy_round_trip( empty, "rt_empty.npy" ); +} + +TEST_CASE( "S5 save_as_npy writes the bytes numpy writes", "[S5][S5-R4]" ) +{ + feng::matrix base{ 2, 3 }; + double const values[] = { 1.5, -2.25, 3.0, 4.125, 5.0, -6.5 }; + std::copy( std::begin( values ), std::end( values ), base.data() ); + auto const path = s5_r4::temp_path( "numpy_f8.npy" ); + REQUIRE( base.save_as_npy( path ) ); + auto const ours = s5_r4::file_bytes( path ); + auto const numpys = s5_r4::file_bytes( "./tests/fixtures/s5/f8_le.npy" ); + REQUIRE( numpys.size() == 128 + 6 * 8 ); + REQUIRE( ours.size() == numpys.size() ); + std::size_t const compare = std::endian::native == std::endian::little ? numpys.size() : 128; // payload is native order + for ( std::size_t k = 0; k != compare; ++k ) + { + if ( k == 18 && std::endian::native != std::endian::little ) continue; // the '<' of the descr + INFO( "byte " << k ); + REQUIRE( ours[k] == numpys[k] ); + } + std::filesystem::remove( path ); + + feng::matrix u{ 1, 2, std::uint8_t{ 7 } }; + REQUIRE( u.save_as_npy( path ) ); + auto const ub = s5_r4::file_bytes( path ); + std::string const header( ub.begin() + 10, ub.begin() + 10 + 60 ); + REQUIRE( header.find( "{'descr': '|u1', 'fortran_order': False, 'shape': (1, 2), }" ) == 0 ); + std::filesystem::remove( path ); +} + +TEST_CASE( "S5 a free three-channel save_as_bmp file loads back to its channels", "[S5][S5-R3]" ) +{ + // channels with minimum 0 and maximum 255 are written unscaled, so load_bmp must return them exactly + feng::matrix r{ 3, 4 }, g{ 3, 4 }, b{ 3, 4 }; + for ( std::size_t k = 0; k != r.size(); ++k ) + { + r.data()[k] = static_cast( k * 23 ); + g.data()[k] = static_cast( ( 11 - k ) * 19 ); + b.data()[k] = static_cast( ( k * 7 ) % 12 * 21 ); + } + for ( auto* m : { &r, &g, &b } ) + { + ( *m )[0][1] = 0.0; + ( *m )[2][2] = 255.0; + } + auto const path = s5_r4::temp_path( "rgb.bmp" ); + REQUIRE( feng::save_as_bmp( path, r, g, b ) ); + auto const img = feng::load_bmp( path ); + REQUIRE( img ); + for ( auto const& channel : *img ) { REQUIRE( channel.row() == 3 ); REQUIRE( channel.col() == 4 ); } + for ( std::size_t y = 0; y != 3; ++y ) + for ( std::size_t x = 0; x != 4; ++x ) + { + INFO( "pixel " << y << ", " << x ); + REQUIRE( ( *img )[0][y][x] == static_cast( r[y][x] ) ); + REQUIRE( ( *img )[1][y][x] == static_cast( g[y][x] ) ); + REQUIRE( ( *img )[2][y][x] == static_cast( b[y][x] ) ); + } + std::filesystem::remove( path ); +} + +namespace s5_r4 +{ + inline std::string bytes_file( std::string const& name, std::string const& bytes ) + { + auto const path = temp_path( name ); + std::ofstream ofs( path, std::ios::binary | std::ios::trunc ); + ofs.write( bytes.data(), static_cast( bytes.size() ) ); + return path; + } + + inline feng::matrix prefilled() + { + feng::matrix m{ 2, 2 }; + m[0][0] = 1.0; m[0][1] = 2.0; m[1][0] = 3.0; m[1][1] = 4.0; + return m; + } + + inline bool is_prefilled( feng::matrix const& m ) + { + return m.row() == 2 && m.col() == 2 && m[0][0] == 1.0 && m[0][1] == 2.0 && m[1][0] == 3.0 && m[1][1] == 4.0; + } + + // One stderr line naming the loader. + inline bool one_line_naming( std::string const& text, char const* loader ) + { + return text.find( loader ) != std::string::npos && !text.empty() && text.find( '\n' ) == text.size() - 1; + } +} // namespace s5_r4 + +TEST_CASE( "S5 every loader fails on every failure class and keeps a pre-filled destination", "[S5][S5-R4]" ) +{ + using size_type = feng::matrix::size_type; + std::string short_binary( 2 * sizeof( size_type ) + sizeof( double ), '\0' ); // declares 2 x 3, holds one element + size_type const two = 2, three = 3; + std::memcpy( short_binary.data(), &two, sizeof( two ) ); + std::memcpy( short_binary.data() + sizeof( two ), &three, sizeof( three ) ); + std::string npy_f4 = std::string( "\x93NUMPY\x01\x00", 8 ) + '\x76' + '\x00' + "{'descr': ' + { + // the same bytes are a well-formed ', so the double rejection is the dtype + feng::matrix f{ 2, 2 }; + REQUIRE( npy_f4.size() == 128 + 4 ); + REQUIRE( f.load_npy( s5_r4::bytes_file( "f4_ok.npy", npy_f4 ) ) ); + REQUIRE( f.row() == 1 ); + REQUIRE( f.col() == 1 ); + REQUIRE( f[0][0] == 0.0f ); + } + std::string bmp_rle( 58, '\0' ); // a 1 x 1 BMP with compression 1 (RLE8) + bmp_rle[0] = 'B'; bmp_rle[1] = 'M'; bmp_rle[10] = 54; bmp_rle[14] = 40; bmp_rle[18] = 1; bmp_rle[22] = 1; + bmp_rle[26] = 1; bmp_rle[28] = 24; bmp_rle[30] = 1; + + auto const missing = s5_r4::temp_path( "missing_input" ); + std::filesystem::remove_all( missing ); + auto const directory = std::filesystem::temp_directory_path().string(); + struct failure { char const* what; std::string txt, bin, npy, bmp; }; + std::vector const failures{ + { "missing file", missing, missing, missing, missing }, + { "directory", directory, directory, directory, directory }, + { "empty file", s5_r4::bytes_file( "empty.txt", "" ), s5_r4::bytes_file( "empty.bin", "" ), s5_r4::bytes_file( "empty.npy", "" ), s5_r4::bytes_file( "empty.bmp", "" ) }, + { "3-byte file", s5_r4::bytes_file( "three.txt", "abc" ), s5_r4::bytes_file( "three.bin", "abc" ), s5_r4::bytes_file( "three.npy", "abc" ), s5_r4::bytes_file( "three.bmp", "BMa" ) }, + { "rejected input", s5_r4::bytes_file( "ragged.txt", "1 2\n3\n" ), s5_r4::bytes_file( "short.bin", short_binary ), s5_r4::bytes_file( "f4.npy", npy_f4 ), + s5_r4::bytes_file( "rle.bmp", bmp_rle ) }, + }; + + for ( auto const& f : failures ) + { + INFO( f.what ); + { + auto m = s5_r4::prefilled(); + s5_r4::capture_cerr cap; + bool const ok = m.load_txt( f.txt ); + std::string const text = cap.str(); + REQUIRE( !ok ); + REQUIRE( s5_r4::is_prefilled( m ) ); + REQUIRE( s5_r4::one_line_naming( text, "load_txt" ) ); + } + { + auto m = s5_r4::prefilled(); + s5_r4::capture_cerr cap; + bool const ok = m.load_binary( f.bin ); + std::string const text = cap.str(); + REQUIRE( !ok ); + REQUIRE( s5_r4::is_prefilled( m ) ); + REQUIRE( s5_r4::one_line_naming( text, "load_binary" ) ); + } + { + auto m = s5_r4::prefilled(); + s5_r4::capture_cerr cap; + bool const ok = m.load_npy( f.npy ); + std::string const text = cap.str(); + REQUIRE( !ok ); + REQUIRE( s5_r4::is_prefilled( m ) ); + REQUIRE( s5_r4::one_line_naming( text, "load_npy" ) ); + } + { + // operator>> reports through the stream state (failbit), not stderr + auto m = s5_r4::prefilled(); + std::ifstream ifs( f.txt ); + ifs >> m; + REQUIRE( ifs.fail() ); + REQUIRE( s5_r4::is_prefilled( m ) ); + } + { + s5_r4::capture_cerr cap; + auto const img = feng::load_bmp( f.bmp ); + std::string const text = cap.str(); + REQUIRE( !img ); + REQUIRE( s5_r4::one_line_naming( text, "load_bmp" ) ); + } + } +} diff --git a/tests/cases/s5_r5.hpp b/tests/cases/s5_r5.hpp new file mode 100644 index 0000000..0f4dbf2 --- /dev/null +++ b/tests/cases/s5_r5.hpp @@ -0,0 +1,59 @@ +// S5-R5 (PR-7): replays the fuzz regressions (tests/fuzz/regressions//, if any) and the seed corpora +// (tests/fuzz/corpus//) through the parsers with the harness checks of tests/fuzz/fuzz_checks.hpp. +// Runs from the repo root. +#include "../fuzz/fuzz_checks.hpp" + +#include +#include +#include +#include + +namespace s5_r5 +{ + using check_fn = char const* (*)( std::uint8_t const*, std::size_t ); + + struct target + { + char const* name; + check_fn check; + }; + + inline std::vector< std::filesystem::path > files_under( std::filesystem::path const& dir ) + { + std::vector< std::filesystem::path > ans; + std::error_code ec; + if ( !std::filesystem::is_directory( dir, ec ) ) return ans; + for ( auto const& e : std::filesystem::directory_iterator( dir ) ) + if ( e.is_regular_file() ) ans.push_back( e.path() ); + return ans; + } +} + +TEST_CASE( "S5 fuzz regressions replay without failure", "[S5][S5-R5]" ) +{ + s5_r5::target const targets[] = { + { "npy", fuzz_checks::check_npy }, + { "bmp", fuzz_checks::check_bmp }, + { "binary", fuzz_checks::check_binary }, + { "text", fuzz_checks::check_text }, + }; + for ( auto const& t : targets ) + { + std::size_t seeds = 0; + for ( char const* root : { "./tests/fuzz/regressions/", "./tests/fuzz/corpus/" } ) + { + auto const files = s5_r5::files_under( std::string{ root } + t.name ); + if ( std::string{ root }.find( "corpus" ) != std::string::npos ) seeds = files.size(); + for ( auto const& f : files ) + { + std::vector< std::uint8_t > bytes; + REQUIRE( feng::matrix_details::read_file( f.string().c_str(), bytes ) ); + INFO( "input " << f.string() ); + char const* const failure = t.check( bytes.data(), bytes.size() ); + REQUIRE( failure == nullptr ); + } + } + INFO( "target " << t.name ); + REQUIRE( seeds > 0 ); + } +} diff --git a/tests/cases/s6_r1.hpp b/tests/cases/s6_r1.hpp new file mode 100644 index 0000000..38cce64 --- /dev/null +++ b/tests/cases/s6_r1.hpp @@ -0,0 +1,299 @@ +// S6-R1 (PR-8): one partition drives parallel and both reductions; every index once, init once, every worker +// joined (F11, F12). +#include +#include +#include +#include +#include +#include +#include +#include + +namespace s6_r1 +{ + // Visits [first, last) with `workers` and checks: every index exactly once (atomic counters), nothing outside + // the range touched, and every chunk's plain slot written before return (a missing join is a TSan race). + template< typename I > + void check_visits( I first, I last, std::size_t workers ) + { + std::size_t const n = first < last ? static_cast( last - first ) : 0; + std::size_t const pad = 4; + std::size_t const base = static_cast( first < last ? first : last ); + std::unique_ptr[]> visits{ new std::atomic[ base + n + pad ] }; + for ( std::size_t i = 0; i != base + n + pad; ++i ) visits[i].store( 0 ); + std::vector slots( n, 0 ); // plain, non-atomic; one slot per index, written by its worker + feng::matrix_details::parallel_workers( [&]( I i ) + { + visits[static_cast( i )].fetch_add( 1 ); + slots[static_cast( i - first )] = static_cast( i ) + 1; + }, first, last, workers ); + for ( std::size_t i = 0; i != base + n + pad; ++i ) + { + bool const inside = i >= static_cast( first ) && i < base + n && n != 0; + REQUIRE( visits[i].load() == ( inside ? 1 : 0 ) ); + } + for ( std::size_t j = 0; j != n; ++j ) + REQUIRE( slots[j] == static_cast( first ) + j + 1 ); + } + + template< typename I > + void check_chunks( I first, I last, std::size_t workers ) + { + std::size_t const n = first < last ? static_cast( last - first ) : 0; + std::size_t const w = feng::matrix_details::effective_workers( first, last, workers ); + if ( n == 0 ) { REQUIRE( w == 0 ); return; } + REQUIRE( w == std::clamp( workers, std::size_t{1}, n ) ); + I expected = first; + for ( std::size_t k = 0; k != w; ++k ) + { + auto const [b, e] = feng::matrix_details::chunk_bounds( first, last, workers, k ); + REQUIRE( b == expected ); // contiguous, in order, disjoint + REQUIRE( b < e ); // non-empty + std::size_t const len = static_cast( e - b ); + REQUIRE( len == n / w + ( k < n % w ? 1 : 0 ) ); + expected = e; + } + REQUIRE( expected == last ); // covers [first, last) + } + + // Each chunk sets its own plain slot; the caller reads them after return with no other synchronisation. + inline void check_chunk_slots( std::size_t first, std::size_t last, std::size_t workers ) + { + std::size_t const w = feng::matrix_details::effective_workers( first, last, workers ); + std::vector slot( w, 0 ); + feng::matrix_details::parallel_workers( [&]( std::size_t k ) + { + auto const [b, e] = feng::matrix_details::chunk_bounds( first, last, workers, k ); + slot[k] = e - b; + }, std::size_t{0}, w, w ); + std::size_t total = 0; + for ( auto s : slot ) { REQUIRE( s > 0 ); total += s; } + REQUIRE( total == last - first ); + } + + inline std::vector worker_counts( std::size_t n ) + { + return { 0, 1, 2, 7, 32, n + 1, n + 100 }; + } +} + +TEST_CASE( "S6-R1 parallel visits every index once for injected worker counts", "[S6][S6-R1]" ) +{ + SECTION( "[0, n) for n in 1, 10, 100, 1000" ) + { + for ( std::size_t n : { 1UL, 10UL, 100UL, 1000UL } ) + for ( std::size_t w : s6_r1::worker_counts( n ) ) + { + CAPTURE( n ); CAPTURE( w ); + s6_r1::check_visits( std::size_t{0}, n, w ); + s6_r1::check_chunks( std::size_t{0}, n, w ); + s6_r1::check_chunk_slots( 0, n, w ); + } + } + SECTION( "[5, 17) with 32 workers and other counts" ) + { + s6_r1::check_visits( 5, 17, 32 ); + s6_r1::check_chunks( 5, 17, 32 ); + s6_r1::check_chunk_slots( 5, 17, 32 ); + for ( std::size_t w : s6_r1::worker_counts( 12 ) ) + { + CAPTURE( w ); + s6_r1::check_visits( 5UL, 17UL, w ); + s6_r1::check_chunks( 5UL, 17UL, w ); + s6_r1::check_chunk_slots( 5, 17, w ); + } + s6_r1::check_visits( 1000, 1777, 7 ); + s6_r1::check_chunks( 1000, 1777, 7 ); + } + SECTION( "empty and reversed ranges call nothing" ) + { + for ( std::size_t w : { 0UL, 1UL, 2UL, 7UL, 32UL } ) + { + int calls = 0; + feng::matrix_details::parallel_workers( [&]( int ) { ++calls; }, 3, 3, w ); + feng::matrix_details::parallel_workers( [&]( int ) { ++calls; }, 9, 3, w ); + feng::matrix_details::parallel_workers( [&]( std::size_t ) { ++calls; }, 0UL, 0UL, w ); + REQUIRE( calls == 0 ); + REQUIRE( feng::matrix_details::effective_workers( 3, 3, w ) == 0 ); + REQUIRE( feng::matrix_details::effective_workers( 9, 3, w ) == 0 ); + s6_r1::check_chunks( 3, 3, w ); + } + } + SECTION( "public parallel( func, first, last, threshold ) and parallel( func, last ) visit every index once" ) + { + for ( std::size_t n : { 0UL, 1UL, 33UL, 2000UL } ) + for ( unsigned long threshold : { 0UL, 1024UL } ) + { + CAPTURE( n ); CAPTURE( threshold ); + std::unique_ptr[]> visits{ new std::atomic[ n + 8 ] }; + for ( std::size_t i = 0; i != n + 8; ++i ) visits[i].store( 0 ); + feng::matrix_details::parallel( [&]( std::size_t i ) { visits[i].fetch_add( 1 ); }, std::size_t{3}, n + 3, threshold ); + for ( std::size_t i = 0; i != n + 8; ++i ) + REQUIRE( visits[i].load() == ( i >= 3 && i < n + 3 ? 1 : 0 ) ); + } + std::unique_ptr[]> visits{ new std::atomic[ 50 ] }; + for ( std::size_t i = 0; i != 50; ++i ) visits[i].store( 0 ); + feng::matrix_details::parallel( [&]( std::size_t i ) { visits[i].fetch_add( 1 ); }, std::size_t{50} ); + for ( std::size_t i = 0; i != 50; ++i ) + REQUIRE( visits[i].load() == 1 ); + } + SECTION( "default_workers is 1 for small or serial jobs and at least 1 otherwise" ) + { + REQUIRE( feng::matrix_details::default_workers( 10, 1024 ) == 1 ); + REQUIRE( feng::matrix_details::default_workers( 0, 0 ) == 1 ); + std::size_t const big = feng::matrix_details::default_workers( 1UL << 20, 32 ); +#ifdef FENG_MATRIX_PARALLEL + unsigned const hc = std::thread::hardware_concurrency(); + REQUIRE( big == ( hc == 0 ? 1 : hc ) ); +#else + REQUIRE( big == 1 ); +#endif + } +} + +TEST_CASE( "S6-R1 both reductions include init exactly once", "[S6][S6-R1]" ) +{ + auto const plus = []( auto a, auto b ) { return a + b; }; + + SECTION( "ten elements, init 100, workers 0: 100 + sum, for matrix and matrix" ) + { + feng::matrix md( 2, 5 ); + std::iota( md.begin(), md.end(), 1.0 ); + feng::matrix mi( 2, 5 ); + std::iota( mi.begin(), mi.end(), 1 ); + REQUIRE( feng::matrix_details::reduce( md.begin(), md.end(), 100.0, plus, 0 ) == 155.0 ); + REQUIRE( feng::matrix_details::reduce( mi.begin(), mi.end(), 100, plus, 0 ) == 155 ); + REQUIRE( feng::matrix_details::reduce_impl_private::reduce_impl( md )( plus, 100.0, 0 ) == 155.0 ); + REQUIRE( feng::matrix_details::reduce_impl_private::reduce_impl( mi )( plus, 100, 0 ) == 155 ); + } + SECTION( "every worker count, sizes 0, 1, 10, 1000: both reductions equal std::accumulate" ) + { + for ( std::size_t n : { 0UL, 1UL, 10UL, 1000UL } ) + { + feng::matrix mi( 1, n ); + std::iota( mi.begin(), mi.end(), -7 ); + feng::matrix md( n, 1 ); + std::iota( md.begin(), md.end(), 0.5 ); + int const ei = std::accumulate( mi.begin(), mi.end(), 100, plus ); + double const ed = std::accumulate( md.begin(), md.end(), 100.0, plus ); + for ( std::size_t w : s6_r1::worker_counts( n ) ) + { + CAPTURE( n ); CAPTURE( w ); + REQUIRE( feng::matrix_details::reduce( mi.begin(), mi.end(), 100, plus, w ) == ei ); + REQUIRE( feng::matrix_details::reduce( md.begin(), md.end(), 100.0, plus, w ) == ed ); + REQUIRE( feng::matrix_details::reduce_impl_private::reduce_impl( mi )( plus, 100, w ) == ei ); + REQUIRE( feng::matrix_details::reduce_impl_private::reduce_impl( md )( plus, 100.0, w ) == ed ); + } + // default worker count + REQUIRE( feng::matrix_details::reduce( mi.begin(), mi.end(), 100, plus ) == ei ); + REQUIRE( feng::matrix_details::reduce_impl_private::reduce_impl( mi )( plus, 100 ) == ei ); + REQUIRE( feng::matrix_details::reduce( plus, 100 )( mi ) == ei ); + } + } + SECTION( "string concatenation shows order and a single init (iterator reduce)" ) + { + std::vector v; + for ( char c = 'a'; c != 'a' + 26; ++c ) v.emplace_back( 1, c ); + auto const cat = []( std::string const& a, std::string const& b ) { return a + b; }; + std::string const expected = std::accumulate( v.begin(), v.end(), std::string{ "<" }, cat ); + REQUIRE( expected == " a x + b stored as complex( a, b ); compose( f, g ) = g after f: associative, not commutative. + using C = std::complex; + auto const compose = []( C f, C g ) { return C{ f.real() * g.real(), g.real() * f.imag() + g.imag() }; }; + feng::matrix m( 2, 6 ); + for ( std::size_t i = 0; i != m.size(); ++i ) + m.data()[i] = C{ ( i % 2 == 0 ) ? 2.0 : 1.0, static_cast( i + 1 ) }; + C const init{ 1.0, 3.0 }; + C const expected = std::accumulate( m.begin(), m.end(), init, compose ); + REQUIRE( expected != std::accumulate( m.begin(), m.end(), C{ 1.0, 0.0 }, compose ) ); + for ( std::size_t w : s6_r1::worker_counts( m.size() ) ) + { + CAPTURE( w ); + REQUIRE( feng::matrix_details::reduce_impl_private::reduce_impl( m )( compose, init, w ) == expected ); + REQUIRE( feng::matrix_details::reduce( m.begin(), m.end(), init, compose, w ) == expected ); + } + } +} + +// S10-T7 (S10-R3, S10-R5): the work-based default worker counts (matrix_details::work_workers) keep small calls on +// the caller, so this case, run by the tsan lane, drives the same partition with more than one worker through the +// GEMM row split (injected and default counts), parallel_work and the default reductions on inputs above the grains. +TEST_CASE( "S6-R1 GEMM row split and work-based defaults run the partition on several workers", "[S6][S6-R1]" ) +{ + SECTION( "gemm_blocked: injected worker counts equal the 1-worker run" ) + { + std::size_t const m = 37, k = 29, n = 41; + feng::matrix a( m, k ), b( k, n ); + for ( std::size_t i = 0; i != a.size(); ++i ) a.data()[i] = 0.25 * static_cast( i % 17 ) - 1.0; + for ( std::size_t i = 0; i != b.size(); ++i ) b.data()[i] = 0.5 * static_cast( i % 13 ) - 2.0; + feng::matrix one( m, n ); + feng::matrix_details::gemm_blocked( a.data(), b.data(), one.data(), m, k, n, 1 ); + for ( std::size_t w : { 2UL, 3UL, 7UL, m, m + 5 } ) + { + CAPTURE( w ); + feng::matrix c( m, n ); + feng::matrix_details::gemm_blocked( a.data(), b.data(), c.data(), m, k, n, w ); + REQUIRE( c == one ); + } + } + SECTION( "operator* above the GEMM grain equals the 1-worker run" ) + { + // 128 * 64 * 64 = 2^19 multiply-adds: two workers by default in parallel builds + std::size_t const m = 128, k = 64, n = 64; + REQUIRE( m * k * n >= 2 * feng::matrix_details::gemm_grain ); + feng::matrix a( m, k ), b( k, n ); + for ( std::size_t i = 0; i != a.size(); ++i ) a.data()[i] = 0.125 * static_cast( i % 11 ) - 0.5; + for ( std::size_t i = 0; i != b.size(); ++i ) b.data()[i] = 0.375 * static_cast( i % 7 ) - 1.0; + feng::matrix one( m, n ); + feng::matrix_details::gemm_blocked( a.data(), b.data(), one.data(), m, k, n, 1 ); + REQUIRE( ( a * b ) == one ); + } + SECTION( "parallel_work and elementwise add above the elementwise grain visit every index once" ) + { + std::size_t const n = 2 * feng::matrix_details::elementwise_grain + 3; +#ifdef FENG_MATRIX_PARALLEL + if ( std::thread::hardware_concurrency() > 1 ) + REQUIRE( feng::matrix_details::work_workers( n, feng::matrix_details::elementwise_grain ) == 2 ); +#endif + std::unique_ptr[]> visits{ new std::atomic[ n ] }; + for ( std::size_t i = 0; i != n; ++i ) visits[i].store( 0 ); + std::vector slots( n, 0 ); + feng::matrix_details::parallel_work( [&]( std::size_t i ) { visits[i].fetch_add( 1 ); slots[i] = i + 1; }, + std::size_t{ 0 }, n, 1, feng::matrix_details::elementwise_grain ); + bool once = true; + for ( std::size_t i = 0; i != n; ++i ) once = once && visits[i].load() == 1 && slots[i] == i + 1; + REQUIRE( once ); + + feng::matrix x( 1, n ), y( 1, n ); + std::iota( x.begin(), x.end(), 0 ); + std::iota( y.begin(), y.end(), 7 ); + auto const z = x + y; + bool sums = true; + for ( std::size_t i = 0; i != n; ++i ) sums = sums && z.data()[i] == static_cast( 2 * i + 7 ); + REQUIRE( sums ); + } + SECTION( "default reductions above the reduce grain equal reduce_range with the same worker count" ) + { + std::size_t const n = 3 * feng::matrix_details::reduce_grain + 7; + std::size_t const w = feng::matrix_details::work_workers( n, feng::matrix_details::reduce_grain ); + feng::matrix m( 1, n ); + for ( std::size_t i = 0; i != n; ++i ) m.data()[i] = 0.1 * static_cast( i % 9 ) - 0.3; + auto const plus = []( double p, double q ) { return p + q; }; + auto const at = [&m]( std::size_t i ) noexcept -> double const& { return m.data()[i]; }; + double const chunked = feng::matrix_details::reduce_range( at, n, 0.0, plus, w ); + REQUIRE( feng::matrix_details::reduce( m.begin(), m.end(), 0.0, plus ) == chunked ); + REQUIRE( feng::matrix_details::reduce_impl_private::reduce_impl( m )( plus, 0.0 ) == chunked ); + REQUIRE( feng::sum( m ) == chunked ); + } +} diff --git a/tests/cases/s6_r3.hpp b/tests/cases/s6_r3.hpp new file mode 100644 index 0000000..0d48bfa --- /dev/null +++ b/tests/cases/s6_r3.hpp @@ -0,0 +1,227 @@ +// S6-R3 (PR-8): caller-owned random engines (F17, D-024). Elements come from the engine alone, in row-major +// order; floating parts lie in (0, 1); integral elements follow std::uniform_int_distribution over the full range. +#include +#include +#include +#include +#include + +namespace s6_r3 +{ + template< typename T > + bool in_open_unit( T x ) { return T{0} < x && x < T{1}; } + + template< typename T > + void check_open_interval() + { + std::mt19937 g{ 2026 }; + auto const m = feng::random( 64, 64, g ); + REQUIRE( m.row() == 64 ); + REQUIRE( m.col() == 64 ); + bool all_inside = true; + for ( auto const x : m ) all_inside = all_inside && in_open_unit( x ); + REQUIRE( all_inside ); + } + + template< typename T > + bool element_in_open_unit( T const& x ) + { + if constexpr ( std::is_floating_point_v ) return in_open_unit( x ); + else return in_open_unit( x.real() ) && in_open_unit( x.imag() ); + } + + template< typename M > + bool all_in_open_unit( M const& m ) + { + bool all_inside = true; + for ( auto const& x : m ) all_inside = all_inside && element_in_open_unit( x ); + return all_inside; + } + + // every engine overload: random( r, c, g ), random( n, g ), rand( r, c, g ), rand( n, g ), rand_like, random_like + template< typename T > + void check_open_interval_all_overloads() + { + std::mt19937_64 g{ 2027 }; + auto const a = feng::random( 16, 24, g ); + auto const b = feng::random( 20, g ); + auto const c = feng::rand( 24, 16, g ); + auto const d = feng::rand( 18, g ); + feng::matrix const shape{ 12, 10 }; + auto const e = feng::rand_like( shape, g ); + auto const f = feng::random_like( shape, g ); + REQUIRE( a.row() == 16 ); + REQUIRE( a.col() == 24 ); + REQUIRE( b.row() == 20 ); + REQUIRE( b.col() == 20 ); + REQUIRE( c.row() == 24 ); + REQUIRE( c.col() == 16 ); + REQUIRE( d.row() == 18 ); + REQUIRE( d.col() == 18 ); + REQUIRE( e.row() == 12 ); + REQUIRE( e.col() == 10 ); + REQUIRE( f.row() == 12 ); + REQUIRE( f.col() == 10 ); + REQUIRE( all_in_open_unit( a ) ); + REQUIRE( all_in_open_unit( b ) ); + REQUIRE( all_in_open_unit( c ) ); + REQUIRE( all_in_open_unit( d ) ); + REQUIRE( all_in_open_unit( e ) ); + REQUIRE( all_in_open_unit( f ) ); + } + + template< typename T > + void check_int_oracle() + { + std::mt19937_64 g{ 11 }; + std::mt19937_64 h{ 11 }; + auto const m = feng::random( 9, 7, g ); + using D = std::conditional_t< ( sizeof( T ) < sizeof( int ) ), std::conditional_t< std::is_signed_v, int, unsigned >, T >; + std::uniform_int_distribution dist{ static_cast( std::numeric_limits::min() ), static_cast( std::numeric_limits::max() ) }; + bool same = true; + for ( auto const x : m ) same = same && ( x == static_cast( dist( h ) ) ); + REQUIRE( same ); + } +} + +TEST_CASE( "S6-R3 equally seeded engines give identical matrices", "[S6][S6-R3]" ) +{ + { + std::mt19937 g{ 5 }, h{ 5 }; + REQUIRE( feng::random( 13, 17, g ) == feng::random( 13, 17, h ) ); + REQUIRE( feng::rand( 13, 17, g ) == feng::rand( 13, 17, h ) ); + REQUIRE( feng::rand( 6, g ) == feng::random( 6, h ) ); + feng::matrix const shape{ 4, 9 }; + auto const a = feng::rand_like( shape, g ); + auto const b = feng::random_like( shape, h ); + REQUIRE( a.row() == 4 ); + REQUIRE( a.col() == 9 ); + REQUIRE( a == b ); + } + { + // the legacy seed overload forwards to the same implementation through a local std::mt19937_64 + std::mt19937_64 g{ 7 }; + REQUIRE( feng::rand( 5, 8, 7 ) == feng::random( 5, 8, g ) ); + REQUIRE( feng::rand( 3, 4 ).row() == 3 ); // still the seed overload + } + { + // row-major order: element k is the k-th draw + std::mt19937 g{ 3 }, h{ 3 }; + auto const m = feng::random( 3, 5, g ); + auto const v = feng::random( 1, 15, h ); + bool same = true; + for ( std::size_t r = 0; r != 3; ++r ) + for ( std::size_t c = 0; c != 5; ++c ) + same = same && ( m[r][c] == v[0][r * 5 + c] ); + REQUIRE( same ); + } +} + +TEST_CASE( "S6-R3 floating and complex elements lie in the open interval (0, 1)", "[S6][S6-R3]" ) +{ + s6_r3::check_open_interval(); + s6_r3::check_open_interval(); + s6_r3::check_open_interval(); + std::mt19937 g{ 99 }; + auto const m = feng::random>( 32, 32, g ); + bool all_inside = true; + for ( auto const z : m ) all_inside = all_inside && s6_r3::in_open_unit( z.real() ) && s6_r3::in_open_unit( z.imag() ); + REQUIRE( all_inside ); + // real part is drawn before the imaginary part + std::mt19937 h{ 99 }, k{ 99 }; + auto const z = feng::random>( 1, 1, h )[0][0]; + auto const p = feng::random( 1, 2, k ); + REQUIRE( z.real() == p[0][0] ); + REQUIRE( z.imag() == p[0][1] ); + s6_r3::check_open_interval_all_overloads(); + s6_r3::check_open_interval_all_overloads(); + s6_r3::check_open_interval_all_overloads(); + s6_r3::check_open_interval_all_overloads>(); + s6_r3::check_open_interval_all_overloads>(); + s6_r3::check_open_interval_all_overloads>(); +} + +TEST_CASE( "S6-R3 integral elements follow a uniform integer distribution", "[S6][S6-R3]" ) +{ + s6_r3::check_int_oracle(); + s6_r3::check_int_oracle(); + s6_r3::check_int_oracle(); + s6_r3::check_int_oracle(); + s6_r3::check_int_oracle(); + s6_r3::check_int_oracle(); + s6_r3::check_int_oracle(); + s6_r3::check_int_oracle(); + s6_r3::check_int_oracle(); + s6_r3::check_int_oracle(); + s6_r3::check_int_oracle(); + s6_r3::check_int_oracle(); + std::mt19937 g{ 1 }; + auto const s = feng::random( 32, 32, g ); + auto const u = feng::random( 32, 32, g ); + bool s_neg = false, s_pos = false, u_low = false, u_high = false; + for ( auto const x : s ) { s_neg = s_neg || x < 0; s_pos = s_pos || x > 0; } + for ( auto const x : u ) { u_low = u_low || x < 128; u_high = u_high || x >= 128; } + REQUIRE( s_neg ); + REQUIRE( s_pos ); + REQUIRE( u_low ); + REQUIRE( u_high ); +} + +TEST_CASE( "S6-R3 engines used from concurrent threads do not interfere", "[S6][S6-R3]" ) +{ + feng::matrix a, b; + std::thread ta( [&a]{ std::mt19937 g{ 42 }; a = feng::random( 256, 256, g ); } ); + std::thread tb( [&b]{ std::mt19937 g{ 42 }; b = feng::random( 256, 256, g ); } ); + ta.join(); + tb.join(); + REQUIRE( a.row() == 256 ); + REQUIRE( a.col() == 256 ); + REQUIRE( a == b ); + std::mt19937 g{ 42 }; + REQUIRE( a == feng::random( 256, 256, g ) ); + // a second pair drawing int + feng::matrix p, q; + std::thread tp( [&p]{ std::mt19937 e{ 42 }; p = feng::random( 256, 256, e ); } ); + std::thread tq( [&q]{ std::mt19937 e{ 42 }; q = feng::random( 256, 256, e ); } ); + tp.join(); + tq.join(); + REQUIRE( p.row() == 256 ); + REQUIRE( p.col() == 256 ); + REQUIRE( p == q ); +} + +TEST_CASE( "S6-R3 legacy seed-only overloads keep their shapes and the open interval", "[S6][S6-R3]" ) +{ + feng::matrix const shape{ 5, 3 }; + auto const a = feng::rand( 4, 6, 9 ); + auto const b = feng::rand( 7 ); + auto const c = feng::random( 3, 8 ); + auto const d = feng::random( 6 ); + auto const e = feng::rand_like( shape ); + auto const f = feng::random_like( shape ); + auto const h = feng::randn_like( shape ); + REQUIRE( a.row() == 4 ); + REQUIRE( a.col() == 6 ); + REQUIRE( b.row() == 7 ); + REQUIRE( b.col() == 7 ); + REQUIRE( c.row() == 3 ); + REQUIRE( c.col() == 8 ); + REQUIRE( d.row() == 6 ); + REQUIRE( d.col() == 6 ); + REQUIRE( e.row() == 5 ); + REQUIRE( e.col() == 3 ); + REQUIRE( f.row() == 5 ); + REQUIRE( f.col() == 3 ); + REQUIRE( h.row() == 5 ); + REQUIRE( h.col() == 3 ); + REQUIRE( s6_r3::all_in_open_unit( a ) ); + REQUIRE( s6_r3::all_in_open_unit( b ) ); + REQUIRE( s6_r3::all_in_open_unit( c ) ); + REQUIRE( s6_r3::all_in_open_unit( d ) ); + REQUIRE( s6_r3::all_in_open_unit( e ) ); + REQUIRE( s6_r3::all_in_open_unit( f ) ); + REQUIRE( s6_r3::all_in_open_unit( h ) ); + // the only legacy overload taking a seed: a nonzero seed equals the engine overload with std::mt19937_64{ seed } + std::mt19937_64 g{ 9 }; + REQUIRE( a == feng::random( 4, 6, g ) ); +} diff --git a/tests/cases/s6_r4.hpp b/tests/cases/s6_r4.hpp new file mode 100644 index 0000000..18506de --- /dev/null +++ b/tests/cases/s6_r4.hpp @@ -0,0 +1,132 @@ +// S6-R4 (PR-9): extrema start from element 0; mean, variance and standard_deviation result types, the shared +// ddof parameter (D-025) and empty-input aborts (D-011) (F15). +#include +#include +#include +#include + +#include "s2_death.hpp" + +namespace s6_r4 +{ + template< typename T > + void check_negative_extrema() + { + feng::matrix const m{ 1, 2, { T( -3 ), T( -1 ) } }; + REQUIRE( m.max() == T( -1 ) ); + REQUIRE( m.min() == T( -3 ) ); + auto const less = []( T const& x, T const& y ) { return x < y; }; + REQUIRE( m.max( less ) == T( -1 ) ); + REQUIRE( m.min( less ) == T( -3 ) ); + auto const [lo, hi] = m.minmax(); + REQUIRE( lo == T( -3 ) ); + REQUIRE( hi == T( -1 ) ); + auto const [lo2, hi2] = m.minmax( less ); + REQUIRE( lo2 == T( -3 ) ); + REQUIRE( hi2 == T( -1 ) ); + REQUIRE( feng::max( m ) == T( -1 ) ); + REQUIRE( feng::min( m ) == T( -3 ) ); + feng::matrix const p{ 1, 2, { T( 1 ), T( 3 ) } }; + REQUIRE( p.min() == T( 1 ) ); + REQUIRE( p.max() == T( 3 ) ); + auto const [plo, phi] = p.minmax(); + REQUIRE( plo == T( 1 ) ); + REQUIRE( phi == T( 3 ) ); + REQUIRE( feng::min( p ) == T( 1 ) ); + } + + template< typename M > + using mean_t = decltype( feng::mean( std::declval< M const& >() ) ); + template< typename M > + using var_t = decltype( feng::variance( std::declval< M const& >() ) ); + template< typename M > + using std_t = decltype( feng::standard_deviation( std::declval< M const& >() ) ); + + using mi = feng::matrix; + using mu = feng::matrix; + using mf = feng::matrix; + using md = feng::matrix; + using mc = feng::matrix>; + static_assert( std::is_same_v< std::remove_cv_t< mean_t >, double > ); + static_assert( std::is_same_v< std::remove_cv_t< mean_t >, double > ); + static_assert( std::is_same_v< std::remove_cv_t< mean_t >, float > ); + static_assert( std::is_same_v< std::remove_cv_t< mean_t >, double > ); + static_assert( std::is_same_v< std::remove_cv_t< mean_t >, std::complex > ); + static_assert( std::is_same_v< std::remove_cv_t< var_t >, double > ); + static_assert( std::is_same_v< std::remove_cv_t< var_t >, double > ); + static_assert( std::is_same_v< std::remove_cv_t< var_t >, float > ); + static_assert( std::is_same_v< std::remove_cv_t< var_t >, double > ); + static_assert( std::is_same_v< std::remove_cv_t< var_t >, double > ); + static_assert( std::is_same_v< std::remove_cv_t< std_t >, double > ); + static_assert( std::is_same_v< std::remove_cv_t< std_t >, double > ); + static_assert( std::is_same_v< std::remove_cv_t< std_t >, float > ); + static_assert( std::is_same_v< std::remove_cv_t< std_t >, double > ); + static_assert( std::is_same_v< std::remove_cv_t< std_t >, double > ); + using mcf = feng::matrix>; + static_assert( std::is_same_v< std::remove_cv_t< var_t >, float > ); + static_assert( noexcept( feng::variance( std::declval< md const& >(), 1 ) ) ); + static_assert( noexcept( feng::standard_deviation( std::declval< md const& >(), 1 ) ) ); +} + +TEST_CASE( "S6-R4 max and min of all-negative matrices", "[S6][S6-R4]" ) +{ + s6_r4::check_negative_extrema(); + s6_r4::check_negative_extrema(); + s6_r4::check_negative_extrema(); +} + +TEST_CASE( "S6-R4 mean of integer matrices is a floating value", "[S6][S6-R4]" ) +{ + REQUIRE( feng::mean( feng::matrix{ 1, 2, { 1, 2 } } ) == 1.5 ); + REQUIRE( feng::mean( feng::matrix{ 1, 2, { -1, -2 } } ) == -1.5 ); + REQUIRE( feng::mean( feng::matrix{ 1, 2, { 1U, 2U } } ) == 1.5 ); + REQUIRE( feng::mean( feng::matrix{ 1, 2, { 1.0f, 2.0f } } ) == 1.5f ); + auto const z = feng::mean( feng::matrix>{ 1, 2, { { 1.0, 2.0 }, { 2.0, -4.0 } } } ); + REQUIRE( z == std::complex{ 1.5, -1.0 } ); +} + +TEST_CASE( "S6-R4 variance and standard_deviation share ddof", "[S6][S6-R4]" ) +{ + feng::matrix const m{ 1, 4, { 1, 2, 3, 4 } }; + // mean 2.5; squared deviations 2.25 + 0.25 + 0.25 + 2.25 = 5 + REQUIRE( feng::variance( m ) == 1.25 ); + REQUIRE( feng::variance( m, 0 ) == 1.25 ); + REQUIRE( std::abs( feng::variance( m, 1 ) - 5.0 / 3.0 ) < 1.0e-12 ); + REQUIRE( std::abs( feng::standard_deviation( m ) - std::sqrt( 1.25 ) ) < 1.0e-12 ); + REQUIRE( std::abs( feng::standard_deviation( m, 1 ) - std::sqrt( 5.0 / 3.0 ) ) < 1.0e-12 ); + feng::matrix const d{ 2, 2, { 1.0, 2.0, 3.0, 4.0 } }; + REQUIRE( feng::variance( d ) == 1.25 ); + REQUIRE( feng::standard_deviation( d, 1 ) * feng::standard_deviation( d, 1 ) == Approx( feng::variance( d, 1 ) ) ); + feng::matrix const u{ 1, 2, { 1U, 3U } }; + REQUIRE( feng::variance( u ) == 1.0 ); // no unsigned wrap of x - mean + // complex: |z - mean|^2 with mean (0, 0): |1+i|^2 = |-1-i|^2 = 2 + feng::matrix> const c{ 1, 2, { { 1.0, 1.0 }, { -1.0, -1.0 } } }; + REQUIRE( feng::variance( c ) == 2.0 ); + REQUIRE( feng::standard_deviation( c ) == std::sqrt( 2.0 ) ); + feng::matrix> const cf{ 1, 2, { { 1.0f, 1.0f }, { -1.0f, -1.0f } } }; + auto const vf = feng::variance( cf ); + static_assert( std::is_same_v< std::remove_cv_t< decltype( vf ) >, float > ); + REQUIRE( vf == 2.0f ); +} + +TEST_CASE( "S6-R4 empty inputs abort with a message", "[S6][S6-R4]" ) +{ + S2_REQUIRE_DEATH( []{ feng::matrix const e; (void)e.max(); }, "matrix max: empty matrix" ); + S2_REQUIRE_DEATH( []{ feng::matrix const e; (void)e.min(); }, "matrix min: empty matrix" ); + S2_REQUIRE_DEATH( []{ feng::matrix const e; (void)e.minmax(); }, "matrix minmax: empty matrix" ); + S2_REQUIRE_DEATH( []{ feng::matrix const e; (void)feng::max( e ); }, "feng::max: empty matrix" ); + S2_REQUIRE_DEATH( []{ feng::matrix const e; (void)feng::min( e ); }, "feng::min: empty matrix" ); + S2_REQUIRE_DEATH( []{ feng::matrix const e; (void)feng::mean( e ); }, "feng::mean: empty matrix" ); + S2_REQUIRE_DEATH( []{ feng::matrix const e; (void)feng::variance( e ); }, "feng::variance: " ); + S2_REQUIRE_DEATH( []{ feng::matrix const e; (void)feng::standard_deviation( e ); }, "feng::standard_deviation: " ); + S2_REQUIRE_DEATH( []{ feng::matrix const one{ 1, 1, { 2.0 } }; (void)feng::variance( one, 1 ); }, "feng::variance: " ); + S2_REQUIRE_DEATH( []{ feng::matrix const one{ 1, 1, { 2.0 } }; (void)feng::standard_deviation( one, 1 ); }, "feng::standard_deviation: " ); + S2_REQUIRE_DEATH( []{ feng::matrix const e; (void)e.max( []( double x, double y ) { return x < y; } ); }, "matrix max: empty matrix" ); + S2_REQUIRE_DEATH( []{ feng::matrix const e; (void)e.min( []( double x, double y ) { return x < y; } ); }, "matrix min: empty matrix" ); + S2_REQUIRE_DEATH( []{ feng::matrix const e; (void)e.minmax( []( double x, double y ) { return x < y; } ); }, "matrix minmax: empty matrix" ); + S2_REQUIRE_DEATH( []{ feng::matrix> const e; (void)feng::mean( e ); }, "feng::mean: empty matrix" ); + S2_REQUIRE_DEATH( []{ feng::matrix> const e; (void)feng::variance( e ); }, "feng::variance: needs more elements than ddof" ); + S2_REQUIRE_DEATH( []{ feng::matrix> const e; (void)feng::standard_deviation( e ); }, "feng::standard_deviation: needs more elements than ddof" ); + S2_REQUIRE_DEATH( []{ feng::matrix const e; (void)feng::variance( e ); }, "feng::variance: needs more elements than ddof" ); + S2_REQUIRE_DEATH( []{ feng::matrix const e; (void)feng::standard_deviation( e ); }, "feng::standard_deviation: needs more elements than ddof" ); +} diff --git a/tests/cases/s6_r5.hpp b/tests/cases/s6_r5.hpp new file mode 100644 index 0000000..e308ff4 --- /dev/null +++ b/tests/cases/s6_r5.hpp @@ -0,0 +1,336 @@ +// S6-R5 (PR-9): the promotion policy for scalar and mixed-type arithmetic (D-023, F17). +// matrix (+) matrix gives common_element_t; matrix (+) scalar keeps T unless the scalar's kind +// (integral < floating < complex) is higher; compound assignment keeps T; the result allocator is A rebound. +#include +#include +#include +#include +#include +#include + +#include "./s3_alloc.hpp" + +namespace s6_r5 +{ + using cf = std::complex; + using cd = std::complex; + + template< typename T > + using mat = feng::matrix; + + template< typename T, typename S > using add_t = std::remove_cvref_t< decltype( std::declval< mat const& >() + std::declval< S const& >() ) >; + template< typename T, typename S > using radd_t = std::remove_cvref_t< decltype( std::declval< S const& >() + std::declval< mat const& >() ) >; + template< typename T, typename S > using sub_t = std::remove_cvref_t< decltype( std::declval< mat const& >() - std::declval< S const& >() ) >; + template< typename T, typename S > using rsub_t = std::remove_cvref_t< decltype( std::declval< S const& >() - std::declval< mat const& >() ) >; + template< typename T, typename S > using mul_t = std::remove_cvref_t< decltype( std::declval< mat const& >() * std::declval< S const& >() ) >; + template< typename T, typename S > using rmul_t = std::remove_cvref_t< decltype( std::declval< S const& >() * std::declval< mat const& >() ) >; + template< typename T, typename S > using div_t = std::remove_cvref_t< decltype( std::declval< mat const& >() / std::declval< S const& >() ) >; + template< typename T, typename S > using rdiv_t = std::remove_cvref_t< decltype( std::declval< S const& >() / std::declval< mat const& >() ) >; + + template< typename T, typename U > using madd_t = std::remove_cvref_t< decltype( std::declval< mat const& >() + std::declval< mat const& >() ) >; + template< typename T, typename U > using msub_t = std::remove_cvref_t< decltype( std::declval< mat const& >() - std::declval< mat const& >() ) >; + template< typename T, typename U > using mmul_t = std::remove_cvref_t< decltype( std::declval< mat const& >() * std::declval< mat const& >() ) >; + template< typename T, typename U > using mdiv_t = std::remove_cvref_t< decltype( std::declval< mat const& >() / std::declval< mat const& >() ) >; + + // The expected element type, written out independently of the library's traits. + template< typename T > constexpr int kind() { if constexpr ( std::is_integral_v ) return 0; else if constexpr ( std::is_floating_point_v ) return 1; else return 2; } + template< typename T > struct real_of { using type = T; }; + template< typename X > struct real_of< std::complex > { using type = X; }; + template< typename T, typename U > + using expected_common_t = std::conditional_t< kind() == 2 || kind() == 2, + std::complex< std::common_type_t< typename real_of::type, typename real_of::type > >, + std::common_type_t< T, U > >; + template< typename T, typename S > + using expected_scalar_t = std::conditional_t< ( kind() <= kind() ), T, expected_common_t< T, S > >; + + template< typename T, typename S > + constexpr bool scalar_pair() + { + using R = expected_scalar_t< T, S >; + static_assert( std::is_same_v< feng::matrix_details::scalar_result_t< T, S >, R > ); + static_assert( std::is_same_v< add_t, mat > ); + static_assert( std::is_same_v< radd_t, mat > ); + static_assert( std::is_same_v< sub_t, mat > ); + static_assert( std::is_same_v< rsub_t, mat > ); + static_assert( std::is_same_v< mul_t, mat > ); + static_assert( std::is_same_v< rmul_t, mat > ); + static_assert( std::is_same_v< div_t, mat > ); + static_assert( std::is_same_v< rdiv_t, mat > ); + return true; + } + + template< typename T, typename U > + constexpr bool matrix_pair() + { + using R = expected_common_t< T, U >; + static_assert( std::is_same_v< feng::matrix_details::common_element_t< T, U >, R > ); + static_assert( std::is_same_v< madd_t, mat > ); + static_assert( std::is_same_v< msub_t, mat > ); + static_assert( std::is_same_v< mmul_t, mat > ); + static_assert( std::is_same_v< mdiv_t, mat > ); + return true; + } + + template< typename T, typename... S > + constexpr bool scalar_row() { return ( scalar_pair< T, S >() && ... ); } + template< typename T, typename... U > + constexpr bool matrix_row() { return ( matrix_pair< T, U >() && ... ); } + + template< typename... T > + constexpr bool scalar_grid() { return ( scalar_row< T, int, unsigned, std::uint8_t, float, double, cf, cd >() && ... ); } + template< typename... T > + constexpr bool matrix_grid() { return ( matrix_row< T, int, unsigned, std::uint8_t, float, double, cf, cd >() && ... ); } + + static_assert( scalar_grid< int, unsigned, std::uint8_t, float, double, cf, cd >() ); + static_assert( matrix_grid< int, unsigned, std::uint8_t, float, double, cf, cd >() ); + + // The traits as D-023 states them, spot-checked by hand. + static_assert( feng::matrix_details::element_kind::value == 0 ); + static_assert( feng::matrix_details::element_kind::value == 1 ); + static_assert( feng::matrix_details::element_kind::value == 2 ); + static_assert( std::is_same_v< feng::matrix_details::common_element_t< cf, double >, cd > ); + static_assert( std::is_same_v< feng::matrix_details::common_element_t< int, unsigned >, unsigned > ); + static_assert( std::is_same_v< feng::matrix_details::scalar_result_t< float, double >, float > ); + static_assert( std::is_same_v< feng::matrix_details::scalar_result_t< std::uint8_t, int >, std::uint8_t > ); + static_assert( std::is_same_v< feng::matrix_details::scalar_result_t< int, double >, double > ); + static_assert( std::is_same_v< feng::matrix_details::scalar_result_t< float, cd >, cd > ); + static_assert( std::is_same_v< feng::matrix_details::scalar_result_t< cf, double >, cf > ); + + // A scalar is an arithmetic type or a std::complex: matrices, pointers, valarray and vector are not. + static_assert( feng::matrix_details::matrix_scalar && feng::matrix_details::matrix_scalar ); + static_assert( !feng::matrix_details::matrix_scalar< mat > ); + static_assert( !feng::matrix_details::matrix_scalar< double const* > ); + static_assert( !feng::matrix_details::matrix_scalar< std::valarray > ); + static_assert( !feng::matrix_details::matrix_scalar< std::vector > ); + + // Compound assignment keeps the left element type. + template< typename T, typename S > using cadd_t = decltype( std::declval< mat& >() += std::declval< S const& >() ); + template< typename T, typename S > using cmul_t = decltype( std::declval< mat& >() *= std::declval< S const& >() ); + static_assert( std::is_same_v< cadd_t< int, double >, mat& > ); + static_assert( std::is_same_v< cmul_t< float, double >, mat& > ); + static_assert( std::is_same_v< cadd_t< double, mat >, mat& > ); + static_assert( std::is_same_v< cmul_t< double, mat >, mat& > ); +} + +TEST_CASE( "S6-R5 scalar result types", "[S6][S6-R5]" ) +{ + using namespace s6_r5; + + SECTION( "matrix * 0.5 promotes to double" ) + { + mat const m{ 1, 2, { 1, 2 } }; + auto const r = m * 0.5; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( r ) >, mat > ); + REQUIRE( r[0][0] == 0.5 ); + REQUIRE( r[0][1] == 1.0 ); + auto const l = 0.5 * m; + REQUIRE( l[0][0] == 0.5 ); + REQUIRE( l[0][1] == 1.0 ); + auto const d = m / 4.0; + REQUIRE( d[0][0] == 0.25 ); + REQUIRE( d[0][1] == 0.5 ); + auto const s = 3.5 - m; + REQUIRE( s[0][0] == 2.5 ); + REQUIRE( s[0][1] == 1.5 ); + } + + SECTION( "matrix * 2 compiles and keeps double" ) + { + mat const m{ 1, 2, { 1.5, -2.0 } }; + auto const r = m * 2; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( r ) >, mat > ); + REQUIRE( r[0][0] == 3.0 ); + REQUIRE( r[0][1] == -4.0 ); + auto const p = 1 + m; + REQUIRE( p[0][0] == 2.5 ); + REQUIRE( p[0][1] == -1.0 ); + } + + SECTION( "matrix * 2.0 stays float, the scalar converted to float first" ) + { + mat const m{ 1, 2, { 1.0f, 3.0f } }; + auto const r = m * 2.0; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( r ) >, mat > ); + REQUIRE( r[0][0] == 2.0f ); + REQUIRE( r[0][1] == 6.0f ); + // 0.1 rounded to float first, then multiplied in float. + auto const t = m * 0.1; + REQUIRE( t[0][1] == 3.0f * static_cast( 0.1 ) ); + } + + SECTION( "matrix + 1 stays uint8_t and wraps" ) + { + mat const m{ 1, 2, { std::uint8_t( 10 ), std::uint8_t( 255 ) } }; + auto const r = m + 1; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( r ) >, mat > ); + REQUIRE( r[0][0] == 11 ); + REQUIRE( r[0][1] == 0 ); + auto const s = 1 - m; + REQUIRE( s[0][0] == std::uint8_t( 1 - 10 ) ); + REQUIRE( s[0][1] == 2 ); + } + + SECTION( "complex matrices with real scalars keep the complex type" ) + { + mat const m{ 1, 2, { cd( 1.0, 2.0 ), cd( -1.0, 0.5 ) } }; + auto const r = m * 2; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( r ) >, mat > ); + REQUIRE( r[0][0] == cd( 2.0, 4.0 ) ); + REQUIRE( r[0][1] == cd( -2.0, 1.0 ) ); + auto const a = 1.0 + m; + REQUIRE( a[0][0] == cd( 2.0, 2.0 ) ); + auto const s = 1 - m; + REQUIRE( s[0][1] == cd( 2.0, -0.5 ) ); + mat const f{ 1, 1, { cf( 1.0f, 1.0f ) } }; + auto const g = f * 2.0; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( g ) >, mat > ); + REQUIRE( g[0][0] == cf( 2.0f, 2.0f ) ); + } + + SECTION( "real matrices with complex scalars promote to complex" ) + { + mat const m{ 1, 2, { 1, 2 } }; + auto const r = m * cd( 0.0, 1.0 ); + static_assert( std::is_same_v< std::remove_cvref_t< decltype( r ) >, mat > ); + REQUIRE( r[0][0] == cd( 0.0, 1.0 ) ); + REQUIRE( r[0][1] == cd( 0.0, 2.0 ) ); + } + + SECTION( "scalar / matrix is the scalar times the inverse" ) + { + mat const m{ 2, 2, { 2, 0, 0, 4 } }; + auto const r = 1.0 / m; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( r ) >, mat > ); + REQUIRE( r[0][0] == 0.5 ); + REQUIRE( r[0][1] == 0.0 ); + REQUIRE( r[1][1] == 0.25 ); + mat const d{ 2, 2, { 2.0, 0.0, 0.0, 4.0 } }; + auto const q = 2 / d; + REQUIRE( q[0][0] == 1.0 ); + REQUIRE( q[1][1] == 0.5 ); + } + + SECTION( "compound assignment keeps T, the scalar converted as by static_cast" ) + { + mat m{ 1, 2, { 3, 5 } }; + m *= 0.5; + // 0.5 becomes static_cast( 0.5 ) == 0 before the multiplication. + REQUIRE( m[0][0] == 0 ); + REQUIRE( m[0][1] == 0 ); + mat n{ 1, 2, { 3, 5 } }; + n *= 2.9; + REQUIRE( n[0][0] == 6 ); + REQUIRE( n[0][1] == 10 ); + mat f{ 1, 1, { 1.0f } }; + f += 2.0; + REQUIRE( f[0][0] == 3.0f ); + } + + SECTION( "the result allocator is A rebound to the result type" ) + { + using alloc_i = s3_alloc::tracking_allocator< int, false, false, false >; + using alloc_d = s3_alloc::tracking_allocator< double, false, false, false >; + s3_alloc::reset(); + { + feng::matrix< int, alloc_i > const m{ alloc_i{ 7 }, 1, 2 }; + auto const r = m * 0.5; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( r ) >, feng::matrix< double, alloc_d > > ); + REQUIRE( r.get_allocator().id == 7 ); + auto const l = 2.0 + m; + REQUIRE( l.get_allocator().id == 7 ); + auto const k = m + 1; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( k ) >, feng::matrix< int, alloc_i > > ); + REQUIRE( k.get_allocator().id == 7 ); + } + s3_alloc::require_balanced(); + } +} + +TEST_CASE( "S6-R5 mixed matrix result types", "[S6][S6-R5]" ) +{ + using namespace s6_r5; + + SECTION( "int + double gives double" ) + { + mat const a{ 1, 2, { 1, 2 } }; + mat const b{ 1, 2, { 0.5, 0.25 } }; + auto const r = a + b; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( r ) >, mat > ); + REQUIRE( r[0][0] == 1.5 ); + REQUIRE( r[0][1] == 2.25 ); + auto const s = b - a; + REQUIRE( s[0][0] == -0.5 ); + REQUIRE( s[0][1] == -1.75 ); + } + + SECTION( "double + complex gives complex" ) + { + mat const a{ 1, 1, { 1.0 } }; + mat const b{ 1, 1, { cf( 0.5f, 2.0f ) } }; + auto const r = a + b; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( r ) >, mat > ); + REQUIRE( r[0][0] == cd( 1.5, 2.0 ) ); + } + + SECTION( "product and division of mixed types" ) + { + mat const a{ 1, 2, { 1, 2 } }; + mat const b{ 2, 1, { 0.5, 0.25 } }; + auto const p = a * b; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( p ) >, mat > ); + REQUIRE( p.row() == 1 ); + REQUIRE( p.col() == 1 ); + REQUIRE( p[0][0] == 1.0 ); + mat const c{ 1, 2, { 1, 2 } }; + mat const d{ 2, 2, { 2.0, 0.0, 0.0, 4.0 } }; + auto const q = c / d; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( q ) >, mat > ); + REQUIRE( q[0][0] == 0.5 ); + REQUIRE( q[0][1] == 0.5 ); + } + + SECTION( "same-type results are unchanged" ) + { + mat const a{ 1, 2, { 1, 2 } }; + mat const b{ 1, 2, { 3, 4 } }; + auto const r = a + b; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( r ) >, mat > ); + REQUIRE( r[0][0] == 4 ); + REQUIRE( r[0][1] == 6 ); + } + + SECTION( "compound assignment with another element type keeps T" ) + { + mat a{ 1, 2, { 1, 2 } }; + mat const b{ 1, 2, { 0.5, 1.5 } }; + a += b; + // each element of b becomes static_cast first: 0 and 1. + REQUIRE( a[0][0] == 1 ); + REQUIRE( a[0][1] == 3 ); + mat c{ 1, 2, { 1.0, 2.0 } }; + mat const d{ 1, 2, { 1, 1 } }; + c -= d; + REQUIRE( c[0][0] == 0.0 ); + REQUIRE( c[0][1] == 1.0 ); + } + + SECTION( "the result allocator is the left allocator rebound" ) + { + using alloc_i = s3_alloc::tracking_allocator< int, false, false, false >; + using alloc_d = s3_alloc::tracking_allocator< double, false, false, false >; + s3_alloc::reset(); + { + feng::matrix< int, alloc_i > const a{ alloc_i{ 9 }, 1, 2 }; + mat const b{ 1, 2, { 0.5, 0.5 } }; + auto const r = a + b; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( r ) >, feng::matrix< double, alloc_d > > ); + REQUIRE( r.get_allocator().id == 9 ); + REQUIRE( r[0][0] == 0.5 ); + feng::matrix< int, alloc_i > const c{ alloc_i{ 9 }, 1, 2, 1 }; + auto const s = c + a; + static_assert( std::is_same_v< std::remove_cvref_t< decltype( s ) >, feng::matrix< int, alloc_i > > ); + REQUIRE( s.get_allocator().id == 9 ); + } + s3_alloc::require_balanced(); + } +} diff --git a/tests/cases/s7_r1.hpp b/tests/cases/s7_r1.hpp new file mode 100644 index 0000000..72fc102 --- /dev/null +++ b/tests/cases/s7_r1.hpp @@ -0,0 +1,331 @@ +// S7-R1 (PR-10, PR-2): one partial-pivoting LU object; legacy lu_decomposition and lu_solver rest on it (F13). +// The helpers in namespace s7 are shared with tests/cases/s7_r2.hpp (included after this file). +#include +#include +#include +#include +#include +#include +#include +#include + +#include "./s2_death.hpp" + +namespace s7 +{ + template< typename T > struct real_of { using type = T; }; + template< typename T > struct real_of< std::complex< T > > { using type = T; }; + template< typename T > using real_t = typename real_of< T >::type; + + template< typename T > real_t< T > eps() { return std::numeric_limits< real_t< T > >::epsilon(); } + + template< typename T > using mat = feng::matrix< T >; + + // ∞-norm: largest absolute row sum + template< typename T > + real_t< T > norm_inf( mat< T > const& m ) + { + real_t< T > best{ 0 }; + for ( std::size_t r = 0; r != m.row(); ++r ) + { + real_t< T > s{ 0 }; + for ( std::size_t c = 0; c != m.col(); ++c ) s += std::abs( m[r][c] ); + best = s > best ? s : best; + } + return best; + } + + template< typename T > + mat< T > from_real( std::size_t r, std::size_t c, std::vector< double > const& v ) + { + mat< T > m{ r, c }; + for ( std::size_t i = 0; i != r * c; ++i ) m[i / c][i % c] = static_cast< T >( static_cast< real_t< T > >( v[i] ) ); + return m; + } + + template< typename T > + mat< T > hilbert( std::size_t n ) + { + mat< T > m{ n, n }; + for ( std::size_t i = 0; i != n; ++i ) + for ( std::size_t j = 0; j != n; ++j ) + m[i][j] = T( real_t< T >( 1 ) / real_t< T >( i + j + 1 ) ); + return m; + } + + template< typename T > + mat< T > random_square( std::size_t n, std::uint_least64_t seed ) + { + std::mt19937_64 g{ seed }; + return feng::random< T >( n, n, g ); + } + + // the R09 inputs: identity, diagonal, row swap, 4×4 block exchange, random n = 1..12, Hilbert 8 and the + // near-singular 2×2 + template< typename T > + std::vector< std::pair< std::string, mat< T > > > r09_inputs() + { + std::vector< std::pair< std::string, mat< T > > > v; + v.emplace_back( "identity 5", feng::eye< T >( 5, 5 ) ); + v.emplace_back( "diagonal", from_real< T >( 3, 3, { 2, 0, 0, 0, -3, 0, 0, 0, 0.5 } ) ); + v.emplace_back( "row swap", from_real< T >( 3, 3, { 0, 1, 0, 1, 0, 0, 0, 0, 1 } ) ); + v.emplace_back( "block exchange", from_real< T >( 4, 4, { 0, 0, 1, 2, 0, 0, 3, 4, 5, 6, 0, 0, 7, 8, 0, 0 } ) ); + for ( std::size_t n = 1; n <= 12; ++n ) + v.emplace_back( "random " + std::to_string( n ), random_square< T >( n, 0x5701ULL + n ) ); + if constexpr ( sizeof( real_t< T > ) >= sizeof( double ) ) + { + v.emplace_back( "hilbert 8", hilbert< T >( 8 ) ); + v.emplace_back( "near singular 2x2", from_real< T >( 2, 2, { 1, 1, 1, 1 + 1e-10 } ) ); + } + else + { + // float: Hilbert 8 and 1 + 1e-10 are singular at ε = 1.2e-7 (float_singular_inputs); these keep the + // near-singular shape at float precision + v.emplace_back( "hilbert 4", hilbert< T >( 4 ) ); + v.emplace_back( "near singular 2x2", from_real< T >( 2, 2, { 1, 1, 1, 1 + 1e-3 } ) ); + } + return v; + } + + // the double near-singular inputs read singular in float (D-027 tolerance n·ε·max|U|), with the rank shown + template< typename T > + std::vector< std::pair< mat< T >, std::size_t > > float_singular_inputs() + { + return { { hilbert< T >( 8 ), 6 }, { from_real< T >( 2, 2, { 1, 1, 1, 1 + 1e-10 } ), 1 } }; + } + + template< typename T > + mat< T > permute_rows( mat< T > const& a, std::vector< std::size_t > const& piv ) + { + mat< T > pa{ a.row(), a.col() }; + for ( std::size_t i = 0; i != a.row(); ++i ) + for ( std::size_t j = 0; j != a.col(); ++j ) pa[i][j] = a[piv[i]][j]; + return pa; + } + + template< typename T > + void check_lu_reconstructs() + { + for ( auto const& [name, a] : r09_inputs< T >() ) + { + INFO( name ); + std::size_t const n = a.row(); + auto const f = feng::lu_factor( a ); + REQUIRE( f.status() == feng::linalg_status::ok ); + REQUIRE( f.rank() == n ); + auto const& piv = f.pivots(); + REQUIRE( piv.size() == n ); + mat< T > const l = f.l(), u = f.u(), p = f.p(); + REQUIRE( l.row() == n ); REQUIRE( l.col() == n ); + REQUIRE( u.row() == n ); REQUIRE( u.col() == n ); + for ( std::size_t i = 0; i != n; ++i ) + { + REQUIRE( l[i][i] == T( 1 ) ); + for ( std::size_t j = i + 1; j != n; ++j ) { REQUIRE( l[i][j] == T( 0 ) ); REQUIRE( u[j][i] == T( 0 ) ); } + for ( std::size_t j = 0; j != i; ++j ) REQUIRE( std::abs( l[i][j] ) <= real_t< T >( 1 ) ); // partial pivoting + } + mat< T > const pa = permute_rows( a, piv ); + REQUIRE( norm_inf< T >( p * a - pa ) == real_t< T >( 0 ) ); + real_t< T > const res = norm_inf< T >( pa - l * u ); + REQUIRE( res <= real_t< T >( 4 ) * real_t< T >( n ) * eps< T >() * norm_inf( a ) ); + } + if constexpr ( sizeof( real_t< T > ) < sizeof( double ) ) + for ( auto const& [a, rank] : float_singular_inputs< T >() ) + { + std::size_t const n = a.row(); + auto const f = feng::lu_factor( a ); + REQUIRE( f.status() == feng::linalg_status::singular ); + REQUIRE( f.rank() == rank ); + REQUIRE( norm_inf< T >( permute_rows( a, f.pivots() ) - f.l() * f.u() ) <= real_t< T >( 4 ) * real_t< T >( n ) * eps< T >() * norm_inf( a ) ); + } + } + + // ‖AX − B‖ / (‖A‖‖X‖ + ‖B‖) / (n·ε) + template< typename T > + real_t< T > solve_ratio( mat< T > const& a, mat< T > const& x, mat< T > const& b ) + { + real_t< T > const n = static_cast< real_t< T > >( a.row() ); + return norm_inf< T >( a * x - b ) / ( norm_inf( a ) * norm_inf( x ) + norm_inf( b ) ) / ( n * eps< T >() ); + } +} + +TEST_CASE( "S7-R1 pivoted LU reconstructs P·A on the R09 inputs", "[S7][S7-R1]" ) +{ + s7::check_lu_reconstructs< double >(); + s7::check_lu_reconstructs< std::complex< double > >(); + s7::check_lu_reconstructs< float >(); + s7::check_lu_reconstructs< std::complex< float > >(); +} + +namespace s7 +{ + template< typename T > + void check_pivot_rules() + { + // the first entry of largest |·| is the pivot: rows 1 and 2 tie at 3 in column 0 + auto const a = from_real< T >( 3, 3, { 1, 2, 3, 3, 1, 1, -3, 4, 2 } ); + auto const f = feng::lu_factor( a ); + REQUIRE( f.pivots()[0] == 1 ); + // a zero column skips elimination and the factorization reports singular, rank 2 + auto const z = from_real< T >( 3, 3, { 0, 1, 2, 0, 3, 4, 0, 5, 7 } ); + auto const fz = feng::lu_factor( z ); + REQUIRE( fz.status() == feng::linalg_status::singular ); + REQUIRE( fz.rank() == 2 ); + REQUIRE( norm_inf< T >( permute_rows( z, fz.pivots() ) - fz.l() * fz.u() ) <= real_t< T >( 12 ) * eps< T >() * norm_inf( z ) ); + // [[1, 2, 3], [4, 5, 6], [7, 8, 9]] is singular + auto const s = from_real< T >( 3, 3, { 1, 2, 3, 4, 5, 6, 7, 8, 9 } ); + REQUIRE( feng::lu_factor( s ).status() == feng::linalg_status::singular ); + REQUIRE( feng::lu_factor( s ).rank() == 2 ); + // nonfinite input + auto nf = feng::eye< T >( 3, 3 ); + nf[1][2] = T( std::numeric_limits< real_t< T > >::quiet_NaN() ); + REQUIRE( feng::lu_factor( nf ).status() == feng::linalg_status::nonfinite ); + nf[1][2] = T( std::numeric_limits< real_t< T > >::infinity() ); + REQUIRE( feng::lu_factor( nf ).status() == feng::linalg_status::nonfinite ); + // 0×0 is a valid, ok factorization + auto const e = feng::lu_factor( mat< T >{} ); + REQUIRE( e.status() == feng::linalg_status::ok ); + REQUIRE( e.rank() == 0 ); + } + + template< typename T > + void check_solve() + { + real_t< T > worst{ 0 }; + for ( auto const& [name, a] : r09_inputs< T >() ) + { + INFO( name ); + std::size_t const n = a.row(); + for ( std::size_t k : { std::size_t{ 1 }, std::size_t{ 3 } } ) + { + std::mt19937_64 g{ 0x57B0ULL + n * 7 + k }; + mat< T > const b = feng::random< T >( n, k, g ); + auto const r = feng::lu_factor( a ).solve( b ); + REQUIRE( r.ok() ); + REQUIRE( static_cast< bool >( r ) ); + REQUIRE( r.status == feng::linalg_status::ok ); + REQUIRE( r.value.row() == n ); REQUIRE( r.value.col() == k ); + real_t< T > const ratio = solve_ratio( a, r.value, b ); + worst = ratio > worst ? ratio : worst; + REQUIRE( ratio <= real_t< T >( 16 ) ); + auto const r2 = feng::solve( a, b ); + REQUIRE( r2.ok() ); + REQUIRE( norm_inf< T >( r2.value - r.value ) == real_t< T >( 0 ) ); + } + } + INFO( "worst solve residual ratio " << worst ); + // singular: status singular and an empty value + auto const s = from_real< T >( 3, 3, { 1, 2, 3, 4, 5, 6, 7, 8, 9 } ); + auto const b = from_real< T >( 3, 1, { 1, 1, 1 } ); + auto const r = feng::solve( s, b ); + REQUIRE( !r.ok() ); + REQUIRE( !static_cast< bool >( r ) ); + REQUIRE( r.status == feng::linalg_status::singular ); + REQUIRE( r.value.size() == 0 ); + } + + template< typename T > + void check_legacy() + { + for ( auto const& [name, a] : r09_inputs< T >() ) + { + INFO( name ); + std::size_t const n = a.row(); + mat< T > l, u; + REQUIRE( feng::lu_decomposition( a, l, u ) == 0 ); + // MATLAB's two-output form: L = Pᵀ·L₀ is a permuted unit lower triangle, A = L·U + auto const f = feng::lu_factor( a ); + REQUIRE( norm_inf< T >( l - f.p().transpose() * f.l() ) == real_t< T >( 0 ) ); + REQUIRE( norm_inf< T >( u - f.u() ) == real_t< T >( 0 ) ); + REQUIRE( norm_inf< T >( a - l * u ) <= real_t< T >( 4 ) * real_t< T >( n ) * eps< T >() * norm_inf( a ) ); + auto const lu = feng::lu_decomposition( a ); + REQUIRE( lu.has_value() ); + REQUIRE( norm_inf< T >( std::get< 0 >( *lu ) - l ) == real_t< T >( 0 ) ); + REQUIRE( norm_inf< T >( std::get< 1 >( *lu ) - u ) == real_t< T >( 0 ) ); + + std::mt19937_64 g{ 0x57C0ULL + n }; + mat< T > const b = feng::random< T >( n, 1, g ); + mat< T > x; + REQUIRE( feng::lu_solver( a, x, b ) == 0 ); + REQUIRE( solve_ratio( a, x, b ) <= real_t< T >( 16 ) ); + auto const ox = feng::lu_solver( a, b ); + REQUIRE( ox.has_value() ); + REQUIRE( norm_inf< T >( *ox - x ) == real_t< T >( 0 ) ); + } + // not ok: 1 / nullopt, outputs unchanged + for ( auto const& s : { from_real< T >( 3, 3, { 1, 2, 3, 4, 5, 6, 7, 8, 9 } ), + from_real< T >( 2, 2, { 1, std::numeric_limits< double >::quiet_NaN(), 0, 1 } ) } ) + { + mat< T > l{ 1, 1, T( 7 ) }, u{ 1, 2, T( 9 ) }; + REQUIRE( feng::lu_decomposition( s, l, u ) == 1 ); + REQUIRE( l.row() == 1 ); REQUIRE( l.col() == 1 ); REQUIRE( l[0][0] == T( 7 ) ); + REQUIRE( u.row() == 1 ); REQUIRE( u.col() == 2 ); REQUIRE( u[0][1] == T( 9 ) ); + REQUIRE( !feng::lu_decomposition( s ).has_value() ); + mat< T > const b{ s.row(), 1, T( 1 ) }; + mat< T > x{ 2, 2, T( 5 ) }; + REQUIRE( feng::lu_solver( s, x, b ) == 1 ); + REQUIRE( x.row() == 2 ); REQUIRE( x.col() == 2 ); REQUIRE( x[1][1] == T( 5 ) ); + REQUIRE( !feng::lu_solver( s, b ).has_value() ); + } + } +} + +TEST_CASE( "S7-R1 pivot choice, rank and status", "[S7][S7-R1]" ) +{ + s7::check_pivot_rules< double >(); + s7::check_pivot_rules< std::complex< double > >(); + s7::check_pivot_rules< float >(); + s7::check_pivot_rules< std::complex< float > >(); +} + +TEST_CASE( "S7-R1 solve meets the residual bound c = 16", "[S7][S7-R1]" ) +{ + s7::check_solve< double >(); + s7::check_solve< std::complex< double > >(); + s7::check_solve< float >(); + s7::check_solve< std::complex< float > >(); +} + +TEST_CASE( "S7-R1 legacy lu_decomposition and lu_solver rest on the LU object", "[S7][S7-R1]" ) +{ + s7::check_legacy< double >(); + s7::check_legacy< std::complex< double > >(); + s7::check_legacy< float >(); + s7::check_legacy< std::complex< float > >(); +} + +namespace s7 +{ + // the callable aborts (SIGABRT) with exactly one contract-violation line on stderr naming `expected` + template< typename Callable > + void require_one_message_death( Callable&& fn, std::string const& expected ) + { + s2_death::outcome const out = s2_death::run( std::forward< Callable >( fn ) ); + INFO( "child stderr: " << out.err ); + REQUIRE( out.signaled ); + REQUIRE( out.signal == SIGABRT ); + REQUIRE( s2_death::line_count( out.err ) == 1 ); + REQUIRE( s2_death::contains( out.err, "contract violation" ) ); + REQUIRE( s2_death::contains( out.err, expected ) ); + } +} + +TEST_CASE( "S7-R1 contract violations abort", "[S7][S7-R1]" ) +{ + feng::matrix< double > const a{ 2, 3, 1.0 }; + feng::matrix< double > const sq = feng::eye< double >( 3, 3 ); + feng::matrix< double > const b{ 2, 1, 1.0 }; + s7::require_one_message_death( [&] { auto f = feng::lu_factor( a ); (void)f; }, "feng::lu_factor: expecting a square matrix" ); + s7::require_one_message_death( [&] { auto d = a.det(); (void)d; }, "matrix::det: expecting a square matrix" ); + s7::require_one_message_death( [&] { auto d = feng::det( a ); (void)d; }, "expecting a square matrix" ); + s7::require_one_message_death( [&] { auto x = a.inverse(); (void)x; }, "matrix::inverse: expecting a square matrix" ); + s7::require_one_message_death( [&] { auto x = feng::inverse( a ); (void)x; }, "expecting a square matrix" ); + s7::require_one_message_death( [&] { feng::matrix< double > out; auto st = feng::inverse( a, out ); (void)st; }, "feng::inverse: expecting a square matrix" ); + s7::require_one_message_death( [&] { auto r = feng::lu_factor( sq ).solve( b ); (void)r; }, "feng::lu_factorization::solve: B must have 3 rows" ); + s7::require_one_message_death( [&] { auto r = feng::solve( sq, b ); (void)r; }, "feng::solve: expecting a square A and B with as many rows" ); + std::complex< float > const one{ 1.0f, 0.0f }; + feng::matrix< std::complex< float > > const ca{ 3, 2, one }; + s7::require_one_message_death( [&] { auto f = feng::lu_factor( ca ); (void)f; }, "feng::lu_factor: expecting a square matrix" ); + s7::require_one_message_death( [&] { auto d = ca.det(); (void)d; }, "matrix::det: expecting a square matrix" ); + s7::require_one_message_death( [&] { auto x = ca.inverse(); (void)x; }, "matrix::inverse: expecting a square matrix" ); +} diff --git a/tests/cases/s7_r2.hpp b/tests/cases/s7_r2.hpp new file mode 100644 index 0000000..b554c62 --- /dev/null +++ b/tests/cases/s7_r2.hpp @@ -0,0 +1,120 @@ +// S7-R2 (PR-10, PR-2): det and inverse from the LU factors (F13). Uses the helpers of tests/cases/s7_r1.hpp. +namespace s7 +{ + template< typename T > + void check_det() + { + using R = real_t< T >; + // block exchange: 4 within 8·n·ε·|4| + auto const bx = from_real< T >( 4, 4, { 0, 0, 1, 2, 0, 0, 3, 4, 5, 6, 0, 0, 7, 8, 0, 0 } ); + REQUIRE( std::abs( bx.det() - T( 4 ) ) <= R( 8 ) * R( 4 ) * eps< T >() * R( 4 ) ); + REQUIRE( std::abs( feng::det( bx ) - T( 4 ) ) <= R( 8 ) * R( 4 ) * eps< T >() * R( 4 ) ); + REQUIRE( std::abs( feng::lu_factor( bx ).det() - T( 4 ) ) <= R( 8 ) * R( 4 ) * eps< T >() * R( 4 ) ); + // exactly 0 when rank < n: [[1..9]], a zero row, a repeated column + for ( auto const& s : { from_real< T >( 3, 3, { 1, 2, 3, 4, 5, 6, 7, 8, 9 } ), + from_real< T >( 3, 3, { 1, 2, 3, 0, 0, 0, 4, 5, 6 } ), + from_real< T >( 3, 3, { 0.1, 2, 0.1, 0.3, 4, 0.3, 0.7, 6, 0.7 } ) } ) + { + REQUIRE( s.det() == T( 0 ) ); + REQUIRE( feng::det( s ) == T( 0 ) ); + } + // 1 for 0×0 (numpy) + REQUIRE( mat< T >{}.det() == T( 1 ) ); + REQUIRE( feng::det( mat< T >{} ) == T( 1 ) ); + // identity, diagonal and row swap are exact + REQUIRE( feng::eye< T >( 6, 6 ).det() == T( 1 ) ); + REQUIRE( from_real< T >( 3, 3, { 2, 0, 0, 0, -3, 0, 0, 0, 0.5 } ).det() == T( -3 ) ); + REQUIRE( from_real< T >( 3, 3, { 0, 1, 0, 1, 0, 0, 0, 0, 1 } ).det() == T( -1 ) ); + } + + template< typename T > + void check_det_product() + { + using R = real_t< T >; + for ( std::size_t n = 1; n <= 12; ++n ) + { + INFO( "n = " << n ); + // random entries in (0, 1) plus n·I: the bound has no κ factor, and the det error of plain random + // (0, 1) fixtures grows with κ (measured up to 272·n·ε at n = 11) + mat< T > const shift = feng::eye< T >( n, n ) * T( R( n ) ); + mat< T > const a = random_square< T >( n, 0x57D0ULL + n ) + shift; + mat< T > const b = random_square< T >( n, 0x57E0ULL + n ) + shift; + T const dab = ( a * b ).det(), da = a.det(), db = b.det(); + REQUIRE( std::abs( dab - da * db ) <= R( 64 ) * R( n ) * eps< T >() * std::abs( da * db ) ); + } + } + + template< typename T > + void check_inverse() + { + using R = real_t< T >; + for ( auto const& [name, a] : r09_inputs< T >() ) + { + INFO( name ); + std::size_t const n = a.row(); + mat< T > const x = a.inverse(); + REQUIRE( x.row() == n ); REQUIRE( x.col() == n ); + R const kappa = norm_inf( a ) * norm_inf( x ); + R const ratio = norm_inf< T >( a * x - feng::eye< T >( n, n ) ) / ( R( n ) * eps< T >() * kappa ); + REQUIRE( ratio <= R( 16 ) ); + REQUIRE( norm_inf< T >( feng::inverse( a ) - x ) == R( 0 ) ); + REQUIRE( norm_inf< T >( feng::inv( a ) - x ) == R( 0 ) ); + auto const t = feng::try_inverse( a ); + REQUIRE( t.ok() ); + REQUIRE( norm_inf< T >( t.value - x ) == R( 0 ) ); + auto const f = feng::lu_factor( a ).inverse(); + REQUIRE( f.ok() ); + REQUIRE( norm_inf< T >( f.value - x ) == R( 0 ) ); + mat< T > out; + REQUIRE( feng::inverse( a, out ) == feng::linalg_status::ok ); + REQUIRE( norm_inf< T >( out - x ) == R( 0 ) ); + } + // singular and nonfinite: statuses, out unchanged, value-returning forms give 0×0 (D-026) + auto nf = feng::eye< T >( 3, 3 ); + nf[0][1] = T( std::numeric_limits< R >::infinity() ); + std::pair< mat< T >, feng::linalg_status > const bad[] = { + { from_real< T >( 3, 3, { 1, 2, 3, 4, 5, 6, 7, 8, 9 } ), feng::linalg_status::singular }, + { from_real< T >( 2, 2, { 0, 0, 0, 0 } ), feng::linalg_status::singular }, + { nf, feng::linalg_status::nonfinite } }; + for ( auto const& [s, st] : bad ) + { + REQUIRE( s.inverse().size() == 0 ); + REQUIRE( s.inverse().row() == 0 ); + REQUIRE( feng::inverse( s ).size() == 0 ); + REQUIRE( feng::inv( s ).size() == 0 ); + auto const t = feng::try_inverse( s ); + REQUIRE( t.status == st ); + REQUIRE( t.value.size() == 0 ); + REQUIRE( feng::lu_factor( s ).inverse().status == st ); + mat< T > out{ 1, 2, T( 3 ) }; + REQUIRE( feng::inverse( s, out ) == st ); + REQUIRE( out.row() == 1 ); REQUIRE( out.col() == 2 ); REQUIRE( out[0][1] == T( 3 ) ); + } + // 0×0 inverts to 0×0 with status ok + REQUIRE( feng::try_inverse( mat< T >{} ).ok() ); + } +} + +TEST_CASE( "S7-R2 det of the block-exchange and singular matrices", "[S7][S7-R2]" ) +{ + s7::check_det< double >(); + s7::check_det< std::complex< double > >(); + s7::check_det< float >(); + s7::check_det< std::complex< float > >(); +} + +TEST_CASE( "S7-R2 det of a product is the product of dets", "[S7][S7-R2]" ) +{ + s7::check_det_product< double >(); + s7::check_det_product< std::complex< double > >(); + s7::check_det_product< float >(); + s7::check_det_product< std::complex< float > >(); +} + +TEST_CASE( "S7-R2 inverse from the factors and its failure statuses", "[S7][S7-R2]" ) +{ + s7::check_inverse< double >(); + s7::check_inverse< std::complex< double > >(); + s7::check_inverse< float >(); + s7::check_inverse< std::complex< float > >(); +} diff --git a/tests/cases/s7_r3.hpp b/tests/cases/s7_r3.hpp new file mode 100644 index 0000000..4916941 --- /dev/null +++ b/tests/cases/s7_r3.hpp @@ -0,0 +1,434 @@ +// S7-R3 (PR-10, PR-2): one-sided Jacobi SVD for real and complex inputs of every shape, and one pseudoinverse +// with a relative cutoff behind pinverse, pinv and svd_inverse (F13, D-026, D-028). Uses the helpers of +// tests/cases/s7_r1.hpp (namespace s7). Set S7_RATIOS=1 to print the worst measured error/tolerance ratios. +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace s7r3 +{ + using s7::mat; + using s7::real_t; + using s7::eps; + + template< typename T > constexpr bool is_cplx = !std::is_same_v< T, real_t< T > >; + + // worst error/tolerance ratio per check, printed by the last case when S7_RATIOS is set + inline double& worst( int i ) + { + static double w[8] = {}; + return w[i]; + } + inline char const* const worst_name[8] = { "reconstruction", "U^H U = I", "V^H V = I", "singular values", + "AXA = A", "XAX = X", "(AX)^H = AX", "(XA)^H = XA" }; + inline void track( int i, double err, double tol ) + { + double const r = tol > 0 ? err / tol : ( err > 0 ? 1e300 : 0.0 ); + worst( i ) = std::max( worst( i ), r ); + } + + template< typename T > + real_t< T > frob( mat< T > const& m ) + { + real_t< T > s{ 0 }; + for ( auto const& x : m ) s += std::norm( x ); + return std::sqrt( s ); + } + + template< typename T > + T cj( T const& x ) + { + if constexpr ( is_cplx< T > ) return std::conj( x ); + else return x; + } + + // conjugate transpose, written out here so the test does not lean on the library under test + template< typename T > + mat< T > ct( mat< T > const& m ) + { + mat< T > r{ m.col(), m.row() }; + for ( std::size_t i = 0; i != m.row(); ++i ) + for ( std::size_t j = 0; j != m.col(); ++j ) r[j][i] = cj( m[i][j] ); + return r; + } + + template< typename T > + mat< T > eye( std::size_t n ) + { + mat< T > r{ n, n }; + std::fill( r.begin(), r.end(), T( 0 ) ); + for ( std::size_t i = 0; i != n; ++i ) r[i][i] = T( 1 ); + return r; + } + + // a unitary n×n matrix: a product of 3·n² seeded Givens rotations (with random phases for complex T) + template< typename T > + mat< T > givens_unitary( std::size_t n, std::uint_least64_t seed ) + { + using R = real_t< T >; + mat< T > q = eye< T >( n ); + if ( n < 2 ) return q; + std::mt19937_64 g{ seed }; + std::size_t const count = 3 * n * n; + auto const draws = feng::random< double >( count, 4, g ); + R const two_pi = R( 6.283185307179586476925286766559 ); + for ( std::size_t k = 0; k != count; ++k ) + { + std::size_t const i = static_cast< std::size_t >( draws[k][0] * n ) % n; + std::size_t j = static_cast< std::size_t >( draws[k][1] * ( n - 1 ) ) % ( n - 1 ); + if ( j >= i ) ++j; + R const th = two_pi * R( draws[k][2] ); + R const c = std::cos( th ), s = std::sin( th ); + T ph = T( 1 ); + if constexpr ( is_cplx< T > ) ph = std::polar( R( 1 ), two_pi * R( draws[k][3] ) ); + for ( std::size_t r = 0; r != n; ++r ) // q ← q·G on columns i, j + { + T const x = q[r][i], y = q[r][j]; + q[r][i] = c * x - s * ph * y; + q[r][j] = s * cj( ph ) * x + c * y; + } + } + return q; + } + + // Q1·Σ·Q2ᴴ with the chosen singular values on the diagonal of the m×n Σ + template< typename T > + mat< T > with_singular_values( std::size_t m, std::size_t n, std::vector< double > const& sigma, std::uint_least64_t seed ) + { + mat< T > d{ m, n }; + std::fill( d.begin(), d.end(), T( 0 ) ); + for ( std::size_t i = 0; i != sigma.size(); ++i ) d[i][i] = T( real_t< T >( sigma[i] ) ); + return givens_unitary< T >( m, seed ) * d * ct( givens_unitary< T >( n, seed + 1 ) ); + } + + template< typename T > + mat< T > seeded_random( std::size_t r, std::size_t c, std::uint_least64_t seed ) + { + std::mt19937_64 g{ seed }; + return feng::random< T >( r, c, g ); + } + + template< typename T > + struct fixture + { + std::string name; + mat< T > a; + std::vector< double > sigma; // the known singular values, descending; empty when not known + std::size_t rank; + }; + + template< typename T > + std::vector< fixture< T > > fixtures() + { + std::vector< fixture< T > > v; + std::vector< double > const sq{ 4, 2.5, 1, 0.5, 0.125 }; + std::vector< double > const tw{ 3, 2, 1, 0.25 }; + v.push_back( { "square 5x5", with_singular_values< T >( 5, 5, sq, 0x5731ULL ), sq, 5 } ); + v.push_back( { "tall 7x4", with_singular_values< T >( 7, 4, tw, 0x5732ULL ), tw, 4 } ); + v.push_back( { "wide 4x7", with_singular_values< T >( 4, 7, tw, 0x5733ULL ), tw, 4 } ); + v.push_back( { "rank deficient 6x3 * 3x5", seeded_random< T >( 6, 3, 0x5734ULL ) * seeded_random< T >( 3, 5, 0x5735ULL ), {}, 3 } ); + // diagonal with signs (and a phase for complex): singular values are the sorted |d_ii| + mat< T > d{ 4, 4 }; + std::fill( d.begin(), d.end(), T( 0 ) ); + d[0][0] = T( -2 ); d[1][1] = T( 5 ); d[2][2] = T( 0.5 ); d[3][3] = T( 3 ); + if constexpr ( is_cplx< T > ) d[2][2] = T( 0.3, -0.4 ); + v.push_back( { "diagonal 4x4", d, { 5, 3, 2, 0.5 }, 4 } ); + // an exact zero column and an exact zero singular value + mat< T > z{ 3, 3 }; + std::fill( z.begin(), z.end(), T( 0 ) ); + z[0][0] = T( 3 ); z[2][2] = T( 1 ); + v.push_back( { "zero column 3x3", z, { 3, 1, 0 }, 2 } ); + v.push_back( { "rank deficient wide 3x6", with_singular_values< T >( 3, 6, { 2, 1, 0 }, 0x5736ULL ), {}, 2 } ); + return v; + } + + // the fixtures at T's precision: float and complex inputs are built in double and rounded once, so a + // rank-deficient input carries only the ε/2 rounding noise, well under the p·ε·s_1 cutoff (D-028), instead of + // the Givens-product and matrix-product noise accumulated in float + template< typename T > + std::vector< fixture< T > > fixtures_at() + { + if constexpr ( sizeof( real_t< T > ) >= sizeof( double ) ) return fixtures< T >(); + else + { + using W = std::conditional_t< is_cplx< T >, std::complex< double >, double >; + std::vector< fixture< T > > v; + for ( auto const& fx : fixtures< W >() ) v.push_back( { fx.name, fx.a.template astype< T >(), fx.sigma, fx.rank } ); + return v; + } + } + + template< typename T > + void check_factorization( fixture< T > const& fx ) + { + using R = real_t< T >; + INFO( fx.name ); + mat< T > const& a = fx.a; + std::size_t const m = a.row(), n = a.col(), k = std::min( m, n ), p = std::max( m, n ); + auto const f = feng::svd_factor( a ); + REQUIRE( f.status() == feng::linalg_status::ok ); + REQUIRE( f.sweeps() >= 1 ); + REQUIRE( f.u().row() == m ); REQUIRE( f.u().col() == k ); + REQUIRE( f.v().row() == n ); REQUIRE( f.v().col() == k ); + auto const& s = f.s(); + REQUIRE( s.size() == k ); + for ( std::size_t i = 0; i != k; ++i ) + { + REQUIRE( s[i] >= R( 0 ) ); + if ( i ) REQUIRE( s[i] <= s[i - 1] ); + } + mat< T > sd{ k, k }; + std::fill( sd.begin(), sd.end(), T( 0 ) ); + for ( std::size_t i = 0; i != k; ++i ) sd[i][i] = T( s[i] ); + R const pe = R( p ) * eps< T >(); + R const rec = frob< T >( a - f.u() * sd * ct( f.v() ) ); + R const rec_tol = R( 8 ) * pe * frob( a ); + track( 0, rec, rec_tol ); + REQUIRE( rec <= rec_tol ); + R const ou = frob< T >( ct( f.u() ) * f.u() - eye< T >( k ) ); + R const ov = frob< T >( ct( f.v() ) * f.v() - eye< T >( k ) ); + track( 1, ou, R( 8 ) * pe ); + track( 2, ov, R( 8 ) * pe ); + REQUIRE( ou <= R( 8 ) * pe ); + REQUIRE( ov <= R( 8 ) * pe ); + if ( !fx.sigma.empty() ) + for ( std::size_t i = 0; i != k; ++i ) + { + R const err = std::abs( s[i] - R( fx.sigma[i] ) ); + R const tol = R( 8 ) * pe * R( fx.sigma[0] ); + track( 3, err, tol ); + REQUIRE( err <= tol ); + } + REQUIRE( f.rank() == fx.rank ); + } + + // the four Moore–Penrose identities with tolerance 32·p·ε·κ·scale, κ = s_1/s_r over the kept values + template< typename T > + void check_moore_penrose( mat< T > const& a, mat< T > const& x, std::vector< real_t< T > > const& s, std::size_t r ) + { + using R = real_t< T >; + std::size_t const p = std::max( a.row(), a.col() ); + R const kappa = r ? s[0] / s[r - 1] : R( 1 ); + R const tol = R( 32 ) * R( p ) * eps< T >() * kappa; + mat< T > const ax = a * x, xa = x * a; + R const e1 = frob< T >( ax * a - a ), t1 = tol * frob( a ); + R const e2 = frob< T >( xa * x - x ), t2 = tol * frob( x ); + R const e3 = frob< T >( ct( ax ) - ax ); + R const e4 = frob< T >( ct( xa ) - xa ); + track( 4, e1, t1 ); track( 5, e2, t2 ); track( 6, e3, tol ); track( 7, e4, tol ); + REQUIRE( e1 <= t1 ); + REQUIRE( e2 <= t2 ); + REQUIRE( e3 <= tol ); + REQUIRE( e4 <= tol ); + } + + template< typename T > + bool same( mat< T > const& x, mat< T > const& y ) + { + return x.row() == y.row() && x.col() == y.col() && std::equal( x.begin(), x.end(), y.begin() ); + } + + template< typename T > + void check_pinverse() + { + for ( auto const& fx : fixtures_at< T >() ) + { + INFO( fx.name ); + mat< T > const& a = fx.a; + mat< T > x; + REQUIRE( feng::pinverse( a, x ) == feng::linalg_status::ok ); + REQUIRE( x.row() == a.col() ); REQUIRE( x.col() == a.row() ); + auto const f = feng::svd_factor( a ); + std::size_t const r = f.rank(); + REQUIRE( r == fx.rank ); + check_moore_penrose( a, x, f.s(), r ); + // one implementation behind every spelling + REQUIRE( same( feng::pinverse( a ), x ) ); + REQUIRE( same( feng::pinv( a ), x ) ); + REQUIRE( same( feng::svd_inverse( a ), x ) ); + REQUIRE( same( f.pinverse().value, x ) ); + } + } + + template< typename T > + void check_cutoff() + { + using R = real_t< T >; + // diagonal: 1e-20 ≤ 3ε·1 is zeroed, the rest inverted + mat< T > d{ 3, 3 }; + std::fill( d.begin(), d.end(), T( 0 ) ); + d[0][0] = T( 1 ); d[1][1] = T( 0.5 ); d[2][2] = T( R( 1e-20 ) ); + mat< T > x; + REQUIRE( feng::pinverse( d, x ) == feng::linalg_status::ok ); + mat< T > want{ 3, 3 }; + std::fill( want.begin(), want.end(), T( 0 ) ); + want[0][0] = T( 1 ); want[1][1] = T( 2 ); + REQUIRE( frob< T >( x - want ) <= R( 4 ) * eps< T >() ); + REQUIRE( feng::svd_factor( d ).rank() == 2 ); + // rtol = 0 keeps it: the cutoff is the relative rule, nothing else + REQUIRE( feng::pinverse( d, x, R( 0 ) ) == feng::linalg_status::ok ); + REQUIRE( std::abs( x[2][2] - T( R( 1e20 ) ) ) <= R( 4 ) * eps< T >() * R( 1e20 ) ); + // a rotated 5×3 with singular values 1, 0.5, 1e-20 + auto const a = with_singular_values< T >( 5, 3, { 1, 0.5, 1e-20 }, 0x5737ULL ); + auto const f = feng::svd_factor( a ); + REQUIRE( f.ok() ); + REQUIRE( f.rank() == 2 ); + REQUIRE( feng::pinverse( a, x ) == feng::linalg_status::ok ); + check_moore_penrose( a, x, f.s(), 2 ); + REQUIRE( std::abs( frob( x ) - std::sqrt( R( 5 ) ) ) <= R( 32 ) * R( 5 ) * eps< T >() * R( 2 ) * std::sqrt( R( 5 ) ) ); + } + + template< typename T > + void check_legacy_shapes() + { + using R = real_t< T >; + for ( auto [m, n] : { std::pair< std::size_t, std::size_t >{ 7, 4 }, { 4, 7 }, { 5, 5 } } ) + { + INFO( m << "x" << n ); + std::size_t const k = std::min( m, n ); + auto const a = seeded_random< T >( m, n, 0x5738ULL + m ); + mat< T > u, w, v; + REQUIRE( feng::singular_value_decomposition( a, u, w, v ) == 0 ); + REQUIRE( u.row() == m ); REQUIRE( u.col() == k ); + REQUIRE( w.row() == k ); REQUIRE( w.col() == k ); + REQUIRE( v.row() == n ); REQUIRE( v.col() == k ); + for ( std::size_t i = 0; i != k; ++i ) + for ( std::size_t j = 0; j != k; ++j ) + if ( i != j ) REQUIRE( w[i][j] == T( 0 ) ); + REQUIRE( frob< T >( a - u * w * ct( v ) ) <= R( 8 ) * R( std::max( m, n ) ) * eps< T >() * frob( a ) ); + auto const t = feng::singular_value_decomposition( a ); + REQUIRE( t.has_value() ); + REQUIRE( same( std::get< 0 >( *t ), u ) ); + REQUIRE( same( std::get< 1 >( *t ), w ) ); + REQUIRE( same( std::get< 2 >( *t ), v ) ); + auto const t2 = feng::svd( a ); + REQUIRE( t2.has_value() ); + REQUIRE( same( std::get< 1 >( *t2 ), w ) ); + } + } + + template< typename T > + void check_empty() + { + for ( auto [m, n] : { std::pair< std::size_t, std::size_t >{ 0, 3 }, { 3, 0 }, { 0, 0 } } ) + { + INFO( m << "x" << n ); + mat< T > const a{ m, n }; + auto const f = feng::svd_factor( a ); + REQUIRE( f.status() == feng::linalg_status::ok ); + REQUIRE( f.s().empty() ); + REQUIRE( f.u().row() == m ); REQUIRE( f.u().col() == 0 ); + REQUIRE( f.v().row() == n ); REQUIRE( f.v().col() == 0 ); + REQUIRE( f.rank() == 0 ); + mat< T > x; + REQUIRE( feng::pinverse( a, x ) == feng::linalg_status::ok ); + REQUIRE( x.row() == n ); REQUIRE( x.col() == m ); + mat< T > u, w, v; + REQUIRE( feng::singular_value_decomposition( a, u, w, v ) == 0 ); + REQUIRE( u.row() == m ); REQUIRE( w.row() == 0 ); REQUIRE( v.row() == n ); + } + } + + template< typename T > + void check_nonconvergence() + { + auto const a = seeded_random< T >( 8, 8, 0x5739ULL ); + auto const f = feng::svd_factor( a, 1 ); + REQUIRE( f.status() == feng::linalg_status::not_converged ); + REQUIRE( f.sweeps() == 1 ); + auto const r = f.pinverse(); + REQUIRE( r.status == feng::linalg_status::not_converged ); + REQUIRE( r.value.size() == 0 ); + // legacy: 1 and the outputs untouched + mat< T > u{ 2, 2, T( 7 ) }, w{ 1, 3, T( 8 ) }, v{ 3, 1, T( 9 ) }; + mat< T > const u0 = u, w0 = w, v0 = v; + REQUIRE( feng::singular_value_decomposition( a, u, w, v, 1 ) == 1 ); + REQUIRE( same( u, u0 ) ); REQUIRE( same( w, w0 ) ); REQUIRE( same( v, v0 ) ); + // the default sweep limit converges + auto const g = feng::svd_factor( a ); + REQUIRE( g.ok() ); + REQUIRE( g.sweeps() > 1 ); + // a nonfinite input: status, never NaN in an output + mat< T > bad = a; + bad[3][4] = T( std::numeric_limits< real_t< T > >::quiet_NaN() ); + REQUIRE( feng::svd_factor( bad ).status() == feng::linalg_status::nonfinite ); + mat< T > x{ 2, 2, T( 1 ) }; + mat< T > const x0 = x; + REQUIRE( feng::pinverse( bad, x ) == feng::linalg_status::nonfinite ); + REQUIRE( same( x, x0 ) ); + REQUIRE( feng::pinv( bad ).size() == 0 ); + REQUIRE( feng::svd_inverse( bad ).size() == 0 ); + REQUIRE( !feng::svd( bad ).has_value() ); + REQUIRE( feng::singular_value_decomposition( bad, u, w, v ) == 1 ); + REQUIRE( same( u, u0 ) ); + } +} + +TEST_CASE( "S7-R3 svd factors reconstruct with orthonormal U and V and the known singular values", "[S7][S7-R3]" ) +{ + for ( auto const& fx : s7r3::fixtures< double >() ) s7r3::check_factorization( fx ); + for ( auto const& fx : s7r3::fixtures< std::complex< double > >() ) s7r3::check_factorization( fx ); +} + +TEST_CASE( "S7-R3 svd of tall, wide and rank-deficient inputs", "[S7][S7-R3]" ) +{ + // square, tall, wide, diagonal, zero-column and rank-deficient (tall product and wide) inputs in every element + // type; ε is that of the element's real type, ᴴ the conjugate transpose + auto const run = []< typename T >( char const* name, T* ) + { + double saved[8]; + for ( int i = 0; i != 8; ++i ) { saved[i] = s7r3::worst( i ); s7r3::worst( i ) = 0; } + for ( auto const& fx : s7r3::fixtures_at< T >() ) s7r3::check_factorization( fx ); + s7r3::check_pinverse< T >(); + for ( int i = 0; i != 8; ++i ) + { + if ( std::getenv( "S7_RATIOS" ) ) std::printf( "S7-R3 %s worst %s: %.3g of tolerance\n", name, s7r3::worst_name[i], s7r3::worst( i ) ); + s7r3::worst( i ) = std::max( saved[i], s7r3::worst( i ) ); + } + }; + run( "float", static_cast< float* >( nullptr ) ); + run( "double", static_cast< double* >( nullptr ) ); + run( "complex", static_cast< std::complex< float >* >( nullptr ) ); + run( "complex", static_cast< std::complex< double >* >( nullptr ) ); +} + +TEST_CASE( "S7-R3 pseudoinverse meets the Moore-Penrose identities", "[S7][S7-R3]" ) +{ + s7r3::check_pinverse< double >(); + s7r3::check_pinverse< std::complex< double > >(); +} + +TEST_CASE( "S7-R3 pseudoinverse zeroes singular values below the relative cutoff", "[S7][S7-R3]" ) +{ + s7r3::check_cutoff< double >(); + s7r3::check_cutoff< std::complex< double > >(); +} + +TEST_CASE( "S7-R3 legacy singular_value_decomposition returns thin u, diagonal w and v", "[S7][S7-R3]" ) +{ + s7r3::check_legacy_shapes< double >(); + s7r3::check_legacy_shapes< std::complex< double > >(); +} + +TEST_CASE( "S7-R3 svd and pinverse accept 0xn and mx0 inputs", "[S7][S7-R3]" ) +{ + s7r3::check_empty< double >(); + s7r3::check_empty< std::complex< double > >(); +} + +TEST_CASE( "S7-R3 svd reports nonconvergence", "[S7][S7-R3]" ) +{ + s7r3::check_nonconvergence< double >(); + s7r3::check_nonconvergence< std::complex< double > >(); + if ( std::getenv( "S7_RATIOS" ) ) + for ( int i = 0; i != 8; ++i ) std::printf( "S7-R3 worst %s: %.3g of tolerance\n", s7r3::worst_name[i], s7r3::worst( i ) ); +} diff --git a/tests/cases/s7_r4.hpp b/tests/cases/s7_r4.hpp new file mode 100644 index 0000000..99d5971 --- /dev/null +++ b/tests/cases/s7_r4.hpp @@ -0,0 +1,301 @@ +// S7-R4 (PR-10, PR-2): Cholesky with a status and RREF with independent pivot rows and columns (F13, F15). +// Uses the helpers of tests/cases/s7_r1.hpp (namespace s7). The exact RREF references below were derived with +// python3 fractions.Fraction (Gaussian rationals for the complex fixtures) and pasted as literals. +#include +#include +#include +#include +#include +#include + +namespace s7r4 +{ + using s7::mat; + using s7::real_t; + using cd = std::complex< double >; + + template< typename T > + mat< T > from_list( std::size_t r, std::size_t c, std::vector< T > const& v ) + { + mat< T > m{ r, c }; + for ( std::size_t i = 0; i != r * c; ++i ) m[i / c][i % c] = v[i]; + return m; + } + + template< typename T > + T conj_( T const& x ) + { + if constexpr ( std::is_same_v< T, real_t< T > > ) return x; + else return std::conj( x ); + } + + template< typename T > + real_t< T > frob( mat< T > const& m ) + { + real_t< T > s{ 0 }; + for ( auto const& x : m ) s += std::norm( x ); + return std::sqrt( s ); + } + + template< typename T > + bool any_nan( mat< T > const& m ) + { + for ( auto const& x : m ) + if ( std::isnan( std::real( x ) ) || std::isnan( std::imag( x ) ) ) return true; + return false; + } + + struct rref_case + { + std::size_t m, n; + std::vector< double > a, r; + std::vector< std::size_t > pivots; + }; + + inline std::vector< rref_case > real_cases() + { + return { + // square, full rank + { 3, 3, { 2, 1, 1, 1, 3, 2, 1, 0, 0 }, { 1, 0, 0, 0, 1, 0, 0, 0, 1 }, { 0, 1, 2 } }, + // square, rank 2 + { 3, 3, { 1, 2, 3, 4, 5, 6, 7, 8, 9 }, { 1, 0, -1, 0, 1, 2, 0, 0, 0 }, { 0, 1 } }, + // square 4×4, column 1 = 2·column 0 + { 4, 4, { 2, 4, 1, 3, 1, 2, 0, 1, 3, 6, 1, 4, 0, 0, 2, 5 }, { 1, 2, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0 }, { 0, 2, 3 } }, + // tall, column 1 = 2·column 0 + { 4, 3, { 1, 2, 1, 2, 4, 0, 3, 6, 1, 1, 2, 2 }, { 1, 2, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0 }, { 0, 2 } }, + // tall, full column rank + { 3, 2, { 1, 2, 3, 4, 5, 7 }, { 1, 0, 0, 1, 0, 0 }, { 0, 1 } }, + // wide with a zero column 0 and column 2 = 2·column 1, rank 2 + { 3, 5, { 0, 1, 2, 1, 3, 0, 2, 4, 1, 1, 0, 3, 6, 2, 4 }, { 0, 1, 2, 0, -2, 0, 0, 0, 1, 5, 0, 0, 0, 0, 0 }, { 1, 3 } }, + // wide, column 2 = 1.6·column 0 − 0.2·column 1 + { 3, 4, { 2, 1, 3, 1, 1, 3, 1, 2, 3, 4, 4, 5 }, { 1, 0, 1.6, 0, 0, 1, -0.2, 0, 0, 0, 0, 1 }, { 0, 1, 3 } }, + // wide, full row rank + { 2, 3, { 1, 2, 3, 4, 5, 6 }, { 1, 0, -1, 0, 1, 2 }, { 0, 1 } }, + }; + } + + template< typename T > + void check_rref( mat< T > const& a, mat< T > const& ref, std::vector< std::size_t > const& pivots ) + { + auto const [m, n] = a.shape(); + real_t< T > const p = static_cast< real_t< T > >( m > n ? m : n ); + real_t< T > const tol = 64 * p * s7::eps< T >() * ( real_t< T >( 1 ) + s7::norm_inf( a ) ); + auto const res = feng::row_echelon( a ); + REQUIRE( res.status == feng::linalg_status::ok ); + REQUIRE( res.rank == pivots.size() ); + REQUIRE( res.pivot_columns == pivots ); + REQUIRE( res.r.row() == m ); + REQUIRE( res.r.col() == n ); + for ( std::size_t i = 0; i != m; ++i ) + for ( std::size_t j = 0; j != n; ++j ) + CHECK( std::abs( res.r[i][j] - ref[i][j] ) <= tol ); + // pivot entries exactly 1, every other entry of a pivot column exactly 0, rows past the rank exactly 0 + for ( std::size_t k = 0; k != pivots.size(); ++k ) + for ( std::size_t i = 0; i != m; ++i ) + CHECK( res.r[i][pivots[k]] == ( i == k ? T( 1 ) : T( 0 ) ) ); + for ( std::size_t i = pivots.size(); i != m; ++i ) + for ( std::size_t j = 0; j != n; ++j ) + CHECK( res.r[i][j] == T( 0 ) ); + // the legacy spellings forward to row_echelon for every shape + auto const r1 = feng::rref( a ); + auto const r2 = feng::gauss_jordan_elimination( a ); + REQUIRE( r1.has_value() ); + REQUIRE( r2.has_value() ); + CHECK( *r1 == res.r ); + CHECK( *r2 == res.r ); + } + + template< typename T > + void check_rref_real_cases() + { + for ( auto const& c : real_cases() ) + check_rref< T >( s7::from_real< T >( c.m, c.n, c.a ), s7::from_real< T >( c.m, c.n, c.r ), c.pivots ); + } +} // namespace s7r4 + +namespace s7r4 +{ + // the complex fixtures as complex literals, rounded to T (every entry is exact in float) + template< typename T > + mat< T > cplx( std::size_t r, std::size_t c, std::vector< cd > const& v ) + { + return from_list< cd >( r, c, v ).template astype< T >(); + } + + template< typename T > + void check_rref_complex_cases() + { + // complex 3×3 of rank 2: row 2 = row 0 + row 1 + check_rref< T >( cplx< T >( 3, 3, { { 1, 1 }, { 2, 0 }, { 0, 1 }, { 0, 2 }, { 2, -2 }, { 1, 0 }, { 1, 3 }, { 4, -2 }, { 1, 1 } } ), + cplx< T >( 3, 3, { 1, 0, { -0.25, 0.25 }, 0, 1, { 0.25, 0.5 }, 0, 0, 0 } ), { 0, 1 } ); + // complex wide with a zero column and a dependent column + check_rref< T >( cplx< T >( 2, 4, { 0, { 1, 1 }, { 2, 0 }, { 3, -1 }, 0, { 2, 2 }, { 4, 0 }, { 1, 0 } } ), + cplx< T >( 2, 4, { 0, 1, { 1, -1 }, 0, 0, 0, 0, 1 } ), { 1, 3 } ); + } + + template< typename T > + void check_rref_degenerate() + { + // the zero matrix: rank 0, unchanged; a 0×0 input + mat< T > const z{ 2, 3, T( 0 ) }; + auto const rz = feng::row_echelon( z ); + CHECK( rz.rank == 0 ); + CHECK( rz.pivot_columns.empty() ); + CHECK( rz.r == z ); + CHECK( feng::row_echelon( mat< T >{} ).rank == 0 ); + + // nonfinite input: nullopt from the legacy spellings, status nonfinite from row_echelon + mat< T > bad = s7::from_real< T >( 2, 2, { 1, 2, 3, 4 } ); + bad[1][0] = T( std::numeric_limits< real_t< T > >::quiet_NaN() ); + CHECK( feng::row_echelon( bad ).status == feng::linalg_status::nonfinite ); + CHECK_FALSE( feng::rref( bad ).has_value() ); + CHECK_FALSE( feng::gauss_jordan_elimination( bad ).has_value() ); + } +} // namespace s7r4 + +TEST_CASE( "S7-R4 rref matches exact references on every shape", "[S7][S7-R4]" ) +{ + // square, tall, wide and rank-deficient fixtures in float, double, complex and complex + s7r4::check_rref_real_cases< float >(); + s7r4::check_rref_real_cases< double >(); + s7r4::check_rref_real_cases< std::complex< float > >(); + s7r4::check_rref_real_cases< std::complex< double > >(); + s7r4::check_rref_complex_cases< std::complex< float > >(); + s7r4::check_rref_complex_cases< std::complex< double > >(); + s7r4::check_rref_degenerate< float >(); + s7r4::check_rref_degenerate< double >(); + s7r4::check_rref_degenerate< std::complex< float > >(); + s7r4::check_rref_degenerate< std::complex< double > >(); +} + +namespace s7r4 +{ + template< typename T > + void check_cholesky_spd( mat< T > const& a ) + { + std::size_t const n = a.row(); + auto const f = feng::cholesky_factor( a ); + REQUIRE( f.status() == feng::linalg_status::ok ); + REQUIRE( f.ok() ); + mat< T > const& l = f.l(); + REQUIRE( l.row() == n ); + REQUIRE( l.col() == n ); + mat< T > llh{ n, n }; + for ( std::size_t i = 0; i != n; ++i ) + { + CHECK( std::imag( l[i][i] ) == 0 ); + CHECK( std::real( l[i][i] ) > 0 ); + for ( std::size_t j = i + 1; j != n; ++j ) CHECK( l[i][j] == T( 0 ) ); + for ( std::size_t j = 0; j != n; ++j ) + { + T s{ 0 }; + for ( std::size_t k = 0; k != n; ++k ) s += l[i][k] * conj_( l[j][k] ); + llh[i][j] = s; + } + } + mat< T > d{ n, n }; + for ( std::size_t i = 0; i != n; ++i ) + for ( std::size_t j = 0; j != n; ++j ) d[i][j] = llh[i][j] - a[i][j]; + CHECK( frob( d ) <= 8 * real_t< T >( n ) * s7::eps< T >() * frob( a ) ); + + mat< T > out; + CHECK( feng::cholesky_decomposition( a, out ) == 0 ); + CHECK( out == l ); + } + + template< typename T > + void check_cholesky_fails( mat< T > const& a ) + { + auto const f = feng::cholesky_factor( a ); + CHECK( f.status() == feng::linalg_status::not_positive_definite ); + CHECK_FALSE( f.ok() ); + CHECK_FALSE( any_nan( f.l() ) ); + mat< T > out = s7::from_real< T >( 1, 2, { 7, 8 } ); + mat< T > const keep = out; + CHECK( feng::cholesky_decomposition( a, out ) == 1 ); + CHECK( out == keep ); + } + + template< typename T > + void check_cholesky_real_cases() + { + check_cholesky_spd< T >( s7::from_real< T >( 3, 3, { 4, 12, -16, 12, 37, -43, -16, -43, 98 } ) ); + check_cholesky_spd< T >( s7::from_real< T >( 1, 1, { 2 } ) ); + // random SPD: Bᵀ·B + n·I + mat< T > const b = s7::random_square< T >( 6, 77 ); + mat< T > spd{ 6, 6 }; + for ( std::size_t i = 0; i != 6; ++i ) + for ( std::size_t j = 0; j != 6; ++j ) + { + T s = i == j ? T( 6 ) : T( 0 ); + for ( std::size_t k = 0; k != 6; ++k ) s += conj_( b[k][i] ) * b[k][j]; + spd[i][j] = s; + } + check_cholesky_spd< T >( spd ); + + check_cholesky_fails< T >( s7::from_real< T >( 2, 2, { 2, 1, 0, 2 } ) ); // not symmetric + check_cholesky_fails< T >( s7::from_real< T >( 2, 2, { 1, 2, 2, 1 } ) ); // indefinite + check_cholesky_fails< T >( s7::from_real< T >( 2, 2, { -1, 0, 0, -1 } ) ); // negative definite + check_cholesky_fails< T >( s7::from_real< T >( 2, 2, { 1, 1, 1, 1 } ) ); // singular + check_cholesky_fails< T >( s7::from_real< T >( 3, 3, { 1, 2, 3, 2, 4, 6, 3, 6, 9 } ) ); // singular, rank 1 + mat< T > nan_a = s7::from_real< T >( 2, 2, { 4, 1, 1, 3 } ); + nan_a[1][1] = T( std::numeric_limits< real_t< T > >::quiet_NaN() ); + check_cholesky_fails< T >( nan_a ); // nonfinite + mat< T > inf_a = s7::from_real< T >( 2, 2, { 4, 1, 1, 3 } ); + inf_a[0][0] = T( std::numeric_limits< real_t< T > >::infinity() ); + check_cholesky_fails< T >( inf_a ); + } +} // namespace s7r4 + +namespace s7r4 +{ + template< typename T > + void check_cholesky_complex_cases() + { + // Hermitian positive definite + check_cholesky_spd< T >( cplx< T >( 3, 3, { 4, { 1, 2 }, { 0, -1 }, { 1, -2 }, 10, { 2, 1 }, { 0, 1 }, { 2, -1 }, 6 } ) ); + // complex symmetric but not Hermitian, and a diagonal with an imaginary part + check_cholesky_fails< T >( cplx< T >( 2, 2, { 4, { 1, 2 }, { 1, 2 }, 5 } ) ); + check_cholesky_fails< T >( cplx< T >( 2, 2, { { 4, 1 }, 0, 0, 5 } ) ); + } +} // namespace s7r4 + +TEST_CASE( "S7-R4 cholesky factors SPD and HPD inputs", "[S7][S7-R4]" ) +{ + s7r4::check_cholesky_real_cases< float >(); + s7r4::check_cholesky_real_cases< double >(); + s7r4::check_cholesky_real_cases< std::complex< float > >(); + s7r4::check_cholesky_real_cases< std::complex< double > >(); + s7r4::check_cholesky_complex_cases< std::complex< float > >(); + s7r4::check_cholesky_complex_cases< std::complex< double > >(); + // 0×0 is ok and empty + auto const f0 = feng::cholesky_factor( feng::matrix< double >{} ); + CHECK( f0.ok() ); + CHECK( f0.l().size() == 0 ); + auto const f0f = feng::cholesky_factor( feng::matrix< float >{} ); + CHECK( f0f.ok() ); + CHECK( f0f.l().size() == 0 ); +} + +namespace s7r4 +{ + template< typename T > + void check_cholesky_failure_classes() + { + // the four failure classes of S7-R4, each with no NaN in l() and the legacy destination unchanged + check_cholesky_fails< T >( s7::from_real< T >( 3, 3, { 2, 1, 0, 0, 2, 1, 0, 0, 2 } ) ); + check_cholesky_fails< T >( s7::from_real< T >( 3, 3, { 1, 0, 0, 0, -1, 0, 0, 0, 1 } ) ); + check_cholesky_fails< T >( s7::from_real< T >( 3, 3, { 1, 1, 0, 1, 1, 0, 0, 0, 1 } ) ); + mat< T > a = feng::eye< T >( 3, 3 ); + a[2][1] = a[1][2] = T( std::numeric_limits< real_t< T > >::infinity() ); + check_cholesky_fails< T >( a ); + } +} // namespace s7r4 + +TEST_CASE( "S7-R4 cholesky reports a non-SPD input", "[S7][S7-R4]" ) +{ + s7r4::check_cholesky_failure_classes< float >(); + s7r4::check_cholesky_failure_classes< double >(); + s7r4::check_cholesky_failure_classes< std::complex< float > >(); + s7r4::check_cholesky_failure_classes< std::complex< double > >(); +} diff --git a/tests/cases/s7_r5.hpp b/tests/cases/s7_r5.hpp new file mode 100644 index 0000000..74cb1f1 --- /dev/null +++ b/tests/cases/s7_r5.hpp @@ -0,0 +1,332 @@ +// S7-R5 (PR-10, PR-2): qualified expm, cgs and bicgstab (D-029). Uses the helpers of tests/cases/s7_r1.hpp +// (namespace s7) and tests/cases/s7_r4.hpp (namespace s7r4). The scipy references were computed once with +// scipy.linalg.expm (scipy 1.18.1) and pasted as literals; they are not regenerated by the suite. +#include +#include +#include +#include +#include +#include + +namespace s7r5 +{ + using s7::mat; + using s7::real_t; + using cd = std::complex< double >; + + template< typename T > + bool all_finite( mat< T > const& m ) + { + for ( auto const& x : m ) + if ( !std::isfinite( std::real( x ) ) || !std::isfinite( std::imag( x ) ) ) return false; + return true; + } + + template< typename T > + real_t< T > rel_err( mat< T > const& x, mat< T > const& ref ) + { + mat< T > d{ x.row(), x.col() }; + for ( std::size_t i = 0; i != x.row(); ++i ) + for ( std::size_t j = 0; j != x.col(); ++j ) d[i][j] = x[i][j] - ref[i][j]; + return s7r4::frob( d ) / s7r4::frob( ref ); + } + + template< typename T > + void check_expm_exact() + { + // zero → I exactly, also 0×0 + mat< T > const z{ 4, 4, T( 0 ) }; + CHECK( feng::expm( z ) == feng::eye< T >( 4, 4 ) ); + CHECK( feng::expm( mat< T >{} ).size() == 0 ); + // diagonal → diag(exp(a_ii)) exactly, diag(700) finite + mat< T > d{ 3, 3, T( 0 ) }; + d[0][0] = T( 1.5 ); d[1][1] = T( -2.25 ); d[2][2] = T( 700 ); + mat< T > const e = feng::expm( d ); + for ( std::size_t i = 0; i != 3; ++i ) + for ( std::size_t j = 0; j != 3; ++j ) + CHECK( e[i][j] == ( i == j ? std::exp( d[i][i] ) : T( 0 ) ) ); + CHECK( all_finite( e ) ); + mat< T > d700{ 2, 2, T( 0 ) }; + d700[0][0] = d700[1][1] = T( 700 ); + mat< T > const e700 = feng::expm( d700 ); + CHECK( e700[0][0] == std::exp( T( 700 ) ) ); + CHECK( e700[0][1] == T( 0 ) ); + } + + template< typename T > + void check_expm_extreme() + { + // ‖A‖₁ ≈ 1e10, negative definite Jordan block: exp(A) = e^{-1e10}·[[1, 1], [0, 1]] = 0 in double + mat< T > const a = s7::from_real< T >( 2, 2, { -1e10, 1, 0, -1e10 } ); + mat< T > const ea = feng::expm( a ); + REQUIRE( all_finite( ea ) ); + for ( auto const& x : ea ) CHECK( std::abs( x ) <= 1e-300 ); + // nilpotent with ‖A‖₁ = 1e10: exp(A) = I + A + mat< T > const n = s7::from_real< T >( 2, 2, { 0, 1e10, 0, 0 } ); + mat< T > const en = feng::expm( n ); + REQUIRE( all_finite( en ) ); + CHECK( en[0][0] == T( 1 ) ); + CHECK( en[1][1] == T( 1 ) ); + CHECK( en[1][0] == T( 0 ) ); + CHECK( std::abs( en[0][1] - T( 1e10 ) ) <= 1e10 * 64 * s7::eps< T >() ); + // ‖A‖₁ ≈ 1e300, negative definite (−1e300·I plus a small off-diagonal): exp(A) = 0 in double + mat< T > b = s7::from_real< T >( 3, 3, { -1e300, 1, 0, 0, -1e300, 2, 1, 0, -1e300 } ); + mat< T > const eb = feng::expm( b ); + REQUIRE( all_finite( eb ) ); + for ( auto const& x : eb ) CHECK( std::abs( x ) <= 1e-300 ); + } + + template< typename T > + void check_expm_nonfinite() + { + mat< T > a = s7::from_real< T >( 2, 3, { 1, 2, 3, 4, 5, 6 } ); // only the shape matters for the NaN result + mat< T > sq = s7::from_real< T >( 2, 2, { 1, 2, 3, 4 } ); + sq[0][1] = T( std::numeric_limits< real_t< T > >::quiet_NaN() ); + mat< T > const e = feng::expm( sq ); + REQUIRE( e.row() == 2 ); + REQUIRE( e.col() == 2 ); + for ( auto const& x : e ) CHECK( std::isnan( std::real( x ) ) ); + sq[0][1] = T( std::numeric_limits< real_t< T > >::infinity() ); + for ( auto const& x : feng::expm( sq ) ) CHECK( std::isnan( std::real( x ) ) ); + + mat< T > out = a; + CHECK( feng::expm( sq, out ) == feng::linalg_status::nonfinite ); + CHECK( out == a ); + mat< T > const ok_in = s7::from_real< T >( 2, 2, { 0, 1, 0, 0 } ); + CHECK( feng::expm( ok_in, out ) == feng::linalg_status::ok ); + CHECK( out == s7::from_real< T >( 2, 2, { 1, 1, 0, 1 } ) ); + } + + // complex closed forms: zero, a diagonal with complex entries, a nilpotent block, rotation generators + inline void check_expm_complex_closed_form() + { + double const eps = std::numeric_limits< double >::epsilon(); + // zero → I exactly + CHECK( feng::expm( mat< cd >{ 3, 3, cd( 0 ) } ) == feng::eye< cd >( 3, 3 ) ); + // diagonal with complex entries → diag(exp(a_ii)) exactly + mat< cd > d{ 3, 3, cd( 0 ) }; + d[0][0] = cd( 0.5, 2.0 ); d[1][1] = cd( -1.25, -3.0 ); d[2][2] = cd( 0, 3.141592653589793 ); + mat< cd > const ed = feng::expm( d ); + for ( std::size_t i = 0; i != 3; ++i ) + for ( std::size_t j = 0; j != 3; ++j ) CHECK( ed[i][j] == ( i == j ? std::exp( d[i][i] ) : cd( 0 ) ) ); + // nilpotent with a complex entry: exp(A) = I + A + mat< cd > nil{ 2, 2, cd( 0 ) }; + nil[0][1] = cd( 3, -4 ); + mat< cd > const en = feng::expm( nil ); + CHECK( en[0][0] == cd( 1 ) ); + CHECK( en[1][0] == cd( 0 ) ); + CHECK( en[1][1] == cd( 1 ) ); + CHECK( std::abs( en[0][1] - cd( 3, -4 ) ) <= 64 * eps * 5 ); + // rotation generator θ·[[0, −1], [1, 0]] → [[cos θ, −sin θ], [sin θ, cos θ]], and the unitary generator + // iθ·[[0, 1], [1, 0]] → [[cos θ, i sin θ], [i sin θ, cos θ]]; θ = 7.5 takes one squaring + for ( double const th : { 0.75, 7.5 } ) + { + INFO( "theta " << th ); + double const c = std::cos( th ), sn = std::sin( th ); + mat< cd > const rot = s7r4::from_list< cd >( 2, 2, { 0, -th, th, 0 } ); + mat< cd > const want_rot = s7r4::from_list< cd >( 2, 2, { c, -sn, sn, c } ); + CHECK( rel_err( feng::expm( rot ), want_rot ) <= 64 * 2 * eps ); + mat< cd > const uni = s7r4::from_list< cd >( 2, 2, { 0, cd( 0, th ), cd( 0, th ), 0 } ); + mat< cd > const want_uni = s7r4::from_list< cd >( 2, 2, { c, cd( 0, sn ), cd( 0, sn ), c } ); + CHECK( rel_err( feng::expm( uni ), want_uni ) <= 64 * 2 * eps ); + } + } +} // namespace s7r5 + +TEST_CASE( "S7-R5 expm of zero, diagonal and extreme inputs", "[S7][S7-R5]" ) +{ + s7r5::check_expm_exact< double >(); + s7r5::check_expm_exact< std::complex< double > >(); + s7r5::check_expm_extreme< double >(); + s7r5::check_expm_extreme< std::complex< double > >(); + s7r5::check_expm_nonfinite< double >(); + s7r5::check_expm_nonfinite< std::complex< double > >(); + s7r5::check_expm_complex_closed_form(); +} + +TEST_CASE( "S7-R5 expm matches scipy references", "[S7][S7-R5]" ) +{ + using s7r5::cd; + // numpy default_rng(20261002).standard_normal((4, 4))·1.5, ‖A‖₁ = 5.94 (one squaring) + feng::matrix< double > const a = s7::from_real< double >( 4, 4, { + -0.3286716625089818, 1.3936606226964554, 0.9329334716081147, -0.5252792304594791, + -1.4850480371190857, 1.1163261934688191, -1.2310979653076153, -1.3662397422059085, + -4.039856710035755, 0.17777160587229968, -1.3625704897147035, -0.9038952069773634, + 0.08226996372122795, 1.4618974388691317, -0.6853026895932535, 1.0294510878328045 } ); + feng::matrix< double > const ea = s7::from_real< double >( 4, 4, { + -0.24853487600479163, 0.02684645019835008, 0.1877005400720356, -1.4615747395214558, + -0.027256558343341286, 0.615888790885136, -0.3779660471406582, -1.8867428635129848, + -0.6608926016705973, -1.5528185729439463, 0.28883482828891993, 1.0857401605634964, + 0.502285296362307, 3.4658676324552617, -1.522872575272041, 0.8457065877396357 } ); + CHECK( s7r5::rel_err( feng::expm( a ), ea ) <= 1e-12 ); + CHECK( s7r5::rel_err( feng::expm( a.astype< cd >() ), ea.astype< cd >() ) <= 1e-12 ); + + // ‖B‖₁ = 12 (two squarings) + feng::matrix< double > const b = s7::from_real< double >( 4, 4, { + 2.0, -1.0, 4.0, 0.5, 3.0, -4.0, 1.5, 2.0, -2.0, 1.0, 0.0, 6.0, 5.0, -3.0, 2.0, -1.0 } ); + feng::matrix< double > const eb = s7::from_real< double >( 4, 4, { + 39.564648851459076, -16.323279303838085, 50.582129012167115, 51.159244877194794, + 25.995824630946895, -10.660783809409793, 33.15430503812922, 33.46526275298682, + 28.98727302293373, -12.169515854397675, 37.00243940975122, 37.582347827381284, + 31.417721978674933, -13.006598079333658, 40.14491115851765, 40.57335340586671 } ); + CHECK( s7r5::rel_err( feng::expm( b ), eb ) <= 1e-12 ); + + // complex 3×3 + feng::matrix< cd > const c = s7r4::from_list< cd >( 3, 3, { + { 0.5, 0.25 }, { -1.0, 0.5 }, { 0.25, 0 }, { 0, 0.75 }, { -0.5, 0 }, { 1.0, -0.25 }, { 1.5, 0 }, { 0.125, -0.5 }, { -0.25, 1 } } ); + feng::matrix< cd > const ec = s7r4::from_list< cd >( 3, 3, { + { 1.3712706767213854, 0.15597657831225087 }, { -0.9684029456415137, 0.4506778282412952 }, { -0.24284452567538198, 0.3057490979511099 }, + { 0.6926061294227425, 0.8112756454504476 }, { 0.29043359311825423, -0.3757626571722854 }, { 0.5950516957178404, 0.1462719441204775 }, + { 1.5838751698156879, 0.789176073738423 }, { -0.5724612615097182, -0.16102754172923112 }, { 0.43237367668175875, 0.6762943510856989 } } ); + CHECK( s7r5::rel_err( feng::expm( c ), ec ) <= 1e-12 ); +} + +namespace s7r5 +{ + inline double residual( mat< double > const& a, mat< double > const& x, mat< double > const& b ) + { + double s = 0; + for ( std::size_t i = 0; i != a.row(); ++i ) + { + double r = -b[i][0]; + for ( std::size_t j = 0; j != a.col(); ++j ) r += a[i][j] * x[j][0]; + s += r * r; + } + return std::sqrt( s ); + } + + inline double norm2( mat< double > const& b ) { return s7r4::frob( b ); } + + template< typename Solver > + void check_solver( Solver solve ) + { + double const eps = 1e-10; + // SPD 5×5: Bᵀ·B + 5·I + mat< double > const g = s7::random_square< double >( 5, 2026 ); + mat< double > spd{ 5, 5 }; + for ( std::size_t i = 0; i != 5; ++i ) + for ( std::size_t j = 0; j != 5; ++j ) + { + double s = i == j ? 5.0 : 0.0; + for ( std::size_t k = 0; k != 5; ++k ) s += g[k][i] * g[k][j]; + spd[i][j] = s; + } + mat< double > const b = s7::from_real< double >( 5, 1, { 1, -2, 3, 0.5, -1 } ); + mat< double > x; + CHECK( solve( spd, x, b, 100, eps ) == 0 ); + CHECK( residual( spd, x, b ) <= 10 * eps * norm2( b ) ); + + // relative: a b of size 1e8 converges with the same eps + mat< double > const big = b * 1e8; + mat< double > xb; + CHECK( solve( spd, xb, big, 100, eps ) == 0 ); + CHECK( residual( spd, xb, big ) <= 10 * eps * norm2( big ) ); + + // diagonal + mat< double > dg{ 5, 5, 0.0 }; + for ( std::size_t i = 0; i != 5; ++i ) dg[i][i] = double( i + 1 ); + mat< double > xd; + CHECK( solve( dg, xd, b, 100, eps ) == 0 ); + CHECK( residual( dg, xd, b ) <= 10 * eps * norm2( b ) ); + + // zero b: x = 0 and 0 + mat< double > x0 = s7::from_real< double >( 5, 1, { 1, 2, 3, 4, 5 } ); + CHECK( solve( spd, x0, mat< double >{ 5, 1, 0.0 }, 100, eps ) == 0 ); + CHECK( x0 == mat< double >{ 5, 1, 0.0 } ); + + // NaN in b and in A: 1 + mat< double > bn = b; + bn[2][0] = std::numeric_limits< double >::quiet_NaN(); + mat< double > xn; + CHECK( solve( spd, xn, bn, 100, eps ) == 1 ); + mat< double > an = spd; + an[1][3] = std::numeric_limits< double >::infinity(); + CHECK( solve( an, xn, b, 100, eps ) == 1 ); + } + + template< typename Solver > + void check_solver_exhaustion( Solver solve ) + { + // max_loops = 1 on an unconverged 5×5 system: 1, and x holds the (finite) last iterate + mat< double > a{ 5, 5, 0.0 }; + for ( std::size_t i = 0; i != 5; ++i ) + { + a[i][i] = 4.0; + if ( i + 1 != 5 ) a[i][i + 1] = a[i + 1][i] = -1.0; + } + a[0][4] = 0.5; a[4][0] = 0.5; + mat< double > const b = s7::from_real< double >( 5, 1, { 1, -2, 3, 0.5, -1 } ); + mat< double > x; + CHECK( solve( a, x, b, 1, 1e-14 ) == 1 ); + REQUIRE( x.row() == 5 ); + REQUIRE( x.col() == 1 ); + CHECK( all_finite( x ) ); + CHECK( residual( a, x, b ) > 1e-14 * norm2( b ) ); + // more loops converge from the same start + mat< double > y; + CHECK( solve( a, y, b, 100, 1e-12 ) == 0 ); + CHECK( residual( a, y, b ) <= 10 * 1e-12 * norm2( b ) ); + } + + template< typename Solver > + void check_solver_breakdown( Solver solve ) + { + // breakdown: with A = [[0, 1], [1, 0]], b = (2, 0) and the start x₀ = (0, 1) the first residual is + // r = (1, 0) and r̂ᵀ·A·r = 0, the denominator of α: 1, x the finite last iterate x₀, no NaN + mat< double > const a = s7::from_real< double >( 2, 2, { 0, 1, 1, 0 } ); + mat< double > const b = s7::from_real< double >( 2, 1, { 2, 0 } ); + mat< double > const x0 = s7::from_real< double >( 2, 1, { 0, 1 } ); + mat< double > x = x0; + CHECK( solve( a, x, b, 100, 1e-10 ) == 1 ); + REQUIRE( x.row() == 2 ); + REQUIRE( x.col() == 1 ); + CHECK( all_finite( x ) ); + CHECK( x == x0 ); + + // nonfinite A or b: 1, x unchanged and finite + mat< double > const spd = s7::from_real< double >( 3, 3, { 4, 1, 0, 1, 3, 1, 0, 1, 2 } ); + mat< double > const b3 = s7::from_real< double >( 3, 1, { 1, 2, 3 } ); + mat< double > const start = s7::from_real< double >( 3, 1, { 0.5, -1, 2 } ); + for ( int which = 0; which != 4; ++which ) + { + INFO( "nonfinite case " << which ); + mat< double > an = spd, bn = b3; + if ( which == 0 ) an[2][1] = std::numeric_limits< double >::quiet_NaN(); + if ( which == 1 ) an[0][0] = -std::numeric_limits< double >::infinity(); + if ( which == 2 ) bn[1][0] = std::numeric_limits< double >::infinity(); + if ( which == 3 ) bn[0][0] = std::numeric_limits< double >::quiet_NaN(); + mat< double > xs = start; + CHECK( solve( an, xs, bn, 100, 1e-10 ) == 1 ); + CHECK( xs == start ); + CHECK( all_finite( xs ) ); + } + + // a diagonal system with mixed signs and scales converges to the relative tolerance + mat< double > dg{ 5, 5, 0.0 }; + double const diag[5] = { -2, 3, 0.5, -7, 1e3 }; + for ( std::size_t i = 0; i != 5; ++i ) dg[i][i] = diag[i]; + mat< double > const bd = s7::from_real< double >( 5, 1, { 1, -2, 3, 0.5, -1 } ); + mat< double > xd; + CHECK( solve( dg, xd, bd, 100, 1e-10 ) == 0 ); + CHECK( residual( dg, xd, bd ) <= 10 * 1e-10 * norm2( bd ) ); + } +} // namespace s7r5 + +TEST_CASE( "S7-R5 iterative solvers report exhaustion", "[S7][S7-R5]" ) +{ + using M = feng::matrix< double >; + auto cgs = []( M const& a, M& x, M const& b, std::uint_least64_t l, double e ) { return feng::cgs( a, x, b, l, e ); }; + auto cgs2 = []( M const& a, M& x, M const& b, std::uint_least64_t l, double e ) { return feng::conjugate_gradient_squared( a, x, b, l, e ); }; + auto bi = []( M const& a, M& x, M const& b, std::uint_least64_t l, double e ) { return feng::bicgstab( a, x, b, l, e ); }; + auto bi2 = []( M const& a, M& x, M const& b, std::uint_least64_t l, double e ) { return feng::biconjugate_gradient_stabilized_method( a, x, b, l, e ); }; + s7r5::check_solver_exhaustion( cgs ); + s7r5::check_solver_exhaustion( bi ); + s7r5::check_solver( cgs ); + s7r5::check_solver( cgs2 ); + s7r5::check_solver( bi ); + s7r5::check_solver( bi2 ); + s7r5::check_solver_breakdown( cgs ); + s7r5::check_solver_breakdown( cgs2 ); + s7r5::check_solver_breakdown( bi ); + s7r5::check_solver_breakdown( bi2 ); +} diff --git a/tests/cases/s7_r6.hpp b/tests/cases/s7_r6.hpp new file mode 100644 index 0000000..8112534 --- /dev/null +++ b/tests/cases/s7_r6.hpp @@ -0,0 +1,163 @@ +// S7-R6 (PR-10, PR-14): the committed numpy/scipy fixtures in tests/fixtures/s7/ match the S7 APIs within the +// tolerance recorded on each manifest line. The suite runs from the repo root (AGENTS.md), so the paths are +// relative to it. Error measure: ||got - ref||_F / ||ref||_F, or ||got - ref||_F when ref is 0. +// Prints `ORACLE S7 FIXTURES ` for tools/check.sh oracle. +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace s7r6 +{ + inline std::string const dir = "tests/fixtures/s7/"; + + struct line + { + std::string name, op; + double tol = 0; + std::vector< std::string > files; + }; + + template< typename T > + double frob( feng::matrix< T > const& m ) + { + double s = 0; + for ( auto const& x : m ) s += std::norm( x ); + return std::sqrt( s ); + } + + // relative Frobenius error; +inf on a shape mismatch + template< typename T > + double rel_err( feng::matrix< T > const& got, feng::matrix< T > const& ref ) + { + if ( got.row() != ref.row() || got.col() != ref.col() ) return HUGE_VAL; + feng::matrix< T > d{ ref.row(), ref.col() }; + for ( std::size_t i = 0; i != ref.size(); ++i ) d.data()[i] = got.data()[i] - ref.data()[i]; + double const r = frob( ref ); + return r == 0 ? frob( d ) : frob( d ) / r; + } + + template< typename T > + bool load( std::string const& f, feng::matrix< T >& m ) + { + return m.load_npy( dir + f ); + } + + // computes the op on the loaded inputs; returns the error, or -1 (with why set) when a file or the op failed + template< typename T > + double run( line const& c, std::string& why ) + { + using R = s7::real_t< T >; + std::size_t const want = c.op == "lu_solve" ? 3 : 2; + if ( c.files.size() != want ) { why = "expected " + std::to_string( want ) + " files"; return -1; } + // singular values are real ( + bool const real_ref = c.op == "svdvals"; + std::vector< feng::matrix< T > > in( want ); + feng::matrix< R > sref; + for ( std::size_t i = 0; i != want; ++i ) + { + bool const ok = ( real_ref && i + 1 == want ) ? load( c.files[i], sref ) : load( c.files[i], in[i] ); + if ( !ok ) { why = "cannot load " + dir + c.files[i]; return -1; } + } + auto const& a = in[0]; + auto const& ref = in.back(); + if ( c.op == "lu_solve" ) + { + auto x = feng::solve( a, in[1] ); + if ( !x.ok() ) { why = "solve status not ok"; return -1; } + return rel_err( x.value, ref ); + } + if ( c.op == "det" ) + { + feng::matrix< T > d{ 1, 1 }; + d[0][0] = feng::det( a ); + return rel_err( d, ref ); + } + if ( c.op == "inv" ) + { + auto x = feng::try_inverse( a ); + if ( !x.ok() ) { why = "try_inverse status not ok"; return -1; } + return rel_err( x.value, ref ); + } + if ( c.op == "svdvals" ) + { + auto const f = feng::svd_factor( a ); + if ( !f.ok() ) { why = "svd_factor status not ok"; return -1; } + feng::matrix< R > s{ 1, f.s().size() }; + for ( std::size_t i = 0; i != f.s().size(); ++i ) s[0][i] = f.s()[i]; + return rel_err( s, sref ); + } + if ( c.op == "pinv" ) + { + feng::matrix< T > x; + if ( feng::pinverse( a, x, R( -1 ) ) != feng::linalg_status::ok ) { why = "pinverse status not ok"; return -1; } + return rel_err( x, ref ); + } + if ( c.op == "cholesky" ) + { + auto const f = feng::cholesky_factor( a ); + if ( !f.ok() ) { why = "cholesky_factor status not ok"; return -1; } + return rel_err( f.l(), ref ); + } + if ( c.op == "rref" ) + { + auto const r = feng::row_echelon( a ); + if ( r.status != feng::linalg_status::ok ) { why = "row_echelon status not ok"; return -1; } + return rel_err( r.r, ref ); + } + if ( c.op == "expm" ) + { + feng::matrix< T > e; + if ( feng::expm( a, e ) != feng::linalg_status::ok ) { why = "expm status not ok"; return -1; } + return rel_err( e, ref ); + } + why = "unknown op " + c.op; + return -1; + } +} + +TEST_CASE( "S7-R6 committed fixtures match within their tolerances", "[S7][S7-R6][oracle]" ) +{ + std::ifstream in( s7r6::dir + "manifest.txt" ); + REQUIRE( in.good() ); + std::vector< s7r6::line > cases; + std::string text; + while ( std::getline( in, text ) ) + { + if ( text.empty() || text[0] == '#' ) continue; + std::istringstream is( text ); + s7r6::line c; + std::string f; + is >> c.name >> c.op >> c.tol; + INFO( "manifest line: " << text ); + REQUIRE( !is.fail() ); + while ( is >> f ) c.files.push_back( f ); + cases.push_back( c ); + } + REQUIRE( cases.size() > 0 ); + + std::size_t checked = 0; + for ( auto const& c : cases ) + { + // case names are __ (tests/fixtures/s7/make_fixtures.py) + std::string const type = c.name.substr( c.op.size() + 1, c.name.find( '_', c.op.size() + 1 ) - c.op.size() - 1 ); + std::string why; + double err = -1; + if ( type == "f8" ) err = s7r6::run< double >( c, why ); + else if ( type == "c16" ) err = s7r6::run< std::complex< double > >( c, why ); + else why = "unknown element type '" + type + "'"; + INFO( "case " << c.name << " (" << c.op << "): error " << err << ", tol " << c.tol << ( why.empty() ? "" : ", " ) << why ); + CHECK( why.empty() ); + CHECK( err >= 0 ); + CHECK( err <= c.tol ); + if ( why.empty() && err >= 0 && err <= c.tol ) ++checked; + if ( std::getenv( "S7_RATIOS" ) ) std::cout << c.name << " ratio " << err / c.tol << "\n"; + } + CHECK( checked == cases.size() ); + std::cout << "ORACLE S7 FIXTURES " << checked << "\n"; +} diff --git a/tests/cases/s8_fixtures.hpp b/tests/cases/s8_fixtures.hpp new file mode 100644 index 0000000..3539d5f --- /dev/null +++ b/tests/cases/s8_fixtures.hpp @@ -0,0 +1,199 @@ +// S8 oracle fixtures (PR-11, PR-14): the committed numpy/scipy files in tests/fixtures/s8/ match the S8 APIs +// within the absolute tolerance recorded on each manifest line (max |got - ref| over the elements; 0 is exact). +// The suite runs from the repo root (AGENTS.md), so the paths are relative to it. Prints `ORACLE S8 FIXTURES ` +// for tools/check.sh oracle, one line per TEST_CASE (the lane sums them). Ops are dispatched by name in s8fx::run +// (conv) and s8fx::run_fourier (fft2, ifft2, fftshift, ifftshift); the element type is the case-name field after +// the op: f4, f8, i4, c8 or c16. The result type of fft and ifft is checked against D-031 at compile time, and the +// reference file must load as that type, so numpy's result dtype agrees with D-031 too. +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace s8fx +{ + inline std::string const dir = "tests/fixtures/s8/"; + + struct line + { + std::string name, op; + double tol = 0; + std::vector< std::string > files; + }; + + template< typename T > + std::complex< long double > widen( T const& v ) + { + return { static_cast< long double >( std::real( v ) ), static_cast< long double >( std::imag( v ) ) }; + } + + // max |got - ref| (complex modulus for complex elements); +inf on a shape mismatch + template< typename T > + double max_err( feng::matrix< T > const& got, feng::matrix< T > const& ref ) + { + if ( got.row() != ref.row() || got.col() != ref.col() ) return HUGE_VAL; + long double e = 0; + for ( std::size_t i = 0; i != ref.size(); ++i ) + { + long double const d = std::abs( widen( got.data()[i] ) - widen( ref.data()[i] ) ); + if ( !( d <= e ) ) e = d; // NaN propagates + } + return static_cast< double >( e ); + } + + template< typename T > + bool load( std::string const& f, feng::matrix< T >& m ) + { + return m.load_npy( dir + f ); + } + + // loads `want` files into `in`; false with why set when the count or a load fails + template< typename T > + bool load_all( line const& c, std::size_t want, std::vector< feng::matrix< T > >& in, std::string& why ) + { + if ( c.files.size() != want ) { why = "expected " + std::to_string( want ) + " files"; return false; } + in.resize( want ); + for ( std::size_t i = 0; i != want; ++i ) + if ( !load( c.files[i], in[i] ) ) { why = "cannot load " + dir + c.files[i]; return false; } + return true; + } + + // computes the op on the loaded inputs; returns the error, or -1 (with why set) when a file or the op failed + template< typename T > + double run( line const& c, std::string& why ) + { + std::vector< feng::matrix< T > > in; + if ( c.op == "conv_full" || c.op == "conv_same" || c.op == "conv_valid" ) + { + if ( !load_all( c, 3, in, why ) ) return -1; + std::string const mode = c.op.substr( 5 ); + return max_err( feng::conv( in[0], in[1], mode ), in[2] ); + } + why = "unknown op " + c.op; + return -1; + } + + // D-031: float, double and long double give complex of the same real type, complex stays, integers give + // complex + template< typename T > struct fft_result { using type = std::complex< double >; }; + template<> struct fft_result< float > { using type = std::complex< float >; }; + template<> struct fft_result< long double > { using type = std::complex< long double >; }; + template< typename T > struct fft_result< std::complex< T > > { using type = std::complex< T >; }; + + // fft2 / ifft2: files A C, C = numpy.fft.fft2(A) or ifft2(A) of D-031's type; fftshift / ifftshift: files A C of + // A's type + template< typename T > + double run_fourier( line const& c, std::string& why ) + { + using X = typename fft_result< T >::type; + static_assert( std::is_same_v< decltype( feng::fft( std::declval< feng::matrix< T > const& >() ) ), feng::matrix< X > > ); + static_assert( std::is_same_v< decltype( feng::ifft( std::declval< feng::matrix< T > const& >() ) ), feng::matrix< X > > ); + static_assert( std::is_same_v< decltype( feng::fftshift( std::declval< feng::matrix< T > const& >() ) ), feng::matrix< T > > ); + static_assert( std::is_same_v< decltype( feng::ifftshift( std::declval< feng::matrix< T > const& >() ) ), feng::matrix< T > > ); + if ( c.files.size() != 2 ) { why = "expected 2 files"; return -1; } + feng::matrix< T > a; + if ( !load( c.files[0], a ) ) { why = "cannot load " + dir + c.files[0]; return -1; } + if ( c.op == "fft2" || c.op == "ifft2" ) + { + feng::matrix< X > ref; + if ( !load( c.files[1], ref ) ) { why = "cannot load " + dir + c.files[1] + " as D-031's result type"; return -1; } + return max_err( c.op == "fft2" ? feng::fft( a ) : feng::ifft( a ), ref ); + } + if ( c.op == "fftshift" || c.op == "ifftshift" ) + { + feng::matrix< T > ref; + if ( !load( c.files[1], ref ) ) { why = "cannot load " + dir + c.files[1]; return -1; } + return max_err( c.op == "fftshift" ? feng::fftshift( a ) : feng::ifftshift( a ), ref ); + } + why = "unknown op " + c.op; + return -1; + } + + // the manifest lines whose op is one of `ops` + inline std::vector< line > read_manifest( std::vector< std::string > const& ops ) + { + std::ifstream in( dir + "manifest.txt" ); + REQUIRE( in.good() ); + std::vector< line > cases; + std::string text; + while ( std::getline( in, text ) ) + { + if ( text.empty() || text[0] == '#' ) continue; + std::istringstream is( text ); + line c; + std::string f; + is >> c.name >> c.op >> c.tol; + INFO( "manifest line: " << text ); + REQUIRE( !is.fail() ); + while ( is >> f ) c.files.push_back( f ); + for ( auto const& op : ops ) + if ( c.op == op ) cases.push_back( c ); + } + REQUIRE( cases.size() > 0 ); + return cases; + } + + // case names are __ (tests/fixtures/s8/make_fixtures.py) + inline std::string type_of( line const& c ) + { + std::size_t const at = c.op.size() + 1; + return c.name.substr( at, c.name.find( '_', at ) - at ); + } + + // checks every case with `dispatch( case, type, why )` returning the error; prints the ORACLE line + template< typename Dispatch > + void check_cases( std::vector< line > const& cases, Dispatch dispatch ) + { + std::size_t checked = 0; + for ( auto const& c : cases ) + { + std::string why; + double const err = dispatch( c, type_of( c ), why ); + INFO( "case " << c.name << " (" << c.op << "): error " << err << ", tol " << c.tol << ( why.empty() ? "" : ", " ) << why ); + CHECK( why.empty() ); + CHECK( err >= 0 ); + CHECK( err <= c.tol ); + if ( why.empty() && err >= 0 && err <= c.tol ) ++checked; + } + CHECK( checked == cases.size() ); + std::cout << "ORACLE S8 FIXTURES " << checked << "\n"; + } + + inline double fourier( line const& c, std::string const& type, std::string& why ) + { + if ( type == "f4" ) return run_fourier< float >( c, why ); + if ( type == "f8" ) return run_fourier< double >( c, why ); + if ( type == "i4" ) return run_fourier< int >( c, why ); + if ( type == "c8" ) return run_fourier< std::complex< float > >( c, why ); + if ( type == "c16" ) return run_fourier< std::complex< double > >( c, why ); + why = "unknown element type '" + type + "'"; + return -1; + } +} + +TEST_CASE( "S8-R1 conv matches the committed scipy fixtures", "[S8][S8-R1][oracle]" ) +{ + auto const cases = s8fx::read_manifest( { "conv_full", "conv_same", "conv_valid" } ); + s8fx::check_cases( cases, []( s8fx::line const& c, std::string const& type, std::string& why ) -> double + { + if ( type == "f8" ) return s8fx::run< double >( c, why ); + if ( type == "i4" ) return s8fx::run< int >( c, why ); + why = "unknown element type '" + type + "'"; + return -1; + } ); +} + +TEST_CASE( "S8-R2 fft and ifft match the committed numpy fixtures", "[S8][S8-R2][oracle]" ) +{ + s8fx::check_cases( s8fx::read_manifest( { "fft2", "ifft2" } ), s8fx::fourier ); +} + +TEST_CASE( "S8-R3 shifts match the committed numpy fixtures", "[S8][S8-R3][oracle]" ) +{ + s8fx::check_cases( s8fx::read_manifest( { "fftshift", "ifftshift" } ), s8fx::fourier ); +} diff --git a/tests/cases/s8_r1.hpp b/tests/cases/s8_r1.hpp new file mode 100644 index 0000000..64a30c5 --- /dev/null +++ b/tests/cases/s8_r1.hpp @@ -0,0 +1,259 @@ +// S8-R1 (PR-11, F15): conv and conv2 follow scipy.signal.convolve2d: kernel reversed, full (ra+rb-1)x(ca+cb-1), +// same cropped from full at ((rb-1)/2, (cb-1)/2), valid (ra-rb+1)x(ca-cb+1) or from the swapped operands when B +// contains A, otherwise an abort; an unknown mode aborts (D-011); an operand with a zero dimension gives 0x0 for +// full and valid and zeros of A's shape for same (D-030). Literal values below come from scipy 1.18.1. +#include +#include +#include +#include +#include +#include +#include +#include + +#include "./s2_death.hpp" + +namespace s8r1 +{ + template< typename T > + feng::matrix< T > make( std::size_t r, std::size_t c, std::vector< T > const& v ) + { + feng::matrix< T > m{ r, c }; + for ( std::size_t i = 0; i != r * c; ++i ) m.data()[i] = v[i]; + return m; + } + + template< typename T > + bool equal( feng::matrix< T > const& a, feng::matrix< T > const& b ) + { + if ( a.row() != b.row() || a.col() != b.col() ) return false; + for ( std::size_t i = 0; i != a.size(); ++i ) + if ( !( a.data()[i] == b.data()[i] ) ) return false; + return true; + } + + template< typename T > + bool zero_by_zero( feng::matrix< T > const& m ) { return m.row() == 0 && m.col() == 0; } + + // A (3x4) and B (2x3) of the asymmetric case + template< typename T > + feng::matrix< T > A() { return make< T >( 3, 4, { 1, -2, 3, 0, 4, 5, -1, 2, 0, 3, 2, -4 } ); } + template< typename T > + feng::matrix< T > B() { return make< T >( 2, 3, { 1, 0, -1, 2, 1, 3 } ); } + + template< typename Callable > + void require_one_message_death( Callable&& fn, std::string const& expected ) + { + s2_death::outcome const out = s2_death::run( std::forward< Callable >( fn ) ); + INFO( "child stderr: " << out.err ); + REQUIRE( out.signaled ); + REQUIRE( out.signal == SIGABRT ); + REQUIRE( s2_death::line_count( out.err ) == 1 ); + REQUIRE( s2_death::contains( out.err, "contract violation" ) ); + REQUIRE( s2_death::contains( out.err, expected ) ); + REQUIRE_FALSE( s2_death::contains( out.err, "AddressSanitizer" ) ); + REQUIRE_FALSE( s2_death::contains( out.err, "runtime error:" ) ); + } +} + +TEST_CASE( "S8-R1 conv of [1,2] with [3,4] is [3,10,8]", "[S8][S8-R1]" ) +{ + auto const a = s8r1::make< double >( 1, 2, { 1, 2 } ); + auto const b = s8r1::make< double >( 1, 2, { 3, 4 } ); + auto const want = s8r1::make< double >( 1, 3, { 3, 10, 8 } ); + CHECK( s8r1::equal( feng::conv( a, b ), want ) ); + CHECK( s8r1::equal( feng::conv( a, b, "full" ), want ) ); + CHECK( s8r1::equal( feng::conv2( a, b ), want ) ); + CHECK( s8r1::equal( feng::conv2( a, b, std::string{ "full" } ), want ) ); +} + +TEST_CASE( "S8-R1 a 1x1 kernel k gives k*A in every mode", "[S8][S8-R1]" ) +{ + auto const a = s8r1::A< double >(); + auto const k = s8r1::make< double >( 1, 1, { -3 } ); + feng::matrix< double > want = a; + for ( auto& x : want ) x *= -3; + CHECK( s8r1::equal( feng::conv( a, k ), want ) ); + CHECK( s8r1::equal( feng::conv( a, k, "full" ), want ) ); + CHECK( s8r1::equal( feng::conv( a, k, "same" ), want ) ); + CHECK( s8r1::equal( feng::conv( a, k, "valid" ), want ) ); + // the 1x1 operand first: full is commutative, same keeps the 1x1 shape, valid swaps + CHECK( s8r1::equal( feng::conv( k, a ), want ) ); + CHECK( s8r1::equal( feng::conv( k, a, "same" ), s8r1::make< double >( 1, 1, { -15 } ) ) ); // full[1][1] = -3 * A[1][1] + CHECK( s8r1::equal( feng::conv( k, a, "valid" ), want ) ); +} + +TEST_CASE( "S8-R1 asymmetric 3x4 with a 2x3 kernel matches scipy in every mode", "[S8][S8-R1]" ) +{ + auto const a = s8r1::A< double >(); + auto const b = s8r1::B< double >(); + CHECK( s8r1::equal( feng::conv( a, b ), s8r1::make< double >( 4, 6, { 1, -2, 2, 2, -3, 0, 6, 2, 2, -6, 10, -2, 8, 17, 17, 11, -3, 10, 0, 6, 7, 3, 2, -12 } ) ) ); + CHECK( s8r1::equal( feng::conv( a, b, "same" ), s8r1::make< double >( 3, 4, { -2, 2, 2, -3, 2, 2, -6, 10, 17, 17, 11, -3 } ) ) ); + CHECK( s8r1::equal( feng::conv( a, b, "valid" ), s8r1::make< double >( 2, 2, { 2, -6, 17, 11 } ) ) ); + CHECK( s8r1::equal( feng::conv2( a, b, "valid" ), s8r1::make< double >( 2, 2, { 2, -6, 17, 11 } ) ) ); +} + +TEST_CASE( "S8-R1 same mode with even and oversized kernels follows scipy's alignment", "[S8][S8-R1]" ) +{ + auto const a = s8r1::A< double >(); + auto const even = s8r1::make< double >( 2, 2, { 1, 2, 3, 4 } ); + CHECK( s8r1::equal( feng::conv( a, even, "same" ), s8r1::make< double >( 3, 4, { 1, 0, -1, 6, 7, 11, 10, 12, 12, 34, 25, 2 } ) ) ); + feng::matrix< double > over{ 4, 5 }; + for ( std::size_t i = 0; i != over.size(); ++i ) over.data()[i] = double( i + 1 ); + CHECK( s8r1::equal( feng::conv( a, over, "same" ), s8r1::make< double >( 3, 4, { 33, 45, 57, 34, 91, 114, 127, 80, 166, 179, 192, 120 } ) ) ); +} + +TEST_CASE( "S8-R1 valid mode swaps the operands when B contains A", "[S8][S8-R1]" ) +{ + auto const a = s8r1::A< double >(); + feng::matrix< double > over{ 4, 5 }; + for ( std::size_t i = 0; i != over.size(); ++i ) over.data()[i] = double( i + 1 ); + auto const want = s8r1::make< double >( 2, 2, { 114, 127, 179, 192 } ); + CHECK( s8r1::equal( feng::conv( a, over, "valid" ), want ) ); + CHECK( s8r1::equal( feng::conv( over, a, "valid" ), want ) ); +} + +TEST_CASE( "S8-R1 empty operands follow D-030", "[S8][S8-R1]" ) +{ + auto const a = s8r1::A< double >(); + feng::matrix< double > const e00{ 0, 0 }; + feng::matrix< double > const e03{ 0, 3 }; + feng::matrix< double > const e20{ 2, 0 }; + for ( auto const* e : { &e00, &e03, &e20 } ) + { + CHECK( s8r1::zero_by_zero( feng::conv( a, *e ) ) ); + CHECK( s8r1::zero_by_zero( feng::conv( a, *e, "full" ) ) ); + CHECK( s8r1::zero_by_zero( feng::conv( a, *e, "valid" ) ) ); + CHECK( s8r1::equal( feng::conv( a, *e, "same" ), feng::matrix< double >{ 3, 4, 0.0 } ) ); + CHECK( s8r1::zero_by_zero( feng::conv( *e, a ) ) ); + CHECK( s8r1::zero_by_zero( feng::conv( *e, a, "valid" ) ) ); + CHECK( feng::conv( *e, a, "same" ).row() == e->row() ); + CHECK( feng::conv( *e, a, "same" ).col() == e->col() ); + } +} + +TEST_CASE( "S8-R1 int elements are exact", "[S8][S8-R1]" ) +{ + auto const a = s8r1::A< int >(); + auto const b = s8r1::B< int >(); + CHECK( s8r1::equal( feng::conv( a, b ), s8r1::make< int >( 4, 6, { 1, -2, 2, 2, -3, 0, 6, 2, 2, -6, 10, -2, 8, 17, 17, 11, -3, 10, 0, 6, 7, 3, 2, -12 } ) ) ); + CHECK( s8r1::equal( feng::conv( a, b, "same" ), s8r1::make< int >( 3, 4, { -2, 2, 2, -3, 2, 2, -6, 10, 17, 17, 11, -3 } ) ) ); + CHECK( s8r1::equal( feng::conv( a, b, "valid" ), s8r1::make< int >( 2, 2, { 2, -6, 17, 11 } ) ) ); + // complex elements take the same path + using cd = std::complex< double >; + auto const ca = s8r1::make< cd >( 1, 2, { cd{ 1, 1 }, cd{ 2, 0 } } ); + auto const cb = s8r1::make< cd >( 1, 2, { cd{ 3, 0 }, cd{ 0, 4 } } ); + CHECK( s8r1::equal( feng::conv( ca, cb ), s8r1::make< cd >( 1, 3, { cd{ 3, 3 }, cd{ 2, 4 }, cd{ 0, 8 } } ) ) ); +} + +TEST_CASE( "S8-R1 impossible valid shapes and unknown modes abort", "[S8][S8-R1]" ) +{ + auto const a = s8r1::A< double >(); + feng::matrix< double > const wide{ 2, 5, 1.0 }; + feng::matrix< double > const tall{ 4, 3, 1.0 }; + s8r1::require_one_message_death( [&] { auto r = feng::conv( a, wide, "valid" ); (void)r; }, "conv: 'valid' mode" ); + s8r1::require_one_message_death( [&] { auto r = feng::conv( a, tall, "valid" ); (void)r; }, "conv: 'valid' mode" ); + s8r1::require_one_message_death( [&] { auto r = feng::conv( a, a, "Full" ); (void)r; }, "conv: unknown mode 'Full'" ); + s8r1::require_one_message_death( [&] { auto r = feng::conv2( a, a, std::string{ "" } ); (void)r; }, "conv: unknown mode ''" ); +} + +namespace s8r1 +{ + // the oracle: full[i][j] = sum A[p][q] * B[i-p][j-q], accumulated in long double (complex for complex T) + template< typename T > + std::vector< std::complex< long double > > direct_full( feng::matrix< T > const& a, feng::matrix< T > const& b ) + { + std::size_t const R = a.row() + b.row() - 1, C = a.col() + b.col() - 1; + std::vector< std::complex< long double > > out( R * C ); + for ( std::size_t p = 0; p != a.row(); ++p ) + for ( std::size_t q = 0; q != a.col(); ++q ) + for ( std::size_t u = 0; u != b.row(); ++u ) + for ( std::size_t v = 0; v != b.col(); ++v ) + { + std::complex< long double > const x{ static_cast< long double >( std::real( a[p][q] ) ), static_cast< long double >( std::imag( a[p][q] ) ) }; + std::complex< long double > const y{ static_cast< long double >( std::real( b[u][v] ) ), static_cast< long double >( std::imag( b[u][v] ) ) }; + out[( p + u ) * C + q + v] += x * y; + } + return out; + } + + // max |got - full[r0 + i][c0 + j]| over got's elements + template< typename T > + long double crop_err( feng::matrix< T > const& got, std::vector< std::complex< long double > > const& full, std::size_t C, + std::size_t r0, std::size_t c0 ) + { + long double e = 0; + for ( std::size_t i = 0; i != got.row(); ++i ) + for ( std::size_t j = 0; j != got.col(); ++j ) + { + std::complex< long double > const g{ static_cast< long double >( std::real( got[i][j] ) ), static_cast< long double >( std::imag( got[i][j] ) ) }; + long double const d = std::abs( g - full[( r0 + i ) * C + c0 + j] ); + if ( !( d <= e ) ) e = d; + } + return e; + } + + template< typename T > + feng::matrix< T > sample( std::size_t r, std::size_t c, int seed ) + { + feng::matrix< T > m{ r, c }; + for ( std::size_t i = 0; i != r * c; ++i ) + { + int const u = static_cast< int >( ( i * 37 + static_cast< std::size_t >( seed ) * 11 ) % 19 ) - 9; + int const w = static_cast< int >( ( i * 23 + static_cast< std::size_t >( seed ) * 7 ) % 17 ) - 8; + if constexpr ( std::is_integral_v< T > ) m.data()[i] = static_cast< T >( u ); + else if constexpr ( std::is_floating_point_v< T > ) m.data()[i] = static_cast< T >( u ) / T( 7 ); + else m.data()[i] = T{ static_cast< double >( u ) / 7.0, static_cast< double >( w ) / 5.0 }; + } + return m; + } + + template< typename T > + void check_direct( std::size_t ra, std::size_t ca, std::size_t rb, std::size_t cb ) + { + INFO( "A " << ra << "x" << ca << ", B " << rb << "x" << cb ); + auto const a = sample< T >( ra, ca, 1 ); + auto const b = sample< T >( rb, cb, 2 ); + static_assert( std::is_same_v< decltype( feng::conv( a, b ) ), feng::matrix< T > > ); // D-030 + auto const full = direct_full( a, b ); + std::size_t const C = ca + cb - 1; + long double sa = 0, sb = 0; + for ( auto const& x : a ) sa += static_cast< long double >( std::abs( x ) ); + for ( auto const& x : b ) sb += static_cast< long double >( std::abs( x ) ); + long double tol = 0; // integers are exact + if constexpr ( !std::is_integral_v< T > ) + tol = 4.0L * static_cast< long double >( std::numeric_limits< decltype( std::abs( T{} ) ) >::epsilon() ) * sa * sb; + auto const f = feng::conv( a, b ); + REQUIRE( f.row() == ra + rb - 1 ); + REQUIRE( f.col() == C ); + CHECK( crop_err( f, full, C, 0, 0 ) <= tol ); + auto const s = feng::conv( a, b, "same" ); + REQUIRE( s.row() == ra ); + REQUIRE( s.col() == ca ); + CHECK( crop_err( s, full, C, ( rb - 1 ) / 2, ( cb - 1 ) / 2 ) <= tol ); + bool const a_holds = ra >= rb && ca >= cb, b_holds = rb >= ra && cb >= ca; + if ( a_holds || b_holds ) + { + auto const v = feng::conv( a, b, "valid" ); + std::size_t const vr = a_holds ? ra - rb + 1 : rb - ra + 1, vc = a_holds ? ca - cb + 1 : cb - ca + 1; + REQUIRE( v.row() == vr ); + REQUIRE( v.col() == vc ); + CHECK( crop_err( v, full, C, a_holds ? rb - 1 : ra - 1, a_holds ? cb - 1 : ca - 1 ) <= tol ); + } + } + + template< typename T > + void check_direct_all() + { + for ( auto [rb, cb] : { std::pair{ 1, 1 }, std::pair{ 3, 4 }, std::pair{ 2, 2 }, std::pair{ 3, 9 }, std::pair{ 6, 8 } } ) + check_direct< T >( 5, 7, static_cast< std::size_t >( rb ), static_cast< std::size_t >( cb ) ); + } +} + +TEST_CASE( "S8-R1 conv on float, int and complex matches a direct sum", "[S8][S8-R1]" ) +{ + SECTION( "float" ) { s8r1::check_direct_all< float >(); } + SECTION( "int" ) { s8r1::check_direct_all< int >(); } + SECTION( "complex" ) { s8r1::check_direct_all< std::complex< double > >(); } + SECTION( "double" ) { s8r1::check_direct_all< double >(); } +} diff --git a/tests/cases/s8_r2.hpp b/tests/cases/s8_r2.hpp new file mode 100644 index 0000000..c08389e --- /dev/null +++ b/tests/cases/s8_r2.hpp @@ -0,0 +1,335 @@ +// S8-R2 (PR-11, F14): fft and ifft follow numpy.fft.fft2 and ifft2: X[k][l] = sum x[m][n] exp(-2 pi i (km/R + ln/C)), +// ifft uses exp(+...) and divides by R*C; result types per D-031. The oracle is a tests-only O(N^2) direct DFT +// accumulating in long double (D-007). Tolerances: eps of the result's real type, N = R*C, L = ceil(log2(2N)). +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace s8r2 +{ + template< typename T > struct real_of { using type = T; }; + template< typename T > struct real_of< std::complex< T > > { using type = T; }; + template< typename T > using real_of_t = typename real_of< T >::type; + + // L = ceil(log2(2N)) + inline long double big_l( std::size_t n ) + { + long double l = 0; + for ( std::size_t p = 1; p < 2 * n; p *= 2 ) l += 1; + return l; + } + + // the oracle: direct 2-D DFT with sign -1 (fft) or +1 (ifft, without the scale), in long double + template< typename T > + feng::matrix< std::complex< long double > > direct_dft( feng::matrix< T > const& x, int sign ) + { + using cl = std::complex< long double >; + std::size_t const R = x.row(); + std::size_t const C = x.col(); + feng::matrix< cl > out{ R, C }; + long double const two_pi = 2.0L * std::numbers::pi_v< long double >; + // exp(sign 2 pi i j / R) and exp(sign 2 pi i j / C) for every residue j; the phase index is reduced exactly + std::vector< cl > wr( R ), wc( C ); + for ( std::size_t j = 0; j != R; ++j ) + wr[j] = std::polar( 1.0L, static_cast< long double >( sign ) * two_pi * static_cast< long double >( j ) / static_cast< long double >( R ) ); + for ( std::size_t j = 0; j != C; ++j ) + wc[j] = std::polar( 1.0L, static_cast< long double >( sign ) * two_pi * static_cast< long double >( j ) / static_cast< long double >( C ) ); + for ( std::size_t k = 0; k != R; ++k ) + for ( std::size_t l = 0; l != C; ++l ) + { + cl acc{ 0.0L, 0.0L }; + for ( std::size_t m = 0; m != R; ++m ) + { + cl const w_r = wr[( k * m ) % R]; + for ( std::size_t n = 0; n != C; ++n ) + { + cl const v{ static_cast< long double >( std::real( x[m][n] ) ), + static_cast< long double >( std::imag( x[m][n] ) ) }; + acc += v * ( w_r * wc[( l * n ) % C] ); + } + } + out[k][l] = acc; + } + return out; + } + + // deterministic values in [-1, 1] + template< typename T > + feng::matrix< T > sample( std::size_t r, std::size_t c, std::uint64_t seed ) + { + feng::matrix< T > m{ r, c }; + std::uint64_t s = seed * 6364136223846793005ULL + 1442695040888963407ULL; + auto next = [&s]() + { + s = s * 6364136223846793005ULL + 1442695040888963407ULL; + return static_cast< double >( s >> 11 ) / 9007199254740992.0 * 2.0 - 1.0; + }; + using R = real_of_t< T >; + for ( std::size_t i = 0; i != r * c; ++i ) + { + if constexpr ( std::is_integral_v< T > ) + m.data()[i] = static_cast< T >( std::lround( next() * 9.0 ) ); // -9 .. 9 + else if constexpr ( std::is_same_v< T, R > ) + m.data()[i] = static_cast< R >( next() ); + else + { + R const re = static_cast< R >( next() ); + R const im = static_cast< R >( next() ); + m.data()[i] = T{ re, im }; + } + } + return m; + } + + template< typename T > + long double sum_abs( feng::matrix< T > const& x ) + { + long double s = 0; + for ( std::size_t i = 0; i != x.size(); ++i ) s += static_cast< long double >( std::abs( x.data()[i] ) ); + return s; + } + + template< typename T > + long double max_abs( feng::matrix< T > const& x ) + { + long double s = 0; + for ( std::size_t i = 0; i != x.size(); ++i ) + s = std::max( s, static_cast< long double >( std::abs( x.data()[i] ) ) ); + return s; + } + + // max |a - b| elementwise, b in long double + template< typename C > + long double max_diff( feng::matrix< C > const& a, feng::matrix< std::complex< long double > > const& b ) + { + long double d = 0; + for ( std::size_t i = 0; i != a.size(); ++i ) + { + std::complex< long double > const v{ static_cast< long double >( a.data()[i].real() ), + static_cast< long double >( a.data()[i].imag() ) }; + d = std::max( d, std::abs( v - b.data()[i] ) ); + } + return d; + } + + template< typename T > + feng::matrix< std::complex< long double > > widen( feng::matrix< T > const& x ) + { + feng::matrix< std::complex< long double > > out{ x.row(), x.col() }; + for ( std::size_t i = 0; i != x.size(); ++i ) + out.data()[i] = { static_cast< long double >( std::real( x.data()[i] ) ), + static_cast< long double >( std::imag( x.data()[i] ) ) }; + return out; + } + + template< typename T > + void check_against_direct( std::size_t r, std::size_t c ) + { + using R = typename decltype( feng::fft( std::declval< feng::matrix< T > const& >() ) )::value_type::value_type; // D-031 + INFO( r << "x" << c << " " << ( std::is_same_v< T, real_of_t< T > > ? "real" : "complex" ) << " eps " + << std::numeric_limits< R >::epsilon() ); + auto const x = sample< T >( r, c, 17 * r + c ); + long double const n = static_cast< long double >( r * c ); + long double const eps = static_cast< long double >( std::numeric_limits< R >::epsilon() ); + long double const l = big_l( r * c ); + + auto const X = feng::fft( x ); + REQUIRE( X.row() == r ); + REQUIRE( X.col() == c ); + long double const tol_f = 16.0L * l * eps * sum_abs( x ); + long double const df = max_diff( X, direct_dft( x, -1 ) ); + INFO( "fft error " << static_cast< double >( df ) << " tol " << static_cast< double >( tol_f ) ); + CHECK( df <= tol_f ); + + // ifft against the direct inverse DFT divided by R*C + auto const xi = feng::ifft( x ); + auto inv = direct_dft( x, +1 ); + for ( std::size_t i = 0; i != inv.size(); ++i ) inv.data()[i] /= n; + long double const di = max_diff( xi, inv ); + INFO( "ifft error " << static_cast< double >( di ) << " tol " << static_cast< double >( tol_f / n ) ); + CHECK( di <= tol_f / n ); + + // ifft(fft(x)) within 16 L eps max|x| of x + auto const back = feng::ifft( X ); + long double const dr = max_diff( back, widen( x ) ); + long double const tol_r = 16.0L * l * eps * max_abs( x ); + INFO( "roundtrip error " << static_cast< double >( dr ) << " tol " << static_cast< double >( tol_r ) ); + CHECK( dr <= tol_r ); + } + + inline std::pair< std::size_t, std::size_t > const sizes[] = { + { 1, 1 }, { 1, 2 }, { 1, 7 }, { 1, 16 }, { 3, 5 }, { 4, 6 }, { 7, 11 }, { 13, 1 }, + { 31, 17 }, { 8, 8 }, { 16, 32 }, { 64, 64 } }; + + template< typename T > + void check_all_sizes() + { + for ( auto [r, c] : sizes ) check_against_direct< T >( r, c ); + } +} + +TEST_CASE( "S8-R2 fft matches the direct DFT on every listed size", "[S8][S8-R2]" ) +{ + SECTION( "double" ) { s8r2::check_all_sizes< double >(); } + SECTION( "complex" ) { s8r2::check_all_sizes< std::complex< double > >(); } + SECTION( "float" ) { s8r2::check_all_sizes< float >(); } + SECTION( "complex" ) { s8r2::check_all_sizes< std::complex< float > >(); } + SECTION( "long double" ) { s8r2::check_all_sizes< long double >(); } +} + +TEST_CASE( "S8-R2 fft of an off-origin impulse is the closed-form phase ramp", "[S8][S8-R2]" ) +{ + for ( auto [R, C, r0, c0] : { std::array< std::size_t, 4 >{ 8, 8, 3, 5 }, std::array< std::size_t, 4 >{ 7, 11, 2, 9 }, + std::array< std::size_t, 4 >{ 4, 6, 0, 1 }, std::array< std::size_t, 4 >{ 1, 7, 0, 4 } } ) + { + INFO( R << "x" << C << " impulse at " << r0 << "," << c0 ); + feng::matrix< double > x{ R, C }; + std::fill( x.begin(), x.end(), 0.0 ); + x[r0][c0] = 1.0; + auto const X = feng::fft( x ); + long double const tol = 16.0L * s8r2::big_l( R * C ) * std::numeric_limits< double >::epsilon(); + long double const two_pi = 2.0L * std::numbers::pi_v< long double >; + for ( std::size_t k = 0; k != R; ++k ) + for ( std::size_t l = 0; l != C; ++l ) + { + long double const ph = static_cast< long double >( ( k * r0 ) % R ) / static_cast< long double >( R ) + + static_cast< long double >( ( l * c0 ) % C ) / static_cast< long double >( C ); + std::complex< long double > const want = std::polar( 1.0L, -two_pi * ph ); + std::complex< long double > const got{ X[k][l].real(), X[k][l].imag() }; + CHECK( std::abs( got - want ) <= tol ); + } + } +} + +TEST_CASE( "S8-R2 fft of a sinusoid peaks at (a, b) and (-a, -b)", "[S8][S8-R2]" ) +{ + for ( auto [R, C, a, b] : { std::array< std::size_t, 4 >{ 8, 12, 1, 3 }, std::array< std::size_t, 4 >{ 7, 10, 2, 3 }, + std::array< std::size_t, 4 >{ 16, 16, 5, 2 } } ) + { + INFO( R << "x" << C << " frequency " << a << "," << b ); + feng::matrix< double > x{ R, C }; + long double const two_pi = 2.0L * std::numbers::pi_v< long double >; + for ( std::size_t m = 0; m != R; ++m ) + for ( std::size_t n = 0; n != C; ++n ) + { + long double const ph = static_cast< long double >( ( a * m ) % R ) / static_cast< long double >( R ) + + static_cast< long double >( ( b * n ) % C ) / static_cast< long double >( C ); + x[m][n] = static_cast< double >( std::cos( two_pi * ph ) ); + } + auto const X = feng::fft( x ); + long double const tol = 16.0L * s8r2::big_l( R * C ) * std::numeric_limits< double >::epsilon() * s8r2::sum_abs( x ); + std::size_t const ka = ( R - a ) % R; + std::size_t const kb = ( C - b ) % C; + for ( std::size_t k = 0; k != R; ++k ) + for ( std::size_t l = 0; l != C; ++l ) + { + bool const peak = ( k == a && l == b ) || ( k == ka && l == kb ); + long double const want = peak ? static_cast< long double >( R * C ) / 2.0L : 0.0L; + std::complex< long double > const got{ X[k][l].real(), X[k][l].imag() }; + CHECK( std::abs( got - std::complex< long double >{ want, 0.0L } ) <= tol ); + } + } +} + +TEST_CASE( "S8-R2 ifft divides by R*C", "[S8][S8-R2]" ) +{ + for ( auto [R, C] : { std::pair< std::size_t, std::size_t >{ 4, 6 }, std::pair< std::size_t, std::size_t >{ 7, 11 }, + std::pair< std::size_t, std::size_t >{ 8, 8 } } ) + { + INFO( R << "x" << C ); + long double const tol = 16.0L * s8r2::big_l( R * C ) * std::numeric_limits< double >::epsilon() * 3.0L; + // ifft of the constant 3 is 3 at the origin and 0 elsewhere + feng::matrix< double > k{ R, C }; + std::fill( k.begin(), k.end(), 3.0 ); + auto const xi = feng::ifft( k ); + for ( std::size_t i = 0; i != R; ++i ) + for ( std::size_t j = 0; j != C; ++j ) + CHECK( std::abs( xi[i][j] - std::complex< double >{ ( i == 0 && j == 0 ) ? 3.0 : 0.0, 0.0 } ) <= tol ); + // ifft of an origin impulse of height 3*R*C is the constant 3 + feng::matrix< double > d{ R, C }; + std::fill( d.begin(), d.end(), 0.0 ); + d[0][0] = 3.0 * static_cast< double >( R * C ); + auto const xd = feng::ifft( d ); + for ( std::size_t i = 0; i != xd.size(); ++i ) + CHECK( std::abs( xd.data()[i] - std::complex< double >{ 3.0, 0.0 } ) <= tol ); + } +} + +TEST_CASE( "S8-R2 fft and ifft of an empty matrix keep its shape", "[S8][S8-R2]" ) +{ + for ( auto [R, C] : { std::pair< std::size_t, std::size_t >{ 0, 0 }, std::pair< std::size_t, std::size_t >{ 0, 3 }, + std::pair< std::size_t, std::size_t >{ 5, 0 } } ) + { + feng::matrix< double > const e{ R, C }; + auto const X = feng::fft( e ); + auto const x = feng::ifft( e ); + CHECK( X.row() == R ); + CHECK( X.col() == C ); + CHECK( x.row() == R ); + CHECK( x.col() == C ); + feng::matrix< std::complex< float > > const ec{ R, C }; + CHECK( feng::fft( ec ).size() == 0 ); + CHECK( feng::ifft( ec ).size() == 0 ); + } +} + +TEST_CASE( "S8-R2 fft result types follow D-031", "[S8][S8-R2]" ) +{ + using std::declval; + static_assert( std::is_same_v< decltype( feng::fft( declval< feng::matrix< float > const& >() ) ), + feng::matrix< std::complex< float > > > ); + static_assert( std::is_same_v< decltype( feng::fft( declval< feng::matrix< double > const& >() ) ), + feng::matrix< std::complex< double > > > ); + static_assert( std::is_same_v< decltype( feng::fft( declval< feng::matrix< long double > const& >() ) ), + feng::matrix< std::complex< long double > > > ); + static_assert( std::is_same_v< decltype( feng::fft( declval< feng::matrix< int > const& >() ) ), + feng::matrix< std::complex< double > > > ); + static_assert( std::is_same_v< decltype( feng::fft( declval< feng::matrix< std::complex< float > > const& >() ) ), + feng::matrix< std::complex< float > > > ); + static_assert( std::is_same_v< decltype( feng::fft( declval< feng::matrix< std::complex< double > > const& >() ) ), + feng::matrix< std::complex< double > > > ); + static_assert( std::is_same_v< decltype( feng::ifft( declval< feng::matrix< float > const& >() ) ), + feng::matrix< std::complex< float > > > ); + static_assert( std::is_same_v< decltype( feng::ifft( declval< feng::matrix< int > const& >() ) ), + feng::matrix< std::complex< double > > > ); + static_assert( std::is_same_v< decltype( feng::ifft( declval< feng::matrix< long double > const& >() ) ), + feng::matrix< std::complex< long double > > > ); + static_assert( std::is_same_v< decltype( feng::ifft( declval< feng::matrix< std::complex< double > > const& >() ) ), + feng::matrix< std::complex< double > > > ); + static_assert( std::is_same_v< decltype( feng::ifft( declval< feng::matrix< std::complex< float > > const& >() ) ), + feng::matrix< std::complex< float > > > ); + static_assert( std::is_same_v< decltype( feng::ifft( declval< feng::matrix< double > const& >() ) ), + feng::matrix< std::complex< double > > > ); + // values of every D-031 element type against the direct DFT, at the tolerance of the result's real type + for ( auto [r, c] : { std::pair< std::size_t, std::size_t >{ 3, 5 }, std::pair< std::size_t, std::size_t >{ 8, 8 }, + std::pair< std::size_t, std::size_t >{ 7, 11 } } ) + { + s8r2::check_against_direct< float >( r, c ); + s8r2::check_against_direct< long double >( r, c ); + s8r2::check_against_direct< int >( r, c ); + s8r2::check_against_direct< std::complex< float > >( r, c ); + s8r2::check_against_direct< std::complex< double > >( r, c ); + } + // an int input transforms like its double copy + feng::matrix< int > xi{ 3, 5 }; + feng::matrix< double > xd{ 3, 5 }; + for ( std::size_t i = 0; i != xi.size(); ++i ) + { + xi.data()[i] = static_cast< int >( i * 7 % 11 ) - 5; + xd.data()[i] = static_cast< double >( xi.data()[i] ); + } + auto const Xi = feng::fft( xi ); + auto const Xd = feng::fft( xd ); + for ( std::size_t i = 0; i != Xi.size(); ++i ) CHECK( Xi.data()[i] == Xd.data()[i] ); + auto const Ii = feng::ifft( xi ); + auto const Id = feng::ifft( xd ); + for ( std::size_t i = 0; i != Ii.size(); ++i ) CHECK( Ii.data()[i] == Id.data()[i] ); +} diff --git a/tests/cases/s8_r3.hpp b/tests/cases/s8_r3.hpp new file mode 100644 index 0000000..4d00529 --- /dev/null +++ b/tests/cases/s8_r3.hpp @@ -0,0 +1,130 @@ +// S8-R3 (PR-11, F14): fftshift and ifftshift are numpy's rolls by R/2, C/2 and -(R/2), -(C/2) on a matrix of x's +// own type and allocator; no transform is computed. Literal values below come from numpy 2.5.3 on arange(R*C). +#include +#include +#include +#include +#include + +namespace s8r3 +{ + template< typename T > + feng::matrix< T > iota( std::size_t r, std::size_t c ) + { + feng::matrix< T > m{ r, c }; + for ( std::size_t i = 0; i != r * c; ++i ) m.data()[i] = T( static_cast< int >( i ) ); + return m; + } + + template< typename T > + std::vector< T > values( feng::matrix< T > const& m ) + { + return std::vector< T >( m.data(), m.data() + m.size() ); + } + + template< typename T > + bool equal( feng::matrix< T > const& a, feng::matrix< T > const& b ) + { + return a.row() == b.row() && a.col() == b.col() && values( a ) == values( b ); + } + + // the formula: out[(i + dr) mod R][(j + dc) mod C] = x[i][j] + template< typename T > + feng::matrix< T > rolled( feng::matrix< T > const& x, std::ptrdiff_t dr, std::ptrdiff_t dc ) + { + std::ptrdiff_t const R = static_cast< std::ptrdiff_t >( x.row() ); + std::ptrdiff_t const C = static_cast< std::ptrdiff_t >( x.col() ); + feng::matrix< T > out{ x.row(), x.col() }; + for ( std::ptrdiff_t i = 0; i != R; ++i ) + for ( std::ptrdiff_t j = 0; j != C; ++j ) + out[( ( i + dr ) % R + R ) % R][( ( j + dc ) % C + C ) % C] = x[i][j]; + return out; + } + + template< typename T > + void check_size( std::size_t r, std::size_t c ) + { + INFO( r << "x" << c ); + auto const x = iota< T >( r, c ); + auto const s = feng::fftshift( x ); + auto const is = feng::ifftshift( x ); + static_assert( std::is_same_v< std::remove_cvref_t< decltype( s ) >, feng::matrix< T > > ); + static_assert( std::is_same_v< std::remove_cvref_t< decltype( is ) >, feng::matrix< T > > ); + std::ptrdiff_t const hr = static_cast< std::ptrdiff_t >( r / 2 ); + std::ptrdiff_t const hc = static_cast< std::ptrdiff_t >( c / 2 ); + CHECK( equal( s, rolled( x, hr, hc ) ) ); + CHECK( equal( is, rolled( x, -hr, -hc ) ) ); + CHECK( equal( feng::ifftshift( s ), x ) ); + CHECK( equal( feng::fftshift( is ), x ) ); + auto sorted = []( feng::matrix< T > const& m ) + { + auto v = values( m ); + std::sort( v.begin(), v.end(), []( T const& a, T const& b ) { return std::real( a ) < std::real( b ); } ); + return v; + }; + CHECK( sorted( s ) == sorted( x ) ); + CHECK( sorted( is ) == sorted( x ) ); + } + + template< typename T > + feng::matrix< T > make( std::size_t r, std::size_t c, std::vector< int > const& v ) + { + feng::matrix< T > m{ r, c }; + for ( std::size_t i = 0; i != r * c; ++i ) m.data()[i] = T( v[i] ); + return m; + } + + template< typename T > + void check_literals() + { + // numpy.fft.fftshift / ifftshift of numpy.arange(R*C).reshape(R, C) + CHECK( equal( feng::fftshift( iota< T >( 3, 5 ) ), + make< T >( 3, 5, { 13, 14, 10, 11, 12, 3, 4, 0, 1, 2, 8, 9, 5, 6, 7 } ) ) ); + CHECK( equal( feng::ifftshift( iota< T >( 3, 5 ) ), + make< T >( 3, 5, { 7, 8, 9, 5, 6, 12, 13, 14, 10, 11, 2, 3, 4, 0, 1 } ) ) ); + std::vector< int > const s46{ 15, 16, 17, 12, 13, 14, 21, 22, 23, 18, 19, 20, + 3, 4, 5, 0, 1, 2, 9, 10, 11, 6, 7, 8 }; + CHECK( equal( feng::fftshift( iota< T >( 4, 6 ) ), make< T >( 4, 6, s46 ) ) ); + CHECK( equal( feng::ifftshift( iota< T >( 4, 6 ) ), make< T >( 4, 6, s46 ) ) ); + CHECK( equal( feng::fftshift( iota< T >( 5, 5 ) ), + make< T >( 5, 5, { 18, 19, 15, 16, 17, 23, 24, 20, 21, 22, 3, 4, 0, 1, 2, + 8, 9, 5, 6, 7, 13, 14, 10, 11, 12 } ) ) ); + CHECK( equal( feng::ifftshift( iota< T >( 5, 5 ) ), + make< T >( 5, 5, { 12, 13, 14, 10, 11, 17, 18, 19, 15, 16, 22, 23, 24, 20, 21, + 2, 3, 4, 0, 1, 7, 8, 9, 5, 6 } ) ) ); + CHECK( equal( feng::fftshift( iota< T >( 1, 4 ) ), make< T >( 1, 4, { 2, 3, 0, 1 } ) ) ); + CHECK( equal( feng::ifftshift( iota< T >( 1, 4 ) ), make< T >( 1, 4, { 2, 3, 0, 1 } ) ) ); + CHECK( equal( feng::fftshift( iota< T >( 6, 1 ) ), make< T >( 6, 1, { 3, 4, 5, 0, 1, 2 } ) ) ); + CHECK( equal( feng::ifftshift( iota< T >( 6, 1 ) ), make< T >( 6, 1, { 3, 4, 5, 0, 1, 2 } ) ) ); + CHECK( equal( feng::fftshift( iota< T >( 1, 1 ) ), iota< T >( 1, 1 ) ) ); + CHECK( equal( feng::ifftshift( iota< T >( 1, 1 ) ), iota< T >( 1, 1 ) ) ); + } + + template< typename T > + void check_all() + { + check_literals< T >(); + for ( auto [r, c] : { std::pair{ 1, 1 }, std::pair{ 1, 4 }, std::pair{ 3, 5 }, std::pair{ 4, 6 }, + std::pair{ 5, 5 }, std::pair{ 6, 1 } } ) + check_size< T >( static_cast< std::size_t >( r ), static_cast< std::size_t >( c ) ); + for ( auto [r, c] : { std::pair{ 0, 0 }, std::pair{ 0, 3 }, std::pair{ 4, 0 } } ) + { + feng::matrix< T > const e{ static_cast< std::size_t >( r ), static_cast< std::size_t >( c ) }; + auto const s = feng::fftshift( e ); + auto const is = feng::ifftshift( e ); + CHECK( s.row() == e.row() ); + CHECK( s.col() == e.col() ); + CHECK( is.row() == e.row() ); + CHECK( is.col() == e.col() ); + CHECK( s.size() == 0 ); + CHECK( is.size() == 0 ); + } + } +} + +TEST_CASE( "S8-R3 shifts are numpy rolls and invert each other exactly", "[S8][S8-R3]" ) +{ + SECTION( "double" ) { s8r3::check_all< double >(); } + SECTION( "int" ) { s8r3::check_all< int >(); } + SECTION( "complex" ) { s8r3::check_all< std::complex< double > >(); } +} diff --git a/tests/cases/s9_r1.hpp b/tests/cases/s9_r1.hpp new file mode 100644 index 0000000..4834ec1 --- /dev/null +++ b/tests/cases/s9_r1.hpp @@ -0,0 +1,126 @@ +// S9-R1 (PR-12): plain value returns. Factory and shape results are non-const prvalues, so assigning them to an +// existing matrix or move-constructing from them moves the storage instead of copying it (one allocation per +// result, counted with s9_r1::counting_alloc); magic( n ) builds in T and A and keeps magic( n )'s values. +#include +#include +#include +#include +#include +#include + +namespace s9_r1 +{ + struct counts + { + static inline long allocations = 0; + static inline long deallocations = 0; + static void reset() noexcept { allocations = 0; deallocations = 0; } + }; + + template < typename T > + struct counting_alloc + { + using value_type = T; + counting_alloc() noexcept = default; + template < typename U > + counting_alloc( counting_alloc< U > const& ) noexcept {} + T* allocate( std::size_t n ) + { + ++counts::allocations; + return std::allocator< T >{}.allocate( n ); + } + void deallocate( T* p, std::size_t n ) noexcept + { + ++counts::deallocations; + std::allocator< T >{}.deallocate( p, n ); + } + template < typename U > + friend bool operator == ( counting_alloc const&, counting_alloc< U > const& ) noexcept { return true; } + }; + + using cmat = feng::matrix< double, counting_alloc< double > >; + + // decltype of each call is its declared return type; none may be const-qualified. + inline void const_free_returns() noexcept + { + feng::matrix< double > const m{ 3, 3, 1.0 }; + std::mt19937_64 g{ 1 }; + static_assert( !std::is_const_v< decltype( feng::eye< double >( 3 ) ) > ); + static_assert( !std::is_const_v< decltype( feng::zeros< double >( 3, 3 ) ) > ); + static_assert( !std::is_const_v< decltype( feng::ones< double >( 3, 3 ) ) > ); + static_assert( !std::is_const_v< decltype( feng::magic( 4 ) ) > ); + static_assert( !std::is_const_v< decltype( feng::magic< double, counting_alloc< double > >( 4 ) ) > ); + static_assert( !std::is_const_v< decltype( feng::transpose( m ) ) > ); + static_assert( !std::is_const_v< decltype( feng::inverse( m ) ) > ); + static_assert( !std::is_const_v< decltype( feng::diag( m ) ) > ); + static_assert( !std::is_const_v< decltype( feng::tril( m ) ) > ); + static_assert( !std::is_const_v< decltype( feng::triu( m ) ) > ); + static_assert( !std::is_const_v< decltype( feng::fliplr( m ) ) > ); + static_assert( !std::is_const_v< decltype( feng::flipud( m ) ) > ); + static_assert( !std::is_const_v< decltype( feng::rand( 3, 3, g ) ) > ); + static_assert( !std::is_const_v< decltype( m.transpose() ) > ); + static_assert( !std::is_const_v< decltype( m.inverse() ) > ); + static_assert( !std::is_const_v< decltype( m.clone( 0, 2, 0, 2 ) ) > ); + static_assert( !std::is_const_v< decltype( m.astype< float >() ) > ); + } +} + +TEST_CASE( "S9-R1 factory results are moved, not copied", "[S9][S9-R1]" ) +{ + using s9_r1::cmat; + using s9_r1::counts; + s9_r1::const_free_returns(); + s9_r1::counting_alloc< double > const alloc; + + SECTION( "magic> assigned to an existing matrix" ) + { + for ( std::uint_least64_t n : { 3u, 4u, 5u, 6u, 8u, 10u } ) + { + cmat m{ 2, 2 }; + counts::reset(); + m = feng::magic< double, s9_r1::counting_alloc< double > >( n ); + REQUIRE( counts::allocations == 1 ); + REQUIRE( counts::deallocations == 1 ); // the old 2x2 storage only + auto const ref = feng::magic( n ); + REQUIRE( m.row() == n ); + REQUIRE( m.col() == n ); + for ( std::uint_least64_t r = 0; r != n; ++r ) + for ( std::uint_least64_t c = 0; c != n; ++c ) + REQUIRE( m[r][c] == static_cast< double >( ref[r][c] ) ); + } + } + + SECTION( "move construction takes the storage" ) + { + counts::reset(); + auto y = feng::magic< double, s9_r1::counting_alloc< double > >( 5 ); + REQUIRE( counts::allocations == 1 ); + double const* const data = y.data(); + cmat x( std::move( y ) ); + REQUIRE( counts::allocations == 1 ); + REQUIRE( x.data() == data ); + REQUIRE( x[2][2] == 13.0 ); + } + + SECTION( "factory and shape results initialise and assign with one allocation each" ) + { + counts::reset(); + auto z = feng::zeros< double >( alloc, 3, 4 ); + REQUIRE( counts::allocations == 1 ); + auto o = feng::ones< double >( alloc, 4, 3 ); + REQUIRE( counts::allocations == 2 ); + cmat t{ 1, 1 }; + long const before = counts::allocations; + t = z.transpose(); + REQUIRE( counts::allocations == before + 1 ); + t = feng::transpose( o ); + REQUIRE( counts::allocations == before + 2 ); + cmat c( z.clone( 0, 2, 0, 2 ) ); + REQUIRE( counts::allocations == before + 3 ); + REQUIRE( c.row() == 2 ); + REQUIRE( c.col() == 2 ); + REQUIRE( t.row() == 3 ); + REQUIRE( t.col() == 4 ); + REQUIRE( t[0][0] == 1.0 ); + } +} diff --git a/tests/cases/s9_r4.hpp b/tests/cases/s9_r4.hpp new file mode 100644 index 0000000..636ab58 --- /dev/null +++ b/tests/cases/s9_r4.hpp @@ -0,0 +1,148 @@ +// S9-R4 (PR-12, D-032): span access to the storage and rows; std::mdspan and std::submdspan adapters where the +// library has them. In C++20 (macros unset) the gated adapters are not declared, and make_view / +// make_mutable_view plus row_span give the same extents and elements. +#include "./s2_death.hpp" + +#include +#include +#include +#include +#include + +namespace s9_r4 +{ + template < typename M > + concept has_to_mdspan = requires( M& m ) { to_mdspan( m ); }; + + template < typename M > + concept has_submdspan = requires( M& m ) { submdspan( m, { 0, 1 }, { 0, 1 } ); }; + + // a 3x4 matrix holding 10*r + c + inline feng::matrix< double > grid() + { + feng::matrix< double > m{ 3, 4 }; + for ( std::size_t r = 0; r != 3; ++r ) + for ( std::size_t c = 0; c != 4; ++c ) + m[r][c] = static_cast< double >( 10 * r + c ); + return m; + } +} + +TEST_CASE( "S9-R4 span adapters alias the storage and rows", "[S9][S9-R4]" ) +{ + auto m = s9_r4::grid(); + auto const& cm = m; + + static_assert( std::is_same_v< decltype( feng::as_span( m ) ), std::span< double > > ); + static_assert( std::is_same_v< decltype( feng::as_span( cm ) ), std::span< double const > > ); + static_assert( std::is_same_v< decltype( feng::row_span( m, 0 ) ), std::span< double > > ); + static_assert( std::is_same_v< decltype( feng::row_span( cm, 0 ) ), std::span< double const > > ); + + auto const all = feng::as_span( m ); + REQUIRE( all.data() == m.data() ); + REQUIRE( all.size() == m.size() ); + REQUIRE( feng::as_span( cm ).data() == m.data() ); + for ( std::size_t r = 0; r != 3; ++r ) + { + auto const row = feng::row_span( cm, r ); + REQUIRE( row.size() == 4 ); + REQUIRE( row.data() == m.data() + r * 4 ); + for ( std::size_t c = 0; c != 4; ++c ) + REQUIRE( row[c] == m[r][c] ); + } + feng::row_span( m, 1 )[2] = -1.0; + REQUIRE( m[1][2] == -1.0 ); + all[0] = -2.0; + REQUIRE( m[0][0] == -2.0 ); + + feng::matrix< double > empty; + REQUIRE( feng::as_span( empty ).empty() ); + + SECTION( "row_span outside the matrix aborts with one message" ) + { + S2_REQUIRE_DEATH( [&] { auto s = feng::row_span( cm, 3 ); (void)s; }, "row_span: row 3 outside a matrix with 3 rows" ); + } +} + +#if defined( __cpp_lib_mdspan ) +TEST_CASE( "S9-R4 to_mdspan has the matrix extents, layout_right strides and aliases it", "[S9][S9-R4]" ) +{ + auto m = s9_r4::grid(); + auto const& cm = m; + static_assert( s9_r4::has_to_mdspan< feng::matrix< double > > ); + static_assert( std::is_same_v< decltype( feng::to_mdspan( m ) ), std::mdspan< double, std::dextents< std::size_t, 2 > > > ); + static_assert( std::is_same_v< decltype( feng::to_mdspan( cm ) ), std::mdspan< double const, std::dextents< std::size_t, 2 > > > ); + auto const md = feng::to_mdspan( m ); + REQUIRE( md.extent( 0 ) == 3 ); + REQUIRE( md.extent( 1 ) == 4 ); + REQUIRE( md.stride( 0 ) == 4 ); + REQUIRE( md.stride( 1 ) == 1 ); + REQUIRE( md.data_handle() == m.data() ); + for ( std::size_t r = 0; r != 3; ++r ) + for ( std::size_t c = 0; c != 4; ++c ) + REQUIRE( &md[r, c] == &m[r][c] ); + md[2, 3] = -5.0; + REQUIRE( m[2][3] == -5.0 ); + REQUIRE( feng::to_mdspan( cm )[2, 3] == -5.0 ); +} +#endif + +#if defined( __cpp_lib_submdspan ) +TEST_CASE( "S9-R4 submdspan slices the matrix after the view validation", "[S9][S9-R4]" ) +{ + auto m = s9_r4::grid(); + auto const& cm = m; + static_assert( s9_r4::has_submdspan< feng::matrix< double > > ); + auto const s = feng::submdspan( m, { 1, 3 }, { 1, 4 } ); + REQUIRE( s.extent( 0 ) == 2 ); + REQUIRE( s.extent( 1 ) == 3 ); + REQUIRE( s.stride( 0 ) == 4 ); + REQUIRE( s.stride( 1 ) == 1 ); + for ( std::size_t r = 0; r != 2; ++r ) + for ( std::size_t c = 0; c != 3; ++c ) + REQUIRE( &s[r, c] == &m[r + 1][c + 1] ); + auto const cs = feng::submdspan( cm, { 0, 0 }, { 0, 4 } ); + static_assert( std::is_const_v< std::remove_reference_t< decltype( cs[0, 0] ) > > ); + REQUIRE( cs.extent( 0 ) == 0 ); + REQUIRE( cs.extent( 1 ) == 4 ); + + SECTION( "an invalid slice aborts" ) + { + S2_REQUIRE_DEATH( [&] { auto x = feng::submdspan( cm, { 0, 4 }, { 0, 1 } ); (void)x; }, "matrix view: row range [0, 4)" ); + S2_REQUIRE_DEATH( [&] { auto x = feng::submdspan( cm, { 2, 1 }, { 0, 1 } ); (void)x; }, "matrix view: row range [2, 1)" ); + S2_REQUIRE_DEATH( [&] { auto x = feng::submdspan( cm, { 0, 1 }, { 0, 5 } ); (void)x; }, "matrix view: column range [0, 5)" ); + } +} +#endif + +#if !defined( __cpp_lib_mdspan ) || !defined( __cpp_lib_submdspan ) +TEST_CASE( "S9-R4 C++20 fallback: views and row_span give the same extents and elements", "[S9][S9-R4]" ) +{ +#if !defined( __cpp_lib_mdspan ) + static_assert( !s9_r4::has_to_mdspan< feng::matrix< double > > ); +#endif + static_assert( !s9_r4::has_submdspan< feng::matrix< double > > ); + auto m = s9_r4::grid(); + auto const& cm = m; + + auto const v = feng::make_view( cm, { 0, 3 }, { 0, 4 } ); // the whole matrix, as to_mdspan( m ) + REQUIRE( v.row() == 3 ); + REQUIRE( v.col() == 4 ); + for ( std::size_t r = 0; r != 3; ++r ) + { + auto const row = feng::row_span( cm, r ); + for ( std::size_t c = 0; c != 4; ++c ) + { + REQUIRE( v( r, c ) == m[r][c] ); + REQUIRE( row[c] == v( r, c ) ); + } + } + + auto const mv = feng::make_mutable_view( m, { 1, 3 }, { 1, 4 } ); // as submdspan( m, {1, 3}, {1, 4} ) + REQUIRE( mv.row() == 2 ); + REQUIRE( mv.col() == 3 ); + for ( std::size_t r = 0; r != 2; ++r ) + for ( std::size_t c = 0; c != 3; ++c ) + REQUIRE( &mv( r, c ) == &feng::row_span( m, r + 1 )[c + 1] ); +} +#endif diff --git a/tests/cases/s9_r5.hpp b/tests/cases/s9_r5.hpp new file mode 100644 index 0000000..2e8a0b0 --- /dev/null +++ b/tests/cases/s9_r5.hpp @@ -0,0 +1,188 @@ +// S9-R5 (PR-12, D-032): std::expected adapters over linalg_result, the factorization objects and the S5 loaders +// where the library has std::expected. In C++20 the adapters are not declared and the same results stay +// available as linalg_result, the factorization objects' status() and the bool loaders. +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +namespace s9_r5 +{ + template < typename R > + concept has_to_expected = requires( R r ) { to_expected( r ); }; + + template < typename T, typename F > + concept has_load_expected = requires( F f ) { load_expected< T >( std::string{}, f ); }; + + inline std::string temp( std::string const& name ) + { + return ( std::filesystem::temp_directory_path() / ( "feng_s9_r5_" + std::to_string( ::getpid() ) + "_" + name ) ).string(); + } + + inline std::string missing() { return temp( "does_not_exist.npy" ); } + + inline feng::matrix< double > invertible() { return feng::matrix< double >{ 2, 2, { 2.0, 1.0, 1.0, 1.0 } }; } + inline feng::matrix< double > singular() { return feng::matrix< double >{ 2, 2, { 1.0, 2.0, 2.0, 4.0 } }; } +} + +TEST_CASE( "S9-R5 io_status and io_format are always declared", "[S9][S9-R5]" ) +{ + static_assert( std::is_enum_v< feng::io_status > && std::is_enum_v< feng::io_format > ); + REQUIRE( feng::io_status::ok != feng::io_status::failed ); + REQUIRE( feng::io_format::txt != feng::io_format::binary ); + REQUIRE( feng::io_format::binary != feng::io_format::npy ); +} + +#if defined( __cpp_lib_expected ) +TEST_CASE( "S9-R5 to_expected carries value or status", "[S9][S9-R5]" ) +{ + using s9_r5::invertible; + using s9_r5::singular; + using E = std::expected< feng::matrix< double >, feng::linalg_status >; + + SECTION( "linalg_result" ) + { + static_assert( std::is_same_v< decltype( feng::to_expected( feng::try_inverse( invertible() ) ) ), E > ); + static_assert( noexcept( feng::to_expected( std::declval< feng::linalg_result< feng::matrix< double > > >() ) ) ); + auto const ok = feng::to_expected( feng::try_inverse( invertible() ) ); + REQUIRE( ok.has_value() ); + REQUIRE( ( *ok )[0][0] == 1.0 ); + REQUIRE( ( *ok )[0][1] == -1.0 ); + REQUIRE( ( *ok )[1][1] == 2.0 ); + auto const bad = feng::to_expected( feng::try_inverse( singular() ) ); + REQUIRE( !bad.has_value() ); + REQUIRE( bad.error() == feng::linalg_status::singular ); + feng::matrix< double > const b{ 2, 1, { 3.0, 2.0 } }; + static_assert( std::is_same_v< decltype( feng::to_expected( feng::solve( invertible(), b ) ) ), E > ); + auto const s = feng::to_expected( feng::solve( invertible(), b ) ); + REQUIRE( s.has_value() ); + REQUIRE( ( *s )[0][0] == 1.0 ); + REQUIRE( ( *s )[1][0] == 1.0 ); + auto const s_bad = feng::to_expected( feng::solve( singular(), b ) ); + REQUIRE( !s_bad.has_value() ); + REQUIRE( s_bad.error() == feng::linalg_status::singular ); + } + + SECTION( "linalg_result from the factorization members" ) + { + feng::matrix< double > const b{ 2, 1, { 3.0, 2.0 } }; + auto const lu = feng::lu_factor( invertible() ); + auto const lu_singular = feng::lu_factor( singular() ); + static_assert( std::is_same_v< decltype( feng::to_expected( lu.solve( b ) ) ), E > ); + static_assert( std::is_same_v< decltype( feng::to_expected( lu.inverse() ) ), E > ); + auto const x = feng::to_expected( lu.solve( b ) ); + REQUIRE( x.has_value() ); + REQUIRE( ( *x )[0][0] == 1.0 ); + REQUIRE( ( *x )[1][0] == 1.0 ); + REQUIRE( feng::to_expected( lu_singular.solve( b ) ).error_or( feng::linalg_status::ok ) == feng::linalg_status::singular ); + auto const inv = feng::to_expected( lu.inverse() ); + REQUIRE( inv.has_value() ); + REQUIRE( *inv == feng::matrix< double >{ 2, 2, { 1.0, -1.0, -1.0, 2.0 } } ); + REQUIRE( feng::to_expected( lu_singular.inverse() ).error_or( feng::linalg_status::ok ) == feng::linalg_status::singular ); + + auto const svd = feng::svd_factor( invertible() ); + auto const svd_nonfinite = feng::svd_factor( feng::matrix< double >{ 2, 2, { 1.0, std::numeric_limits< double >::quiet_NaN(), 0.0, 1.0 } } ); + static_assert( std::is_same_v< decltype( feng::to_expected( svd.pinverse() ) ), E > ); + auto const p = feng::to_expected( svd.pinverse() ); + REQUIRE( p.has_value() ); + REQUIRE( p->row() == 2 ); + REQUIRE( p->col() == 2 ); + double const expected[2][2] = { { 1.0, -1.0 }, { -1.0, 2.0 } }; + for ( std::size_t r = 0; r != 2; ++r ) + for ( std::size_t c = 0; c != 2; ++c ) + REQUIRE( std::abs( ( *p )[r][c] - expected[r][c] ) < 1.0e-12 ); + auto const p_bad = feng::to_expected( svd_nonfinite.pinverse() ); + REQUIRE( !p_bad.has_value() ); + REQUIRE( p_bad.error() == feng::linalg_status::nonfinite ); + } + + SECTION( "factorization objects" ) + { + auto const lu = feng::to_expected( feng::lu_factor( invertible() ) ); + static_assert( std::is_same_v< std::remove_const_t< decltype( lu ) >, std::expected< feng::lu_factorization< double >, feng::linalg_status > > ); + REQUIRE( lu.has_value() ); + REQUIRE( lu->rank() == 2 ); + auto const lu_bad = feng::to_expected( feng::lu_factor( singular() ) ); + REQUIRE( !lu_bad.has_value() ); + REQUIRE( lu_bad.error() == feng::linalg_status::singular ); + + auto const svd = feng::to_expected( feng::svd_factor( invertible() ) ); + static_assert( std::is_same_v< std::remove_const_t< decltype( svd ) >, std::expected< feng::svd_factorization< double >, feng::linalg_status > > ); + REQUIRE( svd.has_value() ); + auto const svd_bad = feng::to_expected( feng::svd_factor( feng::matrix< double >{ 2, 2, { 1.0, std::numeric_limits< double >::quiet_NaN(), 0.0, 1.0 } } ) ); + REQUIRE( !svd_bad.has_value() ); + REQUIRE( svd_bad.error() == feng::linalg_status::nonfinite ); + + auto const ch = feng::to_expected( feng::cholesky_factor( invertible() ) ); + static_assert( std::is_same_v< std::remove_const_t< decltype( ch ) >, std::expected< feng::cholesky_factorization< double >, feng::linalg_status > > ); + REQUIRE( ch.has_value() ); + auto const ch_bad = feng::to_expected( feng::cholesky_factor( singular() ) ); + REQUIRE( !ch_bad.has_value() ); + REQUIRE( ch_bad.error() == feng::linalg_status::not_positive_definite ); + + auto const re = feng::to_expected( feng::row_echelon( singular() ) ); + static_assert( std::is_same_v< std::remove_const_t< decltype( re ) >, std::expected< feng::rref_result< double >, feng::linalg_status > > ); + REQUIRE( re.has_value() ); + REQUIRE( re->rank == 1 ); + auto const re_bad = feng::to_expected( feng::row_echelon( feng::matrix< double >{ 1, 1, std::numeric_limits< double >::infinity() } ) ); + REQUIRE( !re_bad.has_value() ); + REQUIRE( re_bad.error() == feng::linalg_status::nonfinite ); + } + + SECTION( "load_expected for each format" ) + { + auto const m = invertible(); + std::string const npy = s9_r5::temp( "m.npy" ), txt = s9_r5::temp( "m.txt" ), bin = s9_r5::temp( "m.bin" ); + for ( auto const& path : { npy, txt, bin } ) // fresh files under the run's TMPDIR + std::filesystem::remove( path ); + REQUIRE( m.save_as_npy( npy ) ); + REQUIRE( m.save_as_txt( txt ) ); + REQUIRE( m.save_as_binary( bin ) ); + static_assert( std::is_same_v< decltype( feng::load_expected< double >( npy, feng::io_format::npy ) ), std::expected< feng::matrix< double >, feng::io_status > > ); + static_assert( noexcept( feng::load_expected< double >( npy, feng::io_format::npy ) ) ); + static_assert( s9_r5::has_load_expected< double, feng::io_format > ); + for ( auto const& [path, format] : { std::pair{ npy, feng::io_format::npy }, std::pair{ txt, feng::io_format::txt }, std::pair{ bin, feng::io_format::binary } } ) + { + auto const r = feng::load_expected< double >( path, format ); + REQUIRE( r.has_value() ); + REQUIRE( *r == m ); + auto const bad = feng::load_expected< double >( s9_r5::missing(), format ); + REQUIRE( !bad.has_value() ); + REQUIRE( bad.error() == feng::io_status::failed ); + } + std::filesystem::remove( npy ); + std::filesystem::remove( txt ); + std::filesystem::remove( bin ); + } +} +#else +TEST_CASE( "S9-R5 status results stay available in C++20", "[S9][S9-R5]" ) +{ + static_assert( !s9_r5::has_to_expected< feng::linalg_result< feng::matrix< double > > > ); + static_assert( !s9_r5::has_to_expected< feng::lu_factorization< double > > ); + static_assert( !s9_r5::has_load_expected< double, feng::io_format > ); + + auto const ok = feng::try_inverse( s9_r5::invertible() ); + REQUIRE( ok.ok() ); + REQUIRE( ok.value[1][1] == 2.0 ); + auto const bad = feng::try_inverse( s9_r5::singular() ); + REQUIRE( !bad ); + REQUIRE( bad.status == feng::linalg_status::singular ); + REQUIRE( feng::lu_factor( s9_r5::invertible() ).status() == feng::linalg_status::ok ); + REQUIRE( feng::lu_factor( s9_r5::singular() ).status() == feng::linalg_status::singular ); + + feng::matrix< double > m; + REQUIRE( !m.load_npy( s9_r5::missing() ) ); + std::string const npy = s9_r5::temp( "c20.npy" ); + REQUIRE( s9_r5::invertible().save_as_npy( npy ) ); + REQUIRE( m.load_npy( npy ) ); + REQUIRE( m == s9_r5::invertible() ); + std::filesystem::remove( npy ); +} +#endif diff --git a/tests/cases/s9_r6.hpp b/tests/cases/s9_r6.hpp new file mode 100644 index 0000000..e6230d7 --- /dev/null +++ b/tests/cases/s9_r6.hpp @@ -0,0 +1,181 @@ +// S9-R6 (PR-12): the one elementwise transform helper. matrix_details::transform( m, f ) builds a matrix of m's +// shape whose element type is f's result (or the R of transform< R >) and whose allocator is m's rebound to it; +// the unary family goes through it and keeps its pre-S9 result types and values. +#include "./s3_alloc.hpp" + +#include +#include +#include +#include + +namespace s9_r6 +{ + using alloc = s3_alloc::tracking_allocator< double, true, true, true >; + using amat = feng::matrix< double, alloc >; + template < typename T > + using rebound = feng::matrix< T, s3_alloc::tracking_allocator< T, true, true, true > >; + + inline amat sample( int id, std::size_t r, std::size_t c ) + { + amat m{ alloc{ id }, r, c }; + for ( std::size_t i = 0; i != m.size(); ++i ) + m.data()[i] = 0.25 + 0.5 * static_cast< double >( i % 97 ) - 3.0 * static_cast< double >( i % 5 ); + return m; + } +} + +TEST_CASE( "S9-R6 transform keeps shape, allocator and result type", "[S9][S9-R6]" ) +{ + using s9_r6::amat; + using s9_r6::rebound; + s3_alloc::reset(); + { + auto const m = s9_r6::sample( 7, 3, 5 ); + + SECTION( "the helper: result type from the callable or given" ) + { + auto const t = feng::matrix_details::transform( m, []( double x ) { return static_cast< int >( x * 2.0 ); } ); + static_assert( std::is_same_v< std::remove_const_t< decltype( t ) >, rebound< int > > ); + REQUIRE( t.row() == 3 ); + REQUIRE( t.col() == 5 ); + REQUIRE( t.get_allocator().id == 7 ); + for ( std::size_t i = 0; i != m.size(); ++i ) + REQUIRE( t.data()[i] == static_cast< int >( m.data()[i] * 2.0 ) ); + auto const f = feng::matrix_details::transform< float >( m, []( double x ) { return x + 1.0; } ); + static_assert( std::is_same_v< std::remove_const_t< decltype( f ) >, rebound< float > > ); + REQUIRE( f.get_allocator().id == 7 ); + REQUIRE( f.data()[4] == static_cast< float >( m.data()[4] + 1.0 ) ); + feng::matrix< double > const empty; + REQUIRE( feng::matrix_details::transform( empty, []( double x ) { return x; } ).size() == 0 ); + } + + SECTION( "the unary family keeps its result types and the allocator" ) + { + static_assert( std::is_same_v< decltype( feng::abs( m ) ), amat > ); + static_assert( std::is_same_v< decltype( feng::sqrt( m ) ), amat > ); + static_assert( std::is_same_v< decltype( feng::sqrt( feng::matrix< int >{} ) ), feng::matrix< int > > ); + static_assert( std::is_same_v< decltype( feng::ilogb( m ) ), rebound< int > > ); + static_assert( std::is_same_v< decltype( feng::lrint( m ) ), rebound< long > > ); + static_assert( std::is_same_v< decltype( feng::llrint( m ) ), rebound< long long > > ); + static_assert( std::is_same_v< decltype( feng::lround( m ) ), rebound< long > > ); + static_assert( std::is_same_v< decltype( feng::llround( m ) ), rebound< long long > > ); + auto const a = feng::abs( m ); + auto const l = feng::lround( m ); + auto const g = feng::ilogb( m ); + REQUIRE( a.row() == 3 ); + REQUIRE( a.col() == 5 ); + REQUIRE( a.get_allocator().id == 7 ); + REQUIRE( l.get_allocator().id == 7 ); + REQUIRE( g.get_allocator().id == 7 ); + for ( std::size_t i = 0; i != m.size(); ++i ) + { + REQUIRE( a.data()[i] == std::abs( m.data()[i] ) ); + REQUIRE( l.data()[i] == std::lround( m.data()[i] ) ); + REQUIRE( g.data()[i] == std::ilogb( m.data()[i] ) ); + } + feng::matrix< int > const im{ 1, 3, { 2, 9, 10 } }; + auto const is = feng::sqrt( im ); + REQUIRE( is[0][0] == 1 ); + REQUIRE( is[0][1] == 3 ); + REQUIRE( is[0][2] == 3 ); + } + + SECTION( "a parallel-size input" ) + { + auto const big = s9_r6::sample( 9, 64, 80 ); // 5120 elements, above the parallel threshold + auto const e = feng::exp( big ); + auto const r = feng::llrint( big ); + REQUIRE( e.row() == 64 ); + REQUIRE( e.col() == 80 ); + REQUIRE( e.get_allocator().id == 9 ); + REQUIRE( r.get_allocator().id == 9 ); + for ( std::size_t i = 0; i != big.size(); ++i ) + { + REQUIRE( e.data()[i] == std::exp( big.data()[i] ) ); + REQUIRE( r.data()[i] == std::llrint( big.data()[i] ) ); + } + } + } + s3_alloc::require_balanced(); +} + +// S9-T11: the binary family (matrix-matrix, matrix-scalar, scalar-matrix) and the complex family go through the same +// helper; results keep m's type (binary, conj, proj) or the complex value type (real, imag, abs, arg, norm), and the +// allocator of the matrix operand. +TEST_CASE( "S9-R6 binary and complex families keep their values and types", "[S9][S9-R6]" ) +{ + using s9_r6::amat; + using s9_r6::rebound; + using C = std::complex< double >; + using cmat = rebound< C >; + s3_alloc::reset(); + { + auto const m = s9_r6::sample( 7, 3, 5 ); + auto n = s9_r6::sample( 8, 3, 5 ); + for ( std::size_t i = 0; i != n.size(); ++i ) + n.data()[i] = 1.5 - n.data()[i]; + + SECTION( "binary: matrix-matrix, matrix-scalar and scalar-matrix" ) + { + static_assert( std::is_same_v< decltype( feng::hypot( m, n ) ), amat > ); + static_assert( std::is_same_v< decltype( feng::hypot( m, 2.0 ) ), amat > ); + static_assert( std::is_same_v< decltype( feng::hypot( 2.0, m ) ), amat > ); + static_assert( std::is_same_v< decltype( feng::pow( m, 2 ) ), amat > ); + static_assert( std::is_same_v< decltype( feng::fma( m, n, m ) ), amat > ); + static_assert( std::is_same_v< decltype( feng::ldexp( feng::matrix< int >{}, 2.0 ) ), feng::matrix< int > > ); + auto const mm = feng::atan2( m, n ); + auto const ms = feng::atan2( m, 0.75 ); + auto const sm = feng::atan2( 0.75, m ); + auto const pi = feng::pow( m, 3 ); + auto const f3 = feng::fma( m, n, m ); + REQUIRE( mm.row() == 3 ); + REQUIRE( mm.col() == 5 ); + REQUIRE( mm.get_allocator().id == 7 ); + REQUIRE( ms.get_allocator().id == 7 ); + REQUIRE( sm.get_allocator().id == 7 ); + REQUIRE( f3.get_allocator().id == 7 ); + for ( std::size_t i = 0; i != m.size(); ++i ) + { + REQUIRE( mm.data()[i] == std::atan2( m.data()[i], n.data()[i] ) ); + REQUIRE( ms.data()[i] == std::atan2( m.data()[i], 0.75 ) ); + REQUIRE( sm.data()[i] == std::atan2( 0.75, m.data()[i] ) ); + REQUIRE( pi.data()[i] == std::pow( m.data()[i], 3 ) ); + REQUIRE( f3.data()[i] == std::fma( m.data()[i], n.data()[i], m.data()[i] ) ); + } + } + + SECTION( "complex: real, imag, abs, arg, norm, conj, proj" ) + { + cmat c{ s3_alloc::tracking_allocator< C, true, true, true >{ 5 }, 3, 5 }; + for ( std::size_t i = 0; i != c.size(); ++i ) + c.data()[i] = C{ m.data()[i], n.data()[i] }; + static_assert( std::is_same_v< decltype( feng::real( c ) ), amat > ); + static_assert( std::is_same_v< decltype( feng::imag( c ) ), amat > ); + static_assert( std::is_same_v< decltype( feng::abs( c ) ), amat > ); + static_assert( std::is_same_v< decltype( feng::arg( c ) ), amat > ); + static_assert( std::is_same_v< decltype( feng::norm( c ) ), amat > ); + static_assert( std::is_same_v< decltype( feng::conj( c ) ), cmat > ); + static_assert( std::is_same_v< decltype( feng::proj( c ) ), cmat > ); + static_assert( std::is_same_v< decltype( feng::real( feng::matrix< std::complex< float > >{} ) ), feng::matrix< float > > ); + auto const re = feng::real( c ); + auto const im = feng::imag( c ); + auto const ar = feng::arg( c ); + auto const no = feng::norm( c ); + auto const cj = feng::conj( c ); + REQUIRE( re.row() == 3 ); + REQUIRE( re.col() == 5 ); + REQUIRE( re.get_allocator().id == 5 ); + REQUIRE( no.get_allocator().id == 5 ); + REQUIRE( cj.get_allocator().id == 5 ); + for ( std::size_t i = 0; i != c.size(); ++i ) + { + REQUIRE( re.data()[i] == std::real( c.data()[i] ) ); + REQUIRE( im.data()[i] == std::imag( c.data()[i] ) ); + REQUIRE( ar.data()[i] == std::arg( c.data()[i] ) ); + REQUIRE( no.data()[i] == std::norm( c.data()[i] ) ); + REQUIRE( cj.data()[i] == std::conj( c.data()[i] ) ); + } + } + } + s3_alloc::require_balanced(); +} diff --git a/tests/compile_fail/S3/bool.cc b/tests/compile_fail/S3/bool.cc new file mode 100644 index 0000000..84781d9 --- /dev/null +++ b/tests/compile_fail/S3/bool.cc @@ -0,0 +1,11 @@ +// S3-R3: bool is not a matrix_element (D-012, D-017). +// expect: matrix_element +// expect-gcc: required for the satisfaction of 'matrix_element' +// expect-clang: does not satisfy 'matrix_element' +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix m; + (void)m; +} diff --git a/tests/compile_fail/S3/string.cc b/tests/compile_fail/S3/string.cc new file mode 100644 index 0000000..8b81451 --- /dev/null +++ b/tests/compile_fail/S3/string.cc @@ -0,0 +1,13 @@ +// S3-R3: std::string has a throwing copy constructor, so it is not a matrix_element (D-012). +// expect: matrix_element +// expect-gcc: required for the satisfaction of 'matrix_element' +// expect-clang: does not satisfy 'matrix_element' +#include + +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix m; + (void)m; +} diff --git a/tests/compile_fail/S3/throwing_copy.cc b/tests/compile_fail/S3/throwing_copy.cc new file mode 100644 index 0000000..5ceb768 --- /dev/null +++ b/tests/compile_fail/S3/throwing_copy.cc @@ -0,0 +1,20 @@ +// S3-R3: a type whose copy constructor is not noexcept is not a matrix_element (D-012). +// expect: matrix_element +// expect-gcc: required for the satisfaction of 'matrix_element' +// expect-clang: does not satisfy 'matrix_element' +#include "../../../matrix.hpp" + +struct throwing_copy +{ + throwing_copy() noexcept = default; + throwing_copy( throwing_copy const& ) noexcept( false ) {} + throwing_copy( throwing_copy&& ) noexcept = default; + throwing_copy& operator=( throwing_copy const& ) noexcept = default; + throwing_copy& operator=( throwing_copy&& ) noexcept = default; +}; + +int main() +{ + feng::matrix m; + (void)m; +} diff --git a/tests/compile_fail/S4/make_mutable_view_const_rvalue.cc b/tests/compile_fail/S4/make_mutable_view_const_rvalue.cc new file mode 100644 index 0000000..9d767a5 --- /dev/null +++ b/tests/compile_fail/S4/make_mutable_view_const_rvalue.cc @@ -0,0 +1,18 @@ +// S4-R3: make_mutable_view of a moved const owner is deleted (F06). +// expect: deleted +// expect-gcc: use of deleted function 'feng::mutable_matrix_view feng::make_mutable_view(const matrix&&, +// expect-clang: call to deleted function 'make_mutable_view' +#include "../../../matrix.hpp" + +#include +#include + +int main() +{ + using range = std::pair; + feng::matrix m{ 3, 3 }; + feng::matrix const& cm = m; + (void)cm; + auto v = feng::make_mutable_view( std::move( cm ), { 0, 1 }, { 0, 1 } ); + (void)v; +} diff --git a/tests/compile_fail/S4/make_mutable_view_temporary.cc b/tests/compile_fail/S4/make_mutable_view_temporary.cc new file mode 100644 index 0000000..a914937 --- /dev/null +++ b/tests/compile_fail/S4/make_mutable_view_temporary.cc @@ -0,0 +1,18 @@ +// S4-R3: make_mutable_view of a temporary owner is deleted (F06). +// expect: deleted +// expect-gcc: use of deleted function 'feng::mutable_matrix_view feng::make_mutable_view(matrix&&, +// expect-clang: call to deleted function 'make_mutable_view' +#include "../../../matrix.hpp" + +#include +#include + +int main() +{ + using range = std::pair; + feng::matrix m{ 3, 3 }; + feng::matrix const& cm = m; + (void)cm; + auto v = feng::make_mutable_view( feng::matrix{ 3, 3 }, { 0, 1 }, { 0, 1 } ); + (void)v; +} diff --git a/tests/compile_fail/S4/make_view_const_rvalue.cc b/tests/compile_fail/S4/make_view_const_rvalue.cc new file mode 100644 index 0000000..91d93a1 --- /dev/null +++ b/tests/compile_fail/S4/make_view_const_rvalue.cc @@ -0,0 +1,18 @@ +// S4-R3: make_view of a moved const owner is deleted (F06). +// expect: deleted +// expect-gcc: use of deleted function 'feng::matrix_view feng::make_view(const matrix&&, +// expect-clang: call to deleted function 'make_view' +#include "../../../matrix.hpp" + +#include +#include + +int main() +{ + using range = std::pair; + feng::matrix m{ 3, 3 }; + feng::matrix const& cm = m; + (void)cm; + auto v = feng::make_view( std::move( cm ), { 0, 1 }, { 0, 1 } ); + (void)v; +} diff --git a/tests/compile_fail/S4/make_view_temporary.cc b/tests/compile_fail/S4/make_view_temporary.cc new file mode 100644 index 0000000..7d0b50a --- /dev/null +++ b/tests/compile_fail/S4/make_view_temporary.cc @@ -0,0 +1,11 @@ +// S4-R3: make_view of a temporary owner is a deleted overload (F06). +// expect: deleted +// expect-gcc: use of deleted function 'feng::matrix_view feng::make_view(matrix&& +// expect-clang: call to deleted function 'make_view' +#include "../../../matrix.hpp" + +int main() +{ + auto v = feng::make_view( feng::matrix{ 3, 3 }, { 0, 1 }, { 0, 1 } ); + (void)v; +} diff --git a/tests/compile_fail/S4/matrix_view_const_rvalue.cc b/tests/compile_fail/S4/matrix_view_const_rvalue.cc new file mode 100644 index 0000000..1b8ea80 --- /dev/null +++ b/tests/compile_fail/S4/matrix_view_const_rvalue.cc @@ -0,0 +1,18 @@ +// S4-R3: the matrix_view constructor from a moved const owner is deleted (F06). +// expect: deleted +// expect-gcc: use of deleted function 'feng::matrix_view::matrix_view(const matrix_type&&, +// expect-clang: call to deleted constructor of 'feng::matrix_view>' +#include "../../../matrix.hpp" + +#include +#include + +int main() +{ + using range = std::pair; + feng::matrix m{ 3, 3 }; + feng::matrix const& cm = m; + (void)cm; + feng::matrix_view> v( std::move( cm ), range{ 0, 1 }, range{ 0, 1 } ); + (void)v; +} diff --git a/tests/compile_fail/S4/matrix_view_temporary.cc b/tests/compile_fail/S4/matrix_view_temporary.cc new file mode 100644 index 0000000..833e40c --- /dev/null +++ b/tests/compile_fail/S4/matrix_view_temporary.cc @@ -0,0 +1,18 @@ +// S4-R3: the matrix_view constructor from a temporary owner is deleted (F06). +// expect: deleted +// expect-gcc: use of deleted function 'feng::matrix_view::matrix_view(matrix_type&&, +// expect-clang: call to deleted constructor of 'feng::matrix_view>' +#include "../../../matrix.hpp" + +#include +#include + +int main() +{ + using range = std::pair; + feng::matrix m{ 3, 3 }; + feng::matrix const& cm = m; + (void)cm; + feng::matrix_view> v( feng::matrix{ 3, 3 }, range{ 0, 1 }, range{ 0, 1 } ); + (void)v; +} diff --git a/tests/compile_fail/S4/mutable_matrix_view_const_owner.cc b/tests/compile_fail/S4/mutable_matrix_view_const_owner.cc new file mode 100644 index 0000000..b161de9 --- /dev/null +++ b/tests/compile_fail/S4/mutable_matrix_view_const_owner.cc @@ -0,0 +1,18 @@ +// S4-R3: the mutable_matrix_view constructor from a const owner is deleted (F06). +// expect: deleted +// expect-gcc: use of deleted function 'feng::mutable_matrix_view::mutable_matrix_view(const matrix_type&, +// expect-clang: call to deleted constructor of 'feng::mutable_matrix_view>' +#include "../../../matrix.hpp" + +#include +#include + +int main() +{ + using range = std::pair; + feng::matrix m{ 3, 3 }; + feng::matrix const& cm = m; + (void)cm; + feng::mutable_matrix_view> v( cm, range{ 0, 1 }, range{ 0, 1 } ); + (void)v; +} diff --git a/tests/compile_fail/S4/mutable_matrix_view_const_rvalue.cc b/tests/compile_fail/S4/mutable_matrix_view_const_rvalue.cc new file mode 100644 index 0000000..c31e0d4 --- /dev/null +++ b/tests/compile_fail/S4/mutable_matrix_view_const_rvalue.cc @@ -0,0 +1,18 @@ +// S4-R3: the mutable_matrix_view constructor from a moved const owner is deleted (F06). +// expect: deleted +// expect-gcc: use of deleted function 'feng::mutable_matrix_view::mutable_matrix_view(const matrix_type&&, +// expect-clang: call to deleted constructor of 'feng::mutable_matrix_view>' +#include "../../../matrix.hpp" + +#include +#include + +int main() +{ + using range = std::pair; + feng::matrix m{ 3, 3 }; + feng::matrix const& cm = m; + (void)cm; + feng::mutable_matrix_view> v( std::move( cm ), range{ 0, 1 }, range{ 0, 1 } ); + (void)v; +} diff --git a/tests/compile_fail/S4/mutable_matrix_view_temporary.cc b/tests/compile_fail/S4/mutable_matrix_view_temporary.cc new file mode 100644 index 0000000..9b453e4 --- /dev/null +++ b/tests/compile_fail/S4/mutable_matrix_view_temporary.cc @@ -0,0 +1,18 @@ +// S4-R3: the mutable_matrix_view constructor from a temporary owner is deleted (F06). +// expect: deleted +// expect-gcc: use of deleted function 'feng::mutable_matrix_view::mutable_matrix_view(matrix_type&&, +// expect-clang: call to deleted constructor of 'feng::mutable_matrix_view>' +#include "../../../matrix.hpp" + +#include +#include + +int main() +{ + using range = std::pair; + feng::matrix m{ 3, 3 }; + feng::matrix const& cm = m; + (void)cm; + feng::mutable_matrix_view> v( feng::matrix{ 3, 3 }, range{ 0, 1 }, range{ 0, 1 } ); + (void)v; +} diff --git a/tests/compile_fail/S4/mutable_view_const_owner.cc b/tests/compile_fail/S4/mutable_view_const_owner.cc new file mode 100644 index 0000000..5fa3178 --- /dev/null +++ b/tests/compile_fail/S4/mutable_view_const_owner.cc @@ -0,0 +1,18 @@ +// S4-R3: make_mutable_view of a const owner is deleted (F06). +// expect: deleted +// expect-gcc: use of deleted function 'feng::mutable_matrix_view feng::make_mutable_view(const matrix&, +// expect-clang: call to deleted function 'make_mutable_view' +#include "../../../matrix.hpp" + +#include +#include + +int main() +{ + using range = std::pair; + feng::matrix m{ 3, 3 }; + feng::matrix const& cm = m; + (void)cm; + auto v = feng::make_mutable_view( cm, { 0, 1 }, { 0, 1 } ); + (void)v; +} diff --git a/tests/compile_fail/S7/det_int.cc b/tests/compile_fail/S7/det_int.cc new file mode 100644 index 0000000..0786740 --- /dev/null +++ b/tests/compile_fail/S7/det_int.cc @@ -0,0 +1,12 @@ +// S7-R2, D-027: det of an integral matrix fails the linalg_element constraint (S9-R2, D-034). +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 3, 1 }; + auto d = a.det(); + (void)d; +} diff --git a/tests/compile_fail/S7/inverse_int.cc b/tests/compile_fail/S7/inverse_int.cc new file mode 100644 index 0000000..fbbfbd3 --- /dev/null +++ b/tests/compile_fail/S7/inverse_int.cc @@ -0,0 +1,13 @@ +// S7-R2, D-027: inverse of an integral matrix fails the linalg_element constraint (S9-R2, D-034). +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 2, 2, 1L }; + feng::matrix out; + auto st = feng::inverse( a, out ); + (void)st; +} diff --git a/tests/compile_fail/S7/lu_factor_int.cc b/tests/compile_fail/S7/lu_factor_int.cc new file mode 100644 index 0000000..b9b5daa --- /dev/null +++ b/tests/compile_fail/S7/lu_factor_int.cc @@ -0,0 +1,12 @@ +// S7-R1, D-027: lu_factor of an integral matrix fails the linalg_element constraint (S9-R2, D-034). +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 3, 1 }; + auto f = feng::lu_factor( a ); + (void)f; +} diff --git a/tests/compile_fail/S9/arg_real.cc b/tests/compile_fail/S9/arg_real.cc new file mode 100644 index 0000000..d7bac30 --- /dev/null +++ b/tests/compile_fail/S9/arg_real.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: arg of a real matrix fails the ComplexMatrix constraint. +// expect: ComplexMatrix +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 2, 2, 1.0 }; + auto c = feng::arg( a ); + (void)c; +} diff --git a/tests/compile_fail/S9/cholesky_decomposition_int.cc b/tests/compile_fail/S9/cholesky_decomposition_int.cc new file mode 100644 index 0000000..1de076b --- /dev/null +++ b/tests/compile_fail/S9/cholesky_decomposition_int.cc @@ -0,0 +1,13 @@ +// S9-R2, D-034: the legacy cholesky_decomposition of an integral matrix fails the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 3, 1 }; + feng::matrix l; + auto st = feng::cholesky_decomposition( a, l ); + (void)st; +} diff --git a/tests/compile_fail/S9/cholesky_int.cc b/tests/compile_fail/S9/cholesky_int.cc new file mode 100644 index 0000000..638449f --- /dev/null +++ b/tests/compile_fail/S9/cholesky_int.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: cholesky_factor of an integral matrix fails the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 3, 1 }; + auto f = feng::cholesky_factor( a ); + (void)f; +} diff --git a/tests/compile_fail/S9/conj_real.cc b/tests/compile_fail/S9/conj_real.cc new file mode 100644 index 0000000..44c88b7 --- /dev/null +++ b/tests/compile_fail/S9/conj_real.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: conj of a real matrix fails the ComplexMatrix constraint. +// expect: ComplexMatrix +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 2, 2, 1.0 }; + auto c = feng::conj( a ); + (void)c; +} diff --git a/tests/compile_fail/S9/ctranspose_real.cc b/tests/compile_fail/S9/ctranspose_real.cc new file mode 100644 index 0000000..1608ae7 --- /dev/null +++ b/tests/compile_fail/S9/ctranspose_real.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: ctranspose of a real matrix fails the ComplexMatrix constraint. +// expect: ComplexMatrix +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 2, 3, 1.0 }; + auto c = feng::ctranspose( a ); + (void)c; +} diff --git a/tests/compile_fail/S9/det_int.cc b/tests/compile_fail/S9/det_int.cc new file mode 100644 index 0000000..9b46de5 --- /dev/null +++ b/tests/compile_fail/S9/det_int.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: det of an integral matrix fails the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 3, 1 }; + auto d = feng::det( a ); + (void)d; +} diff --git a/tests/compile_fail/S9/expm_int.cc b/tests/compile_fail/S9/expm_int.cc new file mode 100644 index 0000000..792f425 --- /dev/null +++ b/tests/compile_fail/S9/expm_int.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: expm of an integral matrix fails the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 3, 1 }; + auto e = feng::expm( a ); + (void)e; +} diff --git a/tests/compile_fail/S9/expm_status_int.cc b/tests/compile_fail/S9/expm_status_int.cc new file mode 100644 index 0000000..a033171 --- /dev/null +++ b/tests/compile_fail/S9/expm_status_int.cc @@ -0,0 +1,13 @@ +// S9-R2, D-034: the status overload expm( A, out ) on an integral matrix fails the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 3, 1 }; + feng::matrix out; + auto st = feng::expm( a, out ); + (void)st; +} diff --git a/tests/compile_fail/S9/imag_real.cc b/tests/compile_fail/S9/imag_real.cc new file mode 100644 index 0000000..5dc7d5c --- /dev/null +++ b/tests/compile_fail/S9/imag_real.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: imag of a real matrix fails the ComplexMatrix constraint. +// expect: ComplexMatrix +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 2, 2, 1.0 }; + auto c = feng::imag( a ); + (void)c; +} diff --git a/tests/compile_fail/S9/inverse_int.cc b/tests/compile_fail/S9/inverse_int.cc new file mode 100644 index 0000000..b252803 --- /dev/null +++ b/tests/compile_fail/S9/inverse_int.cc @@ -0,0 +1,14 @@ +// S9-R2, D-034: inverse of an integral matrix, member and free, fails the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 3, 1 }; + auto m = a.inverse(); + auto f = feng::inverse( a ); + (void)m; + (void)f; +} diff --git a/tests/compile_fail/S9/inverse_status_int.cc b/tests/compile_fail/S9/inverse_status_int.cc new file mode 100644 index 0000000..fc7fd14 --- /dev/null +++ b/tests/compile_fail/S9/inverse_status_int.cc @@ -0,0 +1,13 @@ +// S9-R2, D-034: the status overload inverse( A, out ) on an integral matrix fails the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 2, 2, 1L }; + feng::matrix out; + auto st = feng::inverse( a, out ); + (void)st; +} diff --git a/tests/compile_fail/S9/lu_decomposition_int.cc b/tests/compile_fail/S9/lu_decomposition_int.cc new file mode 100644 index 0000000..ba8b6d8 --- /dev/null +++ b/tests/compile_fail/S9/lu_decomposition_int.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: the legacy lu_decomposition of an integral matrix fails the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 3, 1 }; + auto f = feng::lu_decomposition( a ); + (void)f; +} diff --git a/tests/compile_fail/S9/lu_factor_int.cc b/tests/compile_fail/S9/lu_factor_int.cc new file mode 100644 index 0000000..24711f3 --- /dev/null +++ b/tests/compile_fail/S9/lu_factor_int.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: lu_factor of an integral matrix fails the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 3, 1 }; + auto f = feng::lu_factor( a ); + (void)f; +} diff --git a/tests/compile_fail/S9/lu_solver_int.cc b/tests/compile_fail/S9/lu_solver_int.cc new file mode 100644 index 0000000..71b33d9 --- /dev/null +++ b/tests/compile_fail/S9/lu_solver_int.cc @@ -0,0 +1,13 @@ +// S9-R2, D-034: the legacy lu_solver of an integral matrix fails the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 3, 1 }; + feng::matrix const b{ 3, 1, 1 }; + auto x = feng::lu_solver( a, b ); + (void)x; +} diff --git a/tests/compile_fail/S9/nodiscard_inverse_status.cc b/tests/compile_fail/S9/nodiscard_inverse_status.cc new file mode 100644 index 0000000..4820b9d --- /dev/null +++ b/tests/compile_fail/S9/nodiscard_inverse_status.cc @@ -0,0 +1,12 @@ +// S9-R1, D-034: discarding the linalg_status of inverse( A, out ) is diagnosed; under -Werror=unused-result it is an +// error. +// flags: -Werror=unused-result +// expect: nodiscard +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 2, 2, { 2.0, 0.0, 0.0, 2.0 } }; + feng::matrix out; + feng::inverse( a, out ); +} diff --git a/tests/compile_fail/S9/nodiscard_load_npy.cc b/tests/compile_fail/S9/nodiscard_load_npy.cc new file mode 100644 index 0000000..ee3b4f6 --- /dev/null +++ b/tests/compile_fail/S9/nodiscard_load_npy.cc @@ -0,0 +1,10 @@ +// S9-R1, D-034: discarding the bool of a loader is diagnosed; under -Werror=unused-result it is an error. +// flags: -Werror=unused-result +// expect: nodiscard +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix m; + m.load_npy( "./images/64.npy" ); +} diff --git a/tests/compile_fail/S9/nodiscard_lu_factor.cc b/tests/compile_fail/S9/nodiscard_lu_factor.cc new file mode 100644 index 0000000..7bca092 --- /dev/null +++ b/tests/compile_fail/S9/nodiscard_lu_factor.cc @@ -0,0 +1,10 @@ +// S9-R1, D-034: discarding a factorization is diagnosed; under -Werror=unused-result it is an error. +// flags: -Werror=unused-result +// expect: nodiscard +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 2, 2, { 2.0, 0.0, 0.0, 2.0 } }; + feng::lu_factor( a ); +} diff --git a/tests/compile_fail/S9/nodiscard_save.cc b/tests/compile_fail/S9/nodiscard_save.cc new file mode 100644 index 0000000..664e850 --- /dev/null +++ b/tests/compile_fail/S9/nodiscard_save.cc @@ -0,0 +1,10 @@ +// S9-R1, D-034: discarding the bool of a writer is diagnosed; under -Werror=unused-result it is an error. +// flags: -Werror=unused-result +// expect: nodiscard +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const m{ 2, 2, 1.0 }; + m.save_as_npy( "s9_nodiscard_save.npy" ); +} diff --git a/tests/compile_fail/S9/nodiscard_transpose.cc b/tests/compile_fail/S9/nodiscard_transpose.cc new file mode 100644 index 0000000..1e427e0 --- /dev/null +++ b/tests/compile_fail/S9/nodiscard_transpose.cc @@ -0,0 +1,12 @@ +// S9-R1, D-034: discarding a pure function's result, free or member, is diagnosed; under -Werror=unused-result it +// is an error. +// flags: -Werror=unused-result +// expect: nodiscard +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 2, 3, 1.0 }; + feng::transpose( a ); + a.transpose(); +} diff --git a/tests/compile_fail/S9/norm_real.cc b/tests/compile_fail/S9/norm_real.cc new file mode 100644 index 0000000..9929bcc --- /dev/null +++ b/tests/compile_fail/S9/norm_real.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: norm of a real matrix fails the ComplexMatrix constraint. +// expect: ComplexMatrix +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 2, 2, 1.0 }; + auto c = feng::norm( a ); + (void)c; +} diff --git a/tests/compile_fail/S9/pinv_int.cc b/tests/compile_fail/S9/pinv_int.cc new file mode 100644 index 0000000..61f944e --- /dev/null +++ b/tests/compile_fail/S9/pinv_int.cc @@ -0,0 +1,14 @@ +// S9-R2, D-034: the legacy pinv and svd_inverse of an integral matrix fail the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 2, 1 }; + auto p = feng::pinv( a ); + auto q = feng::svd_inverse( a ); + (void)p; + (void)q; +} diff --git a/tests/compile_fail/S9/pinverse_int.cc b/tests/compile_fail/S9/pinverse_int.cc new file mode 100644 index 0000000..6dfe15a --- /dev/null +++ b/tests/compile_fail/S9/pinverse_int.cc @@ -0,0 +1,15 @@ +// S9-R2, D-034: pinverse of an integral matrix, status and value forms, fails the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 2, 1 }; + feng::matrix out; + auto st = feng::pinverse( a, out ); + auto p = feng::pinverse( a ); + (void)st; + (void)p; +} diff --git a/tests/compile_fail/S9/polar_real.cc b/tests/compile_fail/S9/polar_real.cc new file mode 100644 index 0000000..938f3d9 --- /dev/null +++ b/tests/compile_fail/S9/polar_real.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: polar of a real matrix fails the ComplexMatrix constraint. +// expect: ComplexMatrix +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 2, 2, 1.0 }; + auto c = feng::polar( a ); + (void)c; +} diff --git a/tests/compile_fail/S9/proj_real.cc b/tests/compile_fail/S9/proj_real.cc new file mode 100644 index 0000000..4a5e7bf --- /dev/null +++ b/tests/compile_fail/S9/proj_real.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: proj of a real matrix fails the ComplexMatrix constraint. +// expect: ComplexMatrix +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 2, 2, 1.0 }; + auto c = feng::proj( a ); + (void)c; +} diff --git a/tests/compile_fail/S9/real_real.cc b/tests/compile_fail/S9/real_real.cc new file mode 100644 index 0000000..c491a7f --- /dev/null +++ b/tests/compile_fail/S9/real_real.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: real of a real matrix fails the ComplexMatrix constraint. +// expect: ComplexMatrix +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 2, 2, 1.0 }; + auto c = feng::real( a ); + (void)c; +} diff --git a/tests/compile_fail/S9/row_echelon_int.cc b/tests/compile_fail/S9/row_echelon_int.cc new file mode 100644 index 0000000..cad4fda --- /dev/null +++ b/tests/compile_fail/S9/row_echelon_int.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: row_echelon of an integral matrix fails the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 4, 1 }; + auto r = feng::row_echelon( a ); + (void)r; +} diff --git a/tests/compile_fail/S9/rref_int.cc b/tests/compile_fail/S9/rref_int.cc new file mode 100644 index 0000000..9f217c1 --- /dev/null +++ b/tests/compile_fail/S9/rref_int.cc @@ -0,0 +1,14 @@ +// S9-R2, D-034: the legacy rref and gauss_jordan_elimination of an integral matrix fail the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 4, 1 }; + auto r = feng::rref( a ); + auto g = feng::gauss_jordan_elimination( a ); + (void)r; + (void)g; +} diff --git a/tests/compile_fail/S9/solve_int.cc b/tests/compile_fail/S9/solve_int.cc new file mode 100644 index 0000000..f831944 --- /dev/null +++ b/tests/compile_fail/S9/solve_int.cc @@ -0,0 +1,13 @@ +// S9-R2, D-034: solve with an integral matrix fails the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 3, 1 }; + feng::matrix const b{ 3, 1, 1 }; + auto x = feng::solve( a, b ); + (void)x; +} diff --git a/tests/compile_fail/S9/svd_factor_int.cc b/tests/compile_fail/S9/svd_factor_int.cc new file mode 100644 index 0000000..f08a884 --- /dev/null +++ b/tests/compile_fail/S9/svd_factor_int.cc @@ -0,0 +1,12 @@ +// S9-R2, D-034: svd_factor of an integral matrix fails the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 2, 1 }; + auto f = feng::svd_factor( a ); + (void)f; +} diff --git a/tests/compile_fail/S9/svd_int.cc b/tests/compile_fail/S9/svd_int.cc new file mode 100644 index 0000000..fc16d0a --- /dev/null +++ b/tests/compile_fail/S9/svd_int.cc @@ -0,0 +1,14 @@ +// S9-R2, D-034: the legacy singular_value_decomposition and svd of an integral matrix fail the linalg_element constraint. +// expect: linalg_element +// expect-gcc: constraints not satisfied +// expect-clang: does not satisfy +#include "../../../matrix.hpp" + +int main() +{ + feng::matrix const a{ 3, 2, 1 }; + auto f = feng::singular_value_decomposition( a ); + auto g = feng::svd( a ); + (void)f; + (void)g; +} diff --git a/tests/config/tu_a.cc b/tests/config/tu_a.cc new file mode 100644 index 0000000..b6a44ce --- /dev/null +++ b/tests/config/tu_a.cc @@ -0,0 +1,37 @@ +// S9-R3, D-033: one half of the two-TU program of `tools/check.sh config`. The lane compiles it with +// -DCONFIG_EXPECT_PARALLEL=0|1 next to the configuration macros under test; main checks feng::parallel_mode in this +// TU and in tu_b.cc and the value tu_b computes with a matrix. Exit 0 only when all three hold. +#include "../../matrix.hpp" + +#include +#include + +#ifndef CONFIG_EXPECT_PARALLEL +#error "compile with -DCONFIG_EXPECT_PARALLEL=0 or 1" +#endif + +std::uint_least64_t tu_b_parallel_mode(); +double tu_b_trace_of_sum(); + +int main() +{ + int rc = 0; + if ( feng::parallel_mode != CONFIG_EXPECT_PARALLEL ) + { + std::printf( "tu_a: parallel_mode %d, expected %d\n", int( feng::parallel_mode ), int( CONFIG_EXPECT_PARALLEL ) ); + rc = 1; + } + if ( tu_b_parallel_mode() != feng::parallel_mode ) + { + std::printf( "tu_b: parallel_mode %d, tu_a %d\n", int( tu_b_parallel_mode() ), int( feng::parallel_mode ) ); + rc = 1; + } + double const t = tu_b_trace_of_sum(); + if ( t != 12.0 ) + { + std::printf( "tu_b: trace %g, expected 12\n", t ); + rc = 1; + } + if ( rc == 0 ) std::printf( "config: ok\n" ); + return rc; +} diff --git a/tests/config/tu_b.cc b/tests/config/tu_b.cc new file mode 100644 index 0000000..7531d27 --- /dev/null +++ b/tests/config/tu_b.cc @@ -0,0 +1,14 @@ +// S9-R3, D-033: the other half of the two-TU program of `tools/check.sh config`; it uses a matrix. +#include "../../matrix.hpp" + +#include + +std::uint_least64_t tu_b_parallel_mode() { return feng::parallel_mode; } + +double tu_b_trace_of_sum() +{ + feng::matrix a{ 3, 3, 1.0 }; + feng::matrix b{ 3, 3, 3.0 }; + feng::matrix const c = a + b; + return c( 0, 0 ) + c( 1, 1 ) + c( 2, 2 ); +} diff --git a/tests/fixtures/s5/bool.npy b/tests/fixtures/s5/bool.npy new file mode 100644 index 0000000..9c7669e Binary files /dev/null and b/tests/fixtures/s5/bool.npy differ diff --git a/tests/fixtures/s5/c16_be.npy b/tests/fixtures/s5/c16_be.npy new file mode 100644 index 0000000..97c7b36 Binary files /dev/null and b/tests/fixtures/s5/c16_be.npy differ diff --git a/tests/fixtures/s5/c16_le.npy b/tests/fixtures/s5/c16_le.npy new file mode 100644 index 0000000..472b934 Binary files /dev/null and b/tests/fixtures/s5/c16_le.npy differ diff --git a/tests/fixtures/s5/c8_be.npy b/tests/fixtures/s5/c8_be.npy new file mode 100644 index 0000000..ec14c7c Binary files /dev/null and b/tests/fixtures/s5/c8_be.npy differ diff --git a/tests/fixtures/s5/c8_le.npy b/tests/fixtures/s5/c8_le.npy new file mode 100644 index 0000000..0e1994e Binary files /dev/null and b/tests/fixtures/s5/c8_le.npy differ diff --git a/tests/fixtures/s5/f2.npy b/tests/fixtures/s5/f2.npy new file mode 100644 index 0000000..45be057 Binary files /dev/null and b/tests/fixtures/s5/f2.npy differ diff --git a/tests/fixtures/s5/f4_be.npy b/tests/fixtures/s5/f4_be.npy new file mode 100644 index 0000000..d31125b Binary files /dev/null and b/tests/fixtures/s5/f4_be.npy differ diff --git a/tests/fixtures/s5/f4_le.npy b/tests/fixtures/s5/f4_le.npy new file mode 100644 index 0000000..edfeb59 Binary files /dev/null and b/tests/fixtures/s5/f4_le.npy differ diff --git a/tests/fixtures/s5/f8_be.npy b/tests/fixtures/s5/f8_be.npy new file mode 100644 index 0000000..8e2a382 Binary files /dev/null and b/tests/fixtures/s5/f8_be.npy differ diff --git a/tests/fixtures/s5/f8_le.npy b/tests/fixtures/s5/f8_le.npy new file mode 100644 index 0000000..568e6d4 Binary files /dev/null and b/tests/fixtures/s5/f8_le.npy differ diff --git a/tests/fixtures/s5/f8_v2.npy b/tests/fixtures/s5/f8_v2.npy new file mode 100644 index 0000000..92c7c8d Binary files /dev/null and b/tests/fixtures/s5/f8_v2.npy differ diff --git a/tests/fixtures/s5/f8_v3.npy b/tests/fixtures/s5/f8_v3.npy new file mode 100644 index 0000000..cc83368 Binary files /dev/null and b/tests/fixtures/s5/f8_v3.npy differ diff --git a/tests/fixtures/s5/fortran_2x3.npy b/tests/fixtures/s5/fortran_2x3.npy new file mode 100644 index 0000000..099bf2f Binary files /dev/null and b/tests/fixtures/s5/fortran_2x3.npy differ diff --git a/tests/fixtures/s5/i1.npy b/tests/fixtures/s5/i1.npy new file mode 100644 index 0000000..220534a Binary files /dev/null and b/tests/fixtures/s5/i1.npy differ diff --git a/tests/fixtures/s5/i2_be.npy b/tests/fixtures/s5/i2_be.npy new file mode 100644 index 0000000..1b43785 Binary files /dev/null and b/tests/fixtures/s5/i2_be.npy differ diff --git a/tests/fixtures/s5/i2_le.npy b/tests/fixtures/s5/i2_le.npy new file mode 100644 index 0000000..836de4e Binary files /dev/null and b/tests/fixtures/s5/i2_le.npy differ diff --git a/tests/fixtures/s5/i4_be.npy b/tests/fixtures/s5/i4_be.npy new file mode 100644 index 0000000..9db497e Binary files /dev/null and b/tests/fixtures/s5/i4_be.npy differ diff --git a/tests/fixtures/s5/i4_le.npy b/tests/fixtures/s5/i4_le.npy new file mode 100644 index 0000000..d958973 Binary files /dev/null and b/tests/fixtures/s5/i4_le.npy differ diff --git a/tests/fixtures/s5/i8_be.npy b/tests/fixtures/s5/i8_be.npy new file mode 100644 index 0000000..bf601b3 Binary files /dev/null and b/tests/fixtures/s5/i8_be.npy differ diff --git a/tests/fixtures/s5/i8_le.npy b/tests/fixtures/s5/i8_le.npy new file mode 100644 index 0000000..ba666ec Binary files /dev/null and b/tests/fixtures/s5/i8_le.npy differ diff --git a/tests/fixtures/s5/make_npy.py b/tests/fixtures/s5/make_npy.py new file mode 100644 index 0000000..416f17f --- /dev/null +++ b/tests/fixtures/s5/make_npy.py @@ -0,0 +1,57 @@ +#!/usr/bin/env python3 +"""Generate the S5-R1 NPY fixtures with numpy. Run from the repo root: + python3 tests/fixtures/s5/make_npy.py +The values are fixed so tests/cases/s5_r1.hpp can compare against them.""" +import os +import numpy as np + +here = os.path.dirname(os.path.abspath(__file__)) +base = np.array([[1.5, -2.25, 3.0], [4.125, 5.0, -6.5]]) + + +def save(name, arr, **kw): + np.save(os.path.join(here, name), arr, **kw) + + +save("f8_le.npy", base.astype("f8")) +save("f4_le.npy", base.astype("f4")) +save("c8_le.npy", np.array(cx, dtype=" `, files relative +to this directory. Case names are `__`; the element type is the second field of the name. +Ops and their files (all C-order, little-endian ` max(a.shape) * EPS * s[0]] + return float(s[0] / s[-1]) + + +def tol_of(x): + """two significant digits, rounded up, so the manifest text is short and stable.""" + return float("%.1e" % (x * 1.05)) + + +# ---- exact elimination with Fractions (Gaussian rationals for complex) ---- +class Q: + """a + b i with Fraction parts.""" + + __slots__ = ("re", "im") + + def __init__(self, re, im=Fraction(0)): + self.re, self.im = Fraction(re), Fraction(im) + + def __sub__(self, o): + return Q(self.re - o.re, self.im - o.im) + + def __mul__(self, o): + return Q(self.re * o.re - self.im * o.im, self.re * o.im + self.im * o.re) + + def __truediv__(self, o): + d = o.re * o.re + o.im * o.im + return Q((self.re * o.re + self.im * o.im) / d, (self.im * o.re - self.re * o.im) / d) + + def iszero(self): + return self.re == 0 and self.im == 0 + + +def to_q(a): + return [[Q(Fraction(float(np.real(x))), Fraction(float(np.imag(x)))) for x in row] for row in a] + + +def exact_rref(a): + m = to_q(a) + rows, cols = len(m), len(m[0]) + r = 0 + for c in range(cols): + p = next((i for i in range(r, rows) if not m[i][c].iszero()), None) + if p is None: + continue + m[r], m[p] = m[p], m[r] + piv = m[r][c] + m[r] = [x / piv for x in m[r]] + for i in range(rows): + if i != r and not m[i][c].iszero(): + f = m[i][c] + m[i] = [x - f * y for x, y in zip(m[i], m[r])] + r += 1 + if r == rows: + break + out = np.array([[complex(float(x.re), float(x.im)) for x in row] for row in m]) + return out if np.iscomplexobj(a) else out.real.copy() + + +def exact_det(a): + m = to_q(a) + n = len(m) + d = Q(1) + for c in range(n): + p = next((i for i in range(c, n) if not m[i][c].iszero()), None) + if p is None: + return complex(0.0) if np.iscomplexobj(a) else 0.0 + if p != c: + m[c], m[p] = m[p], m[c] + d = Q(0) - d + d = d * m[c][c] + for i in range(c + 1, n): + f = m[i][c] / m[c][c] + m[i] = [x - f * y for x, y in zip(m[i], m[c])] + return complex(float(d.re), float(d.im)) if np.iscomplexobj(a) else float(d.re) + + +# ---- writing ---- +LINES = [] + + +def save(case, role, arr): + arr = np.ascontiguousarray(arr) + arr = arr.astype(" cut] + # every dropped value must sit far below the cutoff so feng and numpy keep the same rank + assert all(x < cut / 10 for x in s[s <= cut]), (tag, s, cut) + assert all(x > cut * 1e6 for x in kept), (tag, s, cut) + x = np.linalg.pinv(a, rtol=p * EPS) + k = float(kept[0] / kept[-1]) + add("pinv_%s_%dx%d%s" % (kind(a), m, n, tag), "pinv", 64 * p * EPS * k * k, [("A", a), ("X", x)]) + + +def cholesky(seed, n, complex_): + b = cplx(seed, n, n) if complex_ else real(seed, n, n) + a = b @ b.conj().T + n * np.eye(n) + a = (a + a.conj().T) / 2 # exactly Hermitian + add("cholesky_%s_%d" % (kind(a), n), "cholesky", 64 * n * EPS * cond(a), [("A", a), ("L", np.linalg.cholesky(a))]) + + +def rref(a, tag=""): + m, n = a.shape + r = exact_rref(a) + pivots = [next(j for j in range(n) if r[i, j] != 0) for i in range(m) if np.any(r[i] != 0)] + k = cond(a[:, pivots]) + add("rref_%s_%dx%d%s" % (kind(a), m, n, tag), "rref", 64 * max(m, n) * EPS * k, [("A", a), ("R", r)]) + + +def expm(a, tag=""): + n = a.shape[0] + nrm = max(1.0, float(np.linalg.norm(a, 1))) + add("expm_%s_%d%s" % (kind(a), n, tag), "expm", 256 * n * EPS * nrm, [("A", a), ("E", scipy.linalg.expm(a))]) + + +def main(): + for f in os.listdir(HERE): + if f.endswith(".npy"): + os.remove(os.path.join(HERE, f)) + + lu_solve(11, well(real(11, 4, 4)), 2) + lu_solve(12, real(12, 8, 8), 3) + lu_solve(13, well(cplx(13, 5, 5)), 2) + + det(real(21, 4, 4)) + det(cplx(23, 6, 6)) + det(np.array([[1.0, 2, 3], [4, 5, 6], [7, 8, 9]]), exact=True) + + inv(well(real(31, 5, 5))) + inv(real(32, 8, 8)) + inv(well(cplx(33, 4, 4))) + + svdvals(real(42, 8, 5)) + svdvals(real(43, 4, 7)) + svdvals(low_rank(44, 6, 6, 3), "_rank3") + svdvals(cplx(45, 7, 4)) + svdvals(cplx(46, 3, 6)) + + pinv(real(51, 6, 4)) + pinv(real(52, 3, 5)) + pinv(low_rank(53, 5, 5, 3), "_rank3") + pinv(cplx(54, 5, 3)) + pinv(low_rank(55, 4, 6, 2, complex_=True), "_rank2") + + cholesky(61, 5, False) + cholesky(62, 8, False) + cholesky(63, 6, True) + + rref(np.array([[2.0, 1, 1], [1, 3, 2], [1, 0, 0]])) + rref(low_rank(71, 4, 4, 2), "_rank2") + rref(low_rank(72, 5, 3, 2), "_rank2") + rref(low_rank(73, 3, 5, 3)) + rref(low_rank(74, 3, 4, 2, complex_=True), "_rank2") + + expm(real(81, 4, 4)) + expm(5 * real(82, 6, 6), "_norm5") + expm(cplx(83, 4, 4)) + + header = [ + "# S7-R6 oracle fixtures, written by tests/fixtures/s7/make_fixtures.py; do not edit by hand.", + "# python %s, numpy %s, scipy %s" % (sys.version.split()[0], np.__version__, scipy.__version__), + "# ", + ] + with open(os.path.join(HERE, "manifest.txt"), "w", newline="\n") as f: + f.write("\n".join(header + LINES) + "\n") + + +if __name__ == "__main__": + main() diff --git a/tests/fixtures/s7/manifest.txt b/tests/fixtures/s7/manifest.txt new file mode 100644 index 0000000..8ab7183 --- /dev/null +++ b/tests/fixtures/s7/manifest.txt @@ -0,0 +1,33 @@ +# S7-R6 oracle fixtures, written by tests/fixtures/s7/make_fixtures.py; do not edit by hand. +# python 3.14.7, numpy 2.5.3, scipy 1.18.1 +# +lu_solve_f8_4x2 lu_solve 1.1e-13 lu_solve_f8_4x2_A.npy lu_solve_f8_4x2_B.npy lu_solve_f8_4x2_X.npy +lu_solve_f8_8x3 lu_solve 1.6e-12 lu_solve_f8_8x3_A.npy lu_solve_f8_8x3_B.npy lu_solve_f8_8x3_X.npy +lu_solve_c16_5x2 lu_solve 1.6e-13 lu_solve_c16_5x2_A.npy lu_solve_c16_5x2_B.npy lu_solve_c16_5x2_X.npy +det_f8_4 det 5e-13 det_f8_4_A.npy det_f8_4_D.npy +det_c16_6 det 1.2e-12 det_c16_6_A.npy det_c16_6_D.npy +det_f8_3_singular det 4.5e-14 det_f8_3_singular_A.npy det_f8_3_singular_D.npy +inv_f8_5 inv 1.4e-13 inv_f8_5_A.npy inv_f8_5_X.npy +inv_f8_8 inv 7.6e-13 inv_f8_8_A.npy inv_f8_8_X.npy +inv_c16_4 inv 1.3e-13 inv_c16_4_A.npy inv_c16_4_X.npy +svdvals_f8_8x5 svdvals 6e-14 svdvals_f8_8x5_A.npy svdvals_f8_8x5_S.npy +svdvals_f8_4x7 svdvals 5.2e-14 svdvals_f8_4x7_A.npy svdvals_f8_4x7_S.npy +svdvals_f8_6x6_rank3 svdvals 4.5e-14 svdvals_f8_6x6_rank3_A.npy svdvals_f8_6x6_rank3_S.npy +svdvals_c16_7x4 svdvals 5.2e-14 svdvals_c16_7x4_A.npy svdvals_c16_7x4_S.npy +svdvals_c16_3x6 svdvals 4.5e-14 svdvals_c16_3x6_A.npy svdvals_c16_3x6_S.npy +pinv_f8_6x4 pinv 4.6e-12 pinv_f8_6x4_A.npy pinv_f8_6x4_X.npy +pinv_f8_3x5 pinv 2e-11 pinv_f8_3x5_A.npy pinv_f8_3x5_X.npy +pinv_f8_5x5_rank3 pinv 9.9e-12 pinv_f8_5x5_rank3_A.npy pinv_f8_5x5_rank3_X.npy +pinv_c16_5x3 pinv 5.2e-13 pinv_c16_5x3_A.npy pinv_c16_5x3_X.npy +pinv_c16_4x6_rank2 pinv 5.9e-13 pinv_c16_4x6_rank2_A.npy pinv_c16_4x6_rank2_X.npy +cholesky_f8_5 cholesky 1.5e-13 cholesky_f8_5_A.npy cholesky_f8_5_L.npy +cholesky_f8_8 cholesky 2.5e-13 cholesky_f8_8_A.npy cholesky_f8_8_L.npy +cholesky_c16_6 cholesky 2.2e-13 cholesky_c16_6_A.npy cholesky_c16_6_L.npy +rref_f8_3x3 rref 1.3e-12 rref_f8_3x3_A.npy rref_f8_3x3_R.npy +rref_f8_4x4_rank2 rref 1e-13 rref_f8_4x4_rank2_A.npy rref_f8_4x4_rank2_R.npy +rref_f8_5x3_rank2 rref 1.7e-13 rref_f8_5x3_rank2_A.npy rref_f8_5x3_rank2_R.npy +rref_f8_3x5 rref 4.2e-13 rref_f8_3x5_A.npy rref_f8_3x5_R.npy +rref_c16_3x4_rank2 rref 3e-13 rref_c16_3x4_rank2_A.npy rref_c16_3x4_rank2_R.npy +expm_f8_4 expm 5.9e-13 expm_f8_4_A.npy expm_f8_4_E.npy +expm_f8_6_norm5 expm 7.3e-12 expm_f8_6_norm5_A.npy expm_f8_6_norm5_E.npy +expm_c16_4 expm 7e-13 expm_c16_4_A.npy expm_c16_4_E.npy diff --git a/tests/fixtures/s7/pinv_c16_4x6_rank2_A.npy b/tests/fixtures/s7/pinv_c16_4x6_rank2_A.npy new file mode 100644 index 0000000..2cc95e0 Binary files /dev/null and b/tests/fixtures/s7/pinv_c16_4x6_rank2_A.npy differ diff --git a/tests/fixtures/s7/pinv_c16_4x6_rank2_X.npy b/tests/fixtures/s7/pinv_c16_4x6_rank2_X.npy new file mode 100644 index 0000000..f6f044d Binary files /dev/null and b/tests/fixtures/s7/pinv_c16_4x6_rank2_X.npy differ diff --git a/tests/fixtures/s7/pinv_c16_5x3_A.npy b/tests/fixtures/s7/pinv_c16_5x3_A.npy new file mode 100644 index 0000000..c13d828 Binary files /dev/null and b/tests/fixtures/s7/pinv_c16_5x3_A.npy differ diff --git a/tests/fixtures/s7/pinv_c16_5x3_X.npy b/tests/fixtures/s7/pinv_c16_5x3_X.npy new file mode 100644 index 0000000..8fa86a0 Binary files /dev/null and b/tests/fixtures/s7/pinv_c16_5x3_X.npy differ diff --git a/tests/fixtures/s7/pinv_f8_3x5_A.npy b/tests/fixtures/s7/pinv_f8_3x5_A.npy new file mode 100644 index 0000000..f4179fe Binary files /dev/null and b/tests/fixtures/s7/pinv_f8_3x5_A.npy differ diff --git a/tests/fixtures/s7/pinv_f8_3x5_X.npy b/tests/fixtures/s7/pinv_f8_3x5_X.npy new file mode 100644 index 0000000..f514656 Binary files /dev/null and b/tests/fixtures/s7/pinv_f8_3x5_X.npy differ diff --git a/tests/fixtures/s7/pinv_f8_5x5_rank3_A.npy b/tests/fixtures/s7/pinv_f8_5x5_rank3_A.npy new file mode 100644 index 0000000..4942104 Binary files /dev/null and b/tests/fixtures/s7/pinv_f8_5x5_rank3_A.npy differ diff --git a/tests/fixtures/s7/pinv_f8_5x5_rank3_X.npy b/tests/fixtures/s7/pinv_f8_5x5_rank3_X.npy new file mode 100644 index 0000000..6cc42ae Binary files /dev/null and b/tests/fixtures/s7/pinv_f8_5x5_rank3_X.npy differ diff --git a/tests/fixtures/s7/pinv_f8_6x4_A.npy b/tests/fixtures/s7/pinv_f8_6x4_A.npy new file mode 100644 index 0000000..7ef48f7 Binary files /dev/null and b/tests/fixtures/s7/pinv_f8_6x4_A.npy differ diff --git a/tests/fixtures/s7/pinv_f8_6x4_X.npy b/tests/fixtures/s7/pinv_f8_6x4_X.npy new file mode 100644 index 0000000..84aa9af Binary files /dev/null and b/tests/fixtures/s7/pinv_f8_6x4_X.npy differ diff --git a/tests/fixtures/s7/rref_c16_3x4_rank2_A.npy b/tests/fixtures/s7/rref_c16_3x4_rank2_A.npy new file mode 100644 index 0000000..af31860 Binary files /dev/null and b/tests/fixtures/s7/rref_c16_3x4_rank2_A.npy differ diff --git a/tests/fixtures/s7/rref_c16_3x4_rank2_R.npy b/tests/fixtures/s7/rref_c16_3x4_rank2_R.npy new file mode 100644 index 0000000..865cfd7 Binary files /dev/null and b/tests/fixtures/s7/rref_c16_3x4_rank2_R.npy differ diff --git a/tests/fixtures/s7/rref_f8_3x3_A.npy b/tests/fixtures/s7/rref_f8_3x3_A.npy new file mode 100644 index 0000000..85a4d78 Binary files /dev/null and b/tests/fixtures/s7/rref_f8_3x3_A.npy differ diff --git a/tests/fixtures/s7/rref_f8_3x3_R.npy b/tests/fixtures/s7/rref_f8_3x3_R.npy new file mode 100644 index 0000000..f5a9945 Binary files /dev/null and b/tests/fixtures/s7/rref_f8_3x3_R.npy differ diff --git a/tests/fixtures/s7/rref_f8_3x5_A.npy b/tests/fixtures/s7/rref_f8_3x5_A.npy new file mode 100644 index 0000000..76a0dd1 Binary files /dev/null and b/tests/fixtures/s7/rref_f8_3x5_A.npy differ diff --git a/tests/fixtures/s7/rref_f8_3x5_R.npy b/tests/fixtures/s7/rref_f8_3x5_R.npy new file mode 100644 index 0000000..423003e Binary files /dev/null and b/tests/fixtures/s7/rref_f8_3x5_R.npy differ diff --git a/tests/fixtures/s7/rref_f8_4x4_rank2_A.npy b/tests/fixtures/s7/rref_f8_4x4_rank2_A.npy new file mode 100644 index 0000000..6c96e41 Binary files /dev/null and b/tests/fixtures/s7/rref_f8_4x4_rank2_A.npy differ diff --git a/tests/fixtures/s7/rref_f8_4x4_rank2_R.npy b/tests/fixtures/s7/rref_f8_4x4_rank2_R.npy new file mode 100644 index 0000000..89f12ad Binary files /dev/null and b/tests/fixtures/s7/rref_f8_4x4_rank2_R.npy differ diff --git a/tests/fixtures/s7/rref_f8_5x3_rank2_A.npy b/tests/fixtures/s7/rref_f8_5x3_rank2_A.npy new file mode 100644 index 0000000..3aea65f Binary files /dev/null and b/tests/fixtures/s7/rref_f8_5x3_rank2_A.npy differ diff --git a/tests/fixtures/s7/rref_f8_5x3_rank2_R.npy b/tests/fixtures/s7/rref_f8_5x3_rank2_R.npy new file mode 100644 index 0000000..57e1dbc Binary files /dev/null and b/tests/fixtures/s7/rref_f8_5x3_rank2_R.npy differ diff --git a/tests/fixtures/s7/svdvals_c16_3x6_A.npy b/tests/fixtures/s7/svdvals_c16_3x6_A.npy new file mode 100644 index 0000000..08ad68f Binary files /dev/null and b/tests/fixtures/s7/svdvals_c16_3x6_A.npy differ diff --git a/tests/fixtures/s7/svdvals_c16_3x6_S.npy b/tests/fixtures/s7/svdvals_c16_3x6_S.npy new file mode 100644 index 0000000..3341744 Binary files /dev/null and b/tests/fixtures/s7/svdvals_c16_3x6_S.npy differ diff --git a/tests/fixtures/s7/svdvals_c16_7x4_A.npy b/tests/fixtures/s7/svdvals_c16_7x4_A.npy new file mode 100644 index 0000000..5bae72e Binary files /dev/null and b/tests/fixtures/s7/svdvals_c16_7x4_A.npy differ diff --git a/tests/fixtures/s7/svdvals_c16_7x4_S.npy b/tests/fixtures/s7/svdvals_c16_7x4_S.npy new file mode 100644 index 0000000..78a9b80 Binary files /dev/null and b/tests/fixtures/s7/svdvals_c16_7x4_S.npy differ diff --git a/tests/fixtures/s7/svdvals_f8_4x7_A.npy b/tests/fixtures/s7/svdvals_f8_4x7_A.npy new file mode 100644 index 0000000..7bec544 Binary files /dev/null and b/tests/fixtures/s7/svdvals_f8_4x7_A.npy differ diff --git a/tests/fixtures/s7/svdvals_f8_4x7_S.npy b/tests/fixtures/s7/svdvals_f8_4x7_S.npy new file mode 100644 index 0000000..dfd1c24 Binary files /dev/null and b/tests/fixtures/s7/svdvals_f8_4x7_S.npy differ diff --git a/tests/fixtures/s7/svdvals_f8_6x6_rank3_A.npy b/tests/fixtures/s7/svdvals_f8_6x6_rank3_A.npy new file mode 100644 index 0000000..eeb41de Binary files /dev/null and b/tests/fixtures/s7/svdvals_f8_6x6_rank3_A.npy differ diff --git a/tests/fixtures/s7/svdvals_f8_6x6_rank3_S.npy b/tests/fixtures/s7/svdvals_f8_6x6_rank3_S.npy new file mode 100644 index 0000000..f6e586e Binary files /dev/null and b/tests/fixtures/s7/svdvals_f8_6x6_rank3_S.npy differ diff --git a/tests/fixtures/s7/svdvals_f8_8x5_A.npy b/tests/fixtures/s7/svdvals_f8_8x5_A.npy new file mode 100644 index 0000000..84774c3 Binary files /dev/null and b/tests/fixtures/s7/svdvals_f8_8x5_A.npy differ diff --git a/tests/fixtures/s7/svdvals_f8_8x5_S.npy b/tests/fixtures/s7/svdvals_f8_8x5_S.npy new file mode 100644 index 0000000..d69b19b Binary files /dev/null and b/tests/fixtures/s7/svdvals_f8_8x5_S.npy differ diff --git a/tests/fixtures/s8/conv_f8_1x6_1x3_A.npy b/tests/fixtures/s8/conv_f8_1x6_1x3_A.npy new file mode 100644 index 0000000..6913767 Binary files /dev/null and b/tests/fixtures/s8/conv_f8_1x6_1x3_A.npy differ diff --git a/tests/fixtures/s8/conv_f8_1x6_1x3_B.npy b/tests/fixtures/s8/conv_f8_1x6_1x3_B.npy new file mode 100644 index 0000000..6056f8c Binary files /dev/null and b/tests/fixtures/s8/conv_f8_1x6_1x3_B.npy differ diff --git a/tests/fixtures/s8/conv_f8_5x1_2x1_A.npy b/tests/fixtures/s8/conv_f8_5x1_2x1_A.npy new file mode 100644 index 0000000..189a1f8 Binary files /dev/null and b/tests/fixtures/s8/conv_f8_5x1_2x1_A.npy differ diff --git a/tests/fixtures/s8/conv_f8_5x1_2x1_B.npy b/tests/fixtures/s8/conv_f8_5x1_2x1_B.npy new file mode 100644 index 0000000..ae38dc9 Binary files /dev/null and b/tests/fixtures/s8/conv_f8_5x1_2x1_B.npy differ diff --git a/tests/fixtures/s8/conv_f8_7x9_10x11_A.npy b/tests/fixtures/s8/conv_f8_7x9_10x11_A.npy new file mode 100644 index 0000000..590e4ec Binary files /dev/null and b/tests/fixtures/s8/conv_f8_7x9_10x11_A.npy differ diff --git a/tests/fixtures/s8/conv_f8_7x9_10x11_B.npy b/tests/fixtures/s8/conv_f8_7x9_10x11_B.npy new file mode 100644 index 0000000..b5fd521 Binary files /dev/null and b/tests/fixtures/s8/conv_f8_7x9_10x11_B.npy differ diff --git a/tests/fixtures/s8/conv_f8_7x9_1x1_A.npy b/tests/fixtures/s8/conv_f8_7x9_1x1_A.npy new file mode 100644 index 0000000..590e4ec Binary files /dev/null and b/tests/fixtures/s8/conv_f8_7x9_1x1_A.npy differ diff --git a/tests/fixtures/s8/conv_f8_7x9_1x1_B.npy b/tests/fixtures/s8/conv_f8_7x9_1x1_B.npy new file mode 100644 index 0000000..29ec695 Binary files /dev/null and b/tests/fixtures/s8/conv_f8_7x9_1x1_B.npy differ diff --git a/tests/fixtures/s8/conv_f8_7x9_2x4_A.npy b/tests/fixtures/s8/conv_f8_7x9_2x4_A.npy new file mode 100644 index 0000000..590e4ec Binary files /dev/null and b/tests/fixtures/s8/conv_f8_7x9_2x4_A.npy differ diff --git a/tests/fixtures/s8/conv_f8_7x9_2x4_B.npy b/tests/fixtures/s8/conv_f8_7x9_2x4_B.npy new file mode 100644 index 0000000..bfce3b9 Binary files /dev/null and b/tests/fixtures/s8/conv_f8_7x9_2x4_B.npy differ diff --git a/tests/fixtures/s8/conv_f8_7x9_3x12_A.npy b/tests/fixtures/s8/conv_f8_7x9_3x12_A.npy new file mode 100644 index 0000000..590e4ec Binary files /dev/null and b/tests/fixtures/s8/conv_f8_7x9_3x12_A.npy differ diff --git a/tests/fixtures/s8/conv_f8_7x9_3x12_B.npy b/tests/fixtures/s8/conv_f8_7x9_3x12_B.npy new file mode 100644 index 0000000..a782511 Binary files /dev/null and b/tests/fixtures/s8/conv_f8_7x9_3x12_B.npy differ diff --git a/tests/fixtures/s8/conv_f8_7x9_3x5_A.npy b/tests/fixtures/s8/conv_f8_7x9_3x5_A.npy new file mode 100644 index 0000000..590e4ec Binary files /dev/null and b/tests/fixtures/s8/conv_f8_7x9_3x5_A.npy differ diff --git a/tests/fixtures/s8/conv_f8_7x9_3x5_B.npy b/tests/fixtures/s8/conv_f8_7x9_3x5_B.npy new file mode 100644 index 0000000..33a6b46 Binary files /dev/null and b/tests/fixtures/s8/conv_f8_7x9_3x5_B.npy differ diff --git a/tests/fixtures/s8/conv_f8_7x9_4x3_A.npy b/tests/fixtures/s8/conv_f8_7x9_4x3_A.npy new file mode 100644 index 0000000..590e4ec Binary files /dev/null and b/tests/fixtures/s8/conv_f8_7x9_4x3_A.npy differ diff --git a/tests/fixtures/s8/conv_f8_7x9_4x3_B.npy b/tests/fixtures/s8/conv_f8_7x9_4x3_B.npy new file mode 100644 index 0000000..76a5e4f Binary files /dev/null and b/tests/fixtures/s8/conv_f8_7x9_4x3_B.npy differ diff --git a/tests/fixtures/s8/conv_full_f8_1x6_1x3_C.npy b/tests/fixtures/s8/conv_full_f8_1x6_1x3_C.npy new file mode 100644 index 0000000..3ad5934 Binary files /dev/null and b/tests/fixtures/s8/conv_full_f8_1x6_1x3_C.npy differ diff --git a/tests/fixtures/s8/conv_full_f8_5x1_2x1_C.npy b/tests/fixtures/s8/conv_full_f8_5x1_2x1_C.npy new file mode 100644 index 0000000..19e316d Binary files /dev/null and b/tests/fixtures/s8/conv_full_f8_5x1_2x1_C.npy differ diff --git a/tests/fixtures/s8/conv_full_f8_7x9_10x11_C.npy b/tests/fixtures/s8/conv_full_f8_7x9_10x11_C.npy new file mode 100644 index 0000000..20c0163 Binary files /dev/null and b/tests/fixtures/s8/conv_full_f8_7x9_10x11_C.npy differ diff --git a/tests/fixtures/s8/conv_full_f8_7x9_1x1_C.npy b/tests/fixtures/s8/conv_full_f8_7x9_1x1_C.npy new file mode 100644 index 0000000..061337b Binary files /dev/null and b/tests/fixtures/s8/conv_full_f8_7x9_1x1_C.npy differ diff --git a/tests/fixtures/s8/conv_full_f8_7x9_2x4_C.npy b/tests/fixtures/s8/conv_full_f8_7x9_2x4_C.npy new file mode 100644 index 0000000..ec7772d Binary files /dev/null and b/tests/fixtures/s8/conv_full_f8_7x9_2x4_C.npy differ diff --git a/tests/fixtures/s8/conv_full_f8_7x9_3x12_C.npy b/tests/fixtures/s8/conv_full_f8_7x9_3x12_C.npy new file mode 100644 index 0000000..ba796be Binary files /dev/null and b/tests/fixtures/s8/conv_full_f8_7x9_3x12_C.npy differ diff --git a/tests/fixtures/s8/conv_full_f8_7x9_3x5_C.npy b/tests/fixtures/s8/conv_full_f8_7x9_3x5_C.npy new file mode 100644 index 0000000..61b2155 Binary files /dev/null and b/tests/fixtures/s8/conv_full_f8_7x9_3x5_C.npy differ diff --git a/tests/fixtures/s8/conv_full_f8_7x9_4x3_C.npy b/tests/fixtures/s8/conv_full_f8_7x9_4x3_C.npy new file mode 100644 index 0000000..8bcafc3 Binary files /dev/null and b/tests/fixtures/s8/conv_full_f8_7x9_4x3_C.npy differ diff --git a/tests/fixtures/s8/conv_full_i4_6x8_1x1_C.npy b/tests/fixtures/s8/conv_full_i4_6x8_1x1_C.npy new file mode 100644 index 0000000..07079d6 Binary files /dev/null and b/tests/fixtures/s8/conv_full_i4_6x8_1x1_C.npy differ diff --git a/tests/fixtures/s8/conv_full_i4_6x8_2x4_C.npy b/tests/fixtures/s8/conv_full_i4_6x8_2x4_C.npy new file mode 100644 index 0000000..4099352 Binary files /dev/null and b/tests/fixtures/s8/conv_full_i4_6x8_2x4_C.npy differ diff --git a/tests/fixtures/s8/conv_full_i4_6x8_3x5_C.npy b/tests/fixtures/s8/conv_full_i4_6x8_3x5_C.npy new file mode 100644 index 0000000..db53ff4 Binary files /dev/null and b/tests/fixtures/s8/conv_full_i4_6x8_3x5_C.npy differ diff --git a/tests/fixtures/s8/conv_full_i4_6x8_9x10_C.npy b/tests/fixtures/s8/conv_full_i4_6x8_9x10_C.npy new file mode 100644 index 0000000..e801cf5 Binary files /dev/null and b/tests/fixtures/s8/conv_full_i4_6x8_9x10_C.npy differ diff --git a/tests/fixtures/s8/conv_i4_6x8_1x1_A.npy b/tests/fixtures/s8/conv_i4_6x8_1x1_A.npy new file mode 100644 index 0000000..c7a06ad Binary files /dev/null and b/tests/fixtures/s8/conv_i4_6x8_1x1_A.npy differ diff --git a/tests/fixtures/s8/conv_i4_6x8_1x1_B.npy b/tests/fixtures/s8/conv_i4_6x8_1x1_B.npy new file mode 100644 index 0000000..c05902e Binary files /dev/null and b/tests/fixtures/s8/conv_i4_6x8_1x1_B.npy differ diff --git a/tests/fixtures/s8/conv_i4_6x8_2x4_A.npy b/tests/fixtures/s8/conv_i4_6x8_2x4_A.npy new file mode 100644 index 0000000..c7a06ad Binary files /dev/null and b/tests/fixtures/s8/conv_i4_6x8_2x4_A.npy differ diff --git a/tests/fixtures/s8/conv_i4_6x8_2x4_B.npy b/tests/fixtures/s8/conv_i4_6x8_2x4_B.npy new file mode 100644 index 0000000..e0ad83e Binary files /dev/null and b/tests/fixtures/s8/conv_i4_6x8_2x4_B.npy differ diff --git a/tests/fixtures/s8/conv_i4_6x8_3x5_A.npy b/tests/fixtures/s8/conv_i4_6x8_3x5_A.npy new file mode 100644 index 0000000..c7a06ad Binary files /dev/null and b/tests/fixtures/s8/conv_i4_6x8_3x5_A.npy differ diff --git a/tests/fixtures/s8/conv_i4_6x8_3x5_B.npy b/tests/fixtures/s8/conv_i4_6x8_3x5_B.npy new file mode 100644 index 0000000..8f83218 Binary files /dev/null and b/tests/fixtures/s8/conv_i4_6x8_3x5_B.npy differ diff --git a/tests/fixtures/s8/conv_i4_6x8_9x10_A.npy b/tests/fixtures/s8/conv_i4_6x8_9x10_A.npy new file mode 100644 index 0000000..c7a06ad Binary files /dev/null and b/tests/fixtures/s8/conv_i4_6x8_9x10_A.npy differ diff --git a/tests/fixtures/s8/conv_i4_6x8_9x10_B.npy b/tests/fixtures/s8/conv_i4_6x8_9x10_B.npy new file mode 100644 index 0000000..5bb683c Binary files /dev/null and b/tests/fixtures/s8/conv_i4_6x8_9x10_B.npy differ diff --git a/tests/fixtures/s8/conv_same_f8_1x6_1x3_C.npy b/tests/fixtures/s8/conv_same_f8_1x6_1x3_C.npy new file mode 100644 index 0000000..c6f1a27 Binary files /dev/null and b/tests/fixtures/s8/conv_same_f8_1x6_1x3_C.npy differ diff --git a/tests/fixtures/s8/conv_same_f8_5x1_2x1_C.npy b/tests/fixtures/s8/conv_same_f8_5x1_2x1_C.npy new file mode 100644 index 0000000..39d0f4f Binary files /dev/null and b/tests/fixtures/s8/conv_same_f8_5x1_2x1_C.npy differ diff --git a/tests/fixtures/s8/conv_same_f8_7x9_10x11_C.npy b/tests/fixtures/s8/conv_same_f8_7x9_10x11_C.npy new file mode 100644 index 0000000..5f9f800 Binary files /dev/null and b/tests/fixtures/s8/conv_same_f8_7x9_10x11_C.npy differ diff --git a/tests/fixtures/s8/conv_same_f8_7x9_1x1_C.npy b/tests/fixtures/s8/conv_same_f8_7x9_1x1_C.npy new file mode 100644 index 0000000..061337b Binary files /dev/null and b/tests/fixtures/s8/conv_same_f8_7x9_1x1_C.npy differ diff --git a/tests/fixtures/s8/conv_same_f8_7x9_2x4_C.npy b/tests/fixtures/s8/conv_same_f8_7x9_2x4_C.npy new file mode 100644 index 0000000..17b71c0 Binary files /dev/null and b/tests/fixtures/s8/conv_same_f8_7x9_2x4_C.npy differ diff --git a/tests/fixtures/s8/conv_same_f8_7x9_3x12_C.npy b/tests/fixtures/s8/conv_same_f8_7x9_3x12_C.npy new file mode 100644 index 0000000..e1b7f96 Binary files /dev/null and b/tests/fixtures/s8/conv_same_f8_7x9_3x12_C.npy differ diff --git a/tests/fixtures/s8/conv_same_f8_7x9_3x5_C.npy b/tests/fixtures/s8/conv_same_f8_7x9_3x5_C.npy new file mode 100644 index 0000000..df8cbd2 Binary files /dev/null and b/tests/fixtures/s8/conv_same_f8_7x9_3x5_C.npy differ diff --git a/tests/fixtures/s8/conv_same_f8_7x9_4x3_C.npy b/tests/fixtures/s8/conv_same_f8_7x9_4x3_C.npy new file mode 100644 index 0000000..30e8633 Binary files /dev/null and b/tests/fixtures/s8/conv_same_f8_7x9_4x3_C.npy differ diff --git a/tests/fixtures/s8/conv_same_i4_6x8_1x1_C.npy b/tests/fixtures/s8/conv_same_i4_6x8_1x1_C.npy new file mode 100644 index 0000000..07079d6 Binary files /dev/null and b/tests/fixtures/s8/conv_same_i4_6x8_1x1_C.npy differ diff --git a/tests/fixtures/s8/conv_same_i4_6x8_2x4_C.npy b/tests/fixtures/s8/conv_same_i4_6x8_2x4_C.npy new file mode 100644 index 0000000..6794672 Binary files /dev/null and b/tests/fixtures/s8/conv_same_i4_6x8_2x4_C.npy differ diff --git a/tests/fixtures/s8/conv_same_i4_6x8_3x5_C.npy b/tests/fixtures/s8/conv_same_i4_6x8_3x5_C.npy new file mode 100644 index 0000000..774e5c1 Binary files /dev/null and b/tests/fixtures/s8/conv_same_i4_6x8_3x5_C.npy differ diff --git a/tests/fixtures/s8/conv_same_i4_6x8_9x10_C.npy b/tests/fixtures/s8/conv_same_i4_6x8_9x10_C.npy new file mode 100644 index 0000000..80c3210 Binary files /dev/null and b/tests/fixtures/s8/conv_same_i4_6x8_9x10_C.npy differ diff --git a/tests/fixtures/s8/conv_valid_f8_1x6_1x3_C.npy b/tests/fixtures/s8/conv_valid_f8_1x6_1x3_C.npy new file mode 100644 index 0000000..a3c5443 Binary files /dev/null and b/tests/fixtures/s8/conv_valid_f8_1x6_1x3_C.npy differ diff --git a/tests/fixtures/s8/conv_valid_f8_5x1_2x1_C.npy b/tests/fixtures/s8/conv_valid_f8_5x1_2x1_C.npy new file mode 100644 index 0000000..660b033 Binary files /dev/null and b/tests/fixtures/s8/conv_valid_f8_5x1_2x1_C.npy differ diff --git a/tests/fixtures/s8/conv_valid_f8_7x9_10x11_C.npy b/tests/fixtures/s8/conv_valid_f8_7x9_10x11_C.npy new file mode 100644 index 0000000..d0c75b0 Binary files /dev/null and b/tests/fixtures/s8/conv_valid_f8_7x9_10x11_C.npy differ diff --git a/tests/fixtures/s8/conv_valid_f8_7x9_1x1_C.npy b/tests/fixtures/s8/conv_valid_f8_7x9_1x1_C.npy new file mode 100644 index 0000000..061337b Binary files /dev/null and b/tests/fixtures/s8/conv_valid_f8_7x9_1x1_C.npy differ diff --git a/tests/fixtures/s8/conv_valid_f8_7x9_2x4_C.npy b/tests/fixtures/s8/conv_valid_f8_7x9_2x4_C.npy new file mode 100644 index 0000000..7b5f05c Binary files /dev/null and b/tests/fixtures/s8/conv_valid_f8_7x9_2x4_C.npy differ diff --git a/tests/fixtures/s8/conv_valid_f8_7x9_3x5_C.npy b/tests/fixtures/s8/conv_valid_f8_7x9_3x5_C.npy new file mode 100644 index 0000000..5aabd16 Binary files /dev/null and b/tests/fixtures/s8/conv_valid_f8_7x9_3x5_C.npy differ diff --git a/tests/fixtures/s8/conv_valid_f8_7x9_4x3_C.npy b/tests/fixtures/s8/conv_valid_f8_7x9_4x3_C.npy new file mode 100644 index 0000000..53203bf Binary files /dev/null and b/tests/fixtures/s8/conv_valid_f8_7x9_4x3_C.npy differ diff --git a/tests/fixtures/s8/conv_valid_i4_6x8_1x1_C.npy b/tests/fixtures/s8/conv_valid_i4_6x8_1x1_C.npy new file mode 100644 index 0000000..07079d6 Binary files /dev/null and b/tests/fixtures/s8/conv_valid_i4_6x8_1x1_C.npy differ diff --git a/tests/fixtures/s8/conv_valid_i4_6x8_2x4_C.npy b/tests/fixtures/s8/conv_valid_i4_6x8_2x4_C.npy new file mode 100644 index 0000000..a5d805f Binary files /dev/null and b/tests/fixtures/s8/conv_valid_i4_6x8_2x4_C.npy differ diff --git a/tests/fixtures/s8/conv_valid_i4_6x8_3x5_C.npy b/tests/fixtures/s8/conv_valid_i4_6x8_3x5_C.npy new file mode 100644 index 0000000..9fbc9d8 Binary files /dev/null and b/tests/fixtures/s8/conv_valid_i4_6x8_3x5_C.npy differ diff --git a/tests/fixtures/s8/conv_valid_i4_6x8_9x10_C.npy b/tests/fixtures/s8/conv_valid_i4_6x8_9x10_C.npy new file mode 100644 index 0000000..4933e9c Binary files /dev/null and b/tests/fixtures/s8/conv_valid_i4_6x8_9x10_C.npy differ diff --git a/tests/fixtures/s8/fft2_c16_1x1_C.npy b/tests/fixtures/s8/fft2_c16_1x1_C.npy new file mode 100644 index 0000000..4f8c507 Binary files /dev/null and b/tests/fixtures/s8/fft2_c16_1x1_C.npy differ diff --git a/tests/fixtures/s8/fft2_c16_1x7_C.npy b/tests/fixtures/s8/fft2_c16_1x7_C.npy new file mode 100644 index 0000000..ace3461 Binary files /dev/null and b/tests/fixtures/s8/fft2_c16_1x7_C.npy differ diff --git a/tests/fixtures/s8/fft2_c16_31x17_C.npy b/tests/fixtures/s8/fft2_c16_31x17_C.npy new file mode 100644 index 0000000..a43ccb3 Binary files /dev/null and b/tests/fixtures/s8/fft2_c16_31x17_C.npy differ diff --git a/tests/fixtures/s8/fft2_c16_3x5_C.npy b/tests/fixtures/s8/fft2_c16_3x5_C.npy new file mode 100644 index 0000000..4f2a500 Binary files /dev/null and b/tests/fixtures/s8/fft2_c16_3x5_C.npy differ diff --git a/tests/fixtures/s8/fft2_c16_4x6_C.npy b/tests/fixtures/s8/fft2_c16_4x6_C.npy new file mode 100644 index 0000000..5bee634 Binary files /dev/null and b/tests/fixtures/s8/fft2_c16_4x6_C.npy differ diff --git a/tests/fixtures/s8/fft2_c16_7x11_C.npy b/tests/fixtures/s8/fft2_c16_7x11_C.npy new file mode 100644 index 0000000..3c99c9e Binary files /dev/null and b/tests/fixtures/s8/fft2_c16_7x11_C.npy differ diff --git a/tests/fixtures/s8/fft2_c16_8x8_C.npy b/tests/fixtures/s8/fft2_c16_8x8_C.npy new file mode 100644 index 0000000..160d46b Binary files /dev/null and b/tests/fixtures/s8/fft2_c16_8x8_C.npy differ diff --git a/tests/fixtures/s8/fft2_c8_3x5_C.npy b/tests/fixtures/s8/fft2_c8_3x5_C.npy new file mode 100644 index 0000000..b05d37e Binary files /dev/null and b/tests/fixtures/s8/fft2_c8_3x5_C.npy differ diff --git a/tests/fixtures/s8/fft2_c8_8x8_C.npy b/tests/fixtures/s8/fft2_c8_8x8_C.npy new file mode 100644 index 0000000..4f8f86f Binary files /dev/null and b/tests/fixtures/s8/fft2_c8_8x8_C.npy differ diff --git a/tests/fixtures/s8/fft2_f4_13x1_C.npy b/tests/fixtures/s8/fft2_f4_13x1_C.npy new file mode 100644 index 0000000..655009d Binary files /dev/null and b/tests/fixtures/s8/fft2_f4_13x1_C.npy differ diff --git a/tests/fixtures/s8/fft2_f4_16x32_C.npy b/tests/fixtures/s8/fft2_f4_16x32_C.npy new file mode 100644 index 0000000..a378d01 Binary files /dev/null and b/tests/fixtures/s8/fft2_f4_16x32_C.npy differ diff --git a/tests/fixtures/s8/fft2_f4_1x16_C.npy b/tests/fixtures/s8/fft2_f4_1x16_C.npy new file mode 100644 index 0000000..a6f04f5 Binary files /dev/null and b/tests/fixtures/s8/fft2_f4_1x16_C.npy differ diff --git a/tests/fixtures/s8/fft2_f4_1x2_C.npy b/tests/fixtures/s8/fft2_f4_1x2_C.npy new file mode 100644 index 0000000..b239666 Binary files /dev/null and b/tests/fixtures/s8/fft2_f4_1x2_C.npy differ diff --git a/tests/fixtures/s8/fft2_f4_3x5_C.npy b/tests/fixtures/s8/fft2_f4_3x5_C.npy new file mode 100644 index 0000000..c09ce22 Binary files /dev/null and b/tests/fixtures/s8/fft2_f4_3x5_C.npy differ diff --git a/tests/fixtures/s8/fft2_f4_7x11_C.npy b/tests/fixtures/s8/fft2_f4_7x11_C.npy new file mode 100644 index 0000000..54583d2 Binary files /dev/null and b/tests/fixtures/s8/fft2_f4_7x11_C.npy differ diff --git a/tests/fixtures/s8/fft2_f8_13x1_C.npy b/tests/fixtures/s8/fft2_f8_13x1_C.npy new file mode 100644 index 0000000..e1e67f1 Binary files /dev/null and b/tests/fixtures/s8/fft2_f8_13x1_C.npy differ diff --git a/tests/fixtures/s8/fft2_f8_16x32_C.npy b/tests/fixtures/s8/fft2_f8_16x32_C.npy new file mode 100644 index 0000000..acbbda0 Binary files /dev/null and b/tests/fixtures/s8/fft2_f8_16x32_C.npy differ diff --git a/tests/fixtures/s8/fft2_f8_1x16_C.npy b/tests/fixtures/s8/fft2_f8_1x16_C.npy new file mode 100644 index 0000000..d38078b Binary files /dev/null and b/tests/fixtures/s8/fft2_f8_1x16_C.npy differ diff --git a/tests/fixtures/s8/fft2_f8_1x1_C.npy b/tests/fixtures/s8/fft2_f8_1x1_C.npy new file mode 100644 index 0000000..6b8ef8f Binary files /dev/null and b/tests/fixtures/s8/fft2_f8_1x1_C.npy differ diff --git a/tests/fixtures/s8/fft2_f8_1x2_C.npy b/tests/fixtures/s8/fft2_f8_1x2_C.npy new file mode 100644 index 0000000..152066b Binary files /dev/null and b/tests/fixtures/s8/fft2_f8_1x2_C.npy differ diff --git a/tests/fixtures/s8/fft2_f8_1x7_C.npy b/tests/fixtures/s8/fft2_f8_1x7_C.npy new file mode 100644 index 0000000..fedc06a Binary files /dev/null and b/tests/fixtures/s8/fft2_f8_1x7_C.npy differ diff --git a/tests/fixtures/s8/fft2_f8_31x17_C.npy b/tests/fixtures/s8/fft2_f8_31x17_C.npy new file mode 100644 index 0000000..154459d Binary files /dev/null and b/tests/fixtures/s8/fft2_f8_31x17_C.npy differ diff --git a/tests/fixtures/s8/fft2_f8_3x5_C.npy b/tests/fixtures/s8/fft2_f8_3x5_C.npy new file mode 100644 index 0000000..3a99e3e Binary files /dev/null and b/tests/fixtures/s8/fft2_f8_3x5_C.npy differ diff --git a/tests/fixtures/s8/fft2_f8_4x6_C.npy b/tests/fixtures/s8/fft2_f8_4x6_C.npy new file mode 100644 index 0000000..967ea03 Binary files /dev/null and b/tests/fixtures/s8/fft2_f8_4x6_C.npy differ diff --git a/tests/fixtures/s8/fft2_f8_64x64_C.npy b/tests/fixtures/s8/fft2_f8_64x64_C.npy new file mode 100644 index 0000000..d7462db Binary files /dev/null and b/tests/fixtures/s8/fft2_f8_64x64_C.npy differ diff --git a/tests/fixtures/s8/fft2_f8_7x11_C.npy b/tests/fixtures/s8/fft2_f8_7x11_C.npy new file mode 100644 index 0000000..f413390 Binary files /dev/null and b/tests/fixtures/s8/fft2_f8_7x11_C.npy differ diff --git a/tests/fixtures/s8/fft2_f8_8x8_C.npy b/tests/fixtures/s8/fft2_f8_8x8_C.npy new file mode 100644 index 0000000..018f3f1 Binary files /dev/null and b/tests/fixtures/s8/fft2_f8_8x8_C.npy differ diff --git a/tests/fixtures/s8/fft2_i4_1x7_C.npy b/tests/fixtures/s8/fft2_i4_1x7_C.npy new file mode 100644 index 0000000..8c822bd Binary files /dev/null and b/tests/fixtures/s8/fft2_i4_1x7_C.npy differ diff --git a/tests/fixtures/s8/fft2_i4_4x6_C.npy b/tests/fixtures/s8/fft2_i4_4x6_C.npy new file mode 100644 index 0000000..5b481a5 Binary files /dev/null and b/tests/fixtures/s8/fft2_i4_4x6_C.npy differ diff --git a/tests/fixtures/s8/fft2_i4_7x11_C.npy b/tests/fixtures/s8/fft2_i4_7x11_C.npy new file mode 100644 index 0000000..d74ad18 Binary files /dev/null and b/tests/fixtures/s8/fft2_i4_7x11_C.npy differ diff --git a/tests/fixtures/s8/fft2_i4_8x8_C.npy b/tests/fixtures/s8/fft2_i4_8x8_C.npy new file mode 100644 index 0000000..19f7c2f Binary files /dev/null and b/tests/fixtures/s8/fft2_i4_8x8_C.npy differ diff --git a/tests/fixtures/s8/fftshift_c16_1x1_C.npy b/tests/fixtures/s8/fftshift_c16_1x1_C.npy new file mode 100644 index 0000000..f72c79c Binary files /dev/null and b/tests/fixtures/s8/fftshift_c16_1x1_C.npy differ diff --git a/tests/fixtures/s8/fftshift_c16_1x4_C.npy b/tests/fixtures/s8/fftshift_c16_1x4_C.npy new file mode 100644 index 0000000..fb66318 Binary files /dev/null and b/tests/fixtures/s8/fftshift_c16_1x4_C.npy differ diff --git a/tests/fixtures/s8/fftshift_c16_3x5_C.npy b/tests/fixtures/s8/fftshift_c16_3x5_C.npy new file mode 100644 index 0000000..b88f741 Binary files /dev/null and b/tests/fixtures/s8/fftshift_c16_3x5_C.npy differ diff --git a/tests/fixtures/s8/fftshift_c16_4x6_C.npy b/tests/fixtures/s8/fftshift_c16_4x6_C.npy new file mode 100644 index 0000000..1c6414c Binary files /dev/null and b/tests/fixtures/s8/fftshift_c16_4x6_C.npy differ diff --git a/tests/fixtures/s8/fftshift_c16_5x5_C.npy b/tests/fixtures/s8/fftshift_c16_5x5_C.npy new file mode 100644 index 0000000..eca6d3d Binary files /dev/null and b/tests/fixtures/s8/fftshift_c16_5x5_C.npy differ diff --git a/tests/fixtures/s8/fftshift_c16_6x1_C.npy b/tests/fixtures/s8/fftshift_c16_6x1_C.npy new file mode 100644 index 0000000..be00df7 Binary files /dev/null and b/tests/fixtures/s8/fftshift_c16_6x1_C.npy differ diff --git a/tests/fixtures/s8/fftshift_f8_1x1_C.npy b/tests/fixtures/s8/fftshift_f8_1x1_C.npy new file mode 100644 index 0000000..24e92d0 Binary files /dev/null and b/tests/fixtures/s8/fftshift_f8_1x1_C.npy differ diff --git a/tests/fixtures/s8/fftshift_f8_1x4_C.npy b/tests/fixtures/s8/fftshift_f8_1x4_C.npy new file mode 100644 index 0000000..5498218 Binary files /dev/null and b/tests/fixtures/s8/fftshift_f8_1x4_C.npy differ diff --git a/tests/fixtures/s8/fftshift_f8_3x5_C.npy b/tests/fixtures/s8/fftshift_f8_3x5_C.npy new file mode 100644 index 0000000..2c25db1 Binary files /dev/null and b/tests/fixtures/s8/fftshift_f8_3x5_C.npy differ diff --git a/tests/fixtures/s8/fftshift_f8_4x6_C.npy b/tests/fixtures/s8/fftshift_f8_4x6_C.npy new file mode 100644 index 0000000..2d65e4e Binary files /dev/null and b/tests/fixtures/s8/fftshift_f8_4x6_C.npy differ diff --git a/tests/fixtures/s8/fftshift_f8_5x5_C.npy b/tests/fixtures/s8/fftshift_f8_5x5_C.npy new file mode 100644 index 0000000..24af7cd Binary files /dev/null and b/tests/fixtures/s8/fftshift_f8_5x5_C.npy differ diff --git a/tests/fixtures/s8/fftshift_f8_6x1_C.npy b/tests/fixtures/s8/fftshift_f8_6x1_C.npy new file mode 100644 index 0000000..c9392af Binary files /dev/null and b/tests/fixtures/s8/fftshift_f8_6x1_C.npy differ diff --git a/tests/fixtures/s8/fftshift_i4_1x1_C.npy b/tests/fixtures/s8/fftshift_i4_1x1_C.npy new file mode 100644 index 0000000..2a7deea Binary files /dev/null and b/tests/fixtures/s8/fftshift_i4_1x1_C.npy differ diff --git a/tests/fixtures/s8/fftshift_i4_1x4_C.npy b/tests/fixtures/s8/fftshift_i4_1x4_C.npy new file mode 100644 index 0000000..f5129a8 Binary files /dev/null and b/tests/fixtures/s8/fftshift_i4_1x4_C.npy differ diff --git a/tests/fixtures/s8/fftshift_i4_3x5_C.npy b/tests/fixtures/s8/fftshift_i4_3x5_C.npy new file mode 100644 index 0000000..7cef88b Binary files /dev/null and b/tests/fixtures/s8/fftshift_i4_3x5_C.npy differ diff --git a/tests/fixtures/s8/fftshift_i4_4x6_C.npy b/tests/fixtures/s8/fftshift_i4_4x6_C.npy new file mode 100644 index 0000000..58e062e Binary files /dev/null and b/tests/fixtures/s8/fftshift_i4_4x6_C.npy differ diff --git a/tests/fixtures/s8/fftshift_i4_5x5_C.npy b/tests/fixtures/s8/fftshift_i4_5x5_C.npy new file mode 100644 index 0000000..f31a02f Binary files /dev/null and b/tests/fixtures/s8/fftshift_i4_5x5_C.npy differ diff --git a/tests/fixtures/s8/fftshift_i4_6x1_C.npy b/tests/fixtures/s8/fftshift_i4_6x1_C.npy new file mode 100644 index 0000000..a984b40 Binary files /dev/null and b/tests/fixtures/s8/fftshift_i4_6x1_C.npy differ diff --git a/tests/fixtures/s8/fourier_c16_1x1_A.npy b/tests/fixtures/s8/fourier_c16_1x1_A.npy new file mode 100644 index 0000000..4f8c507 Binary files /dev/null and b/tests/fixtures/s8/fourier_c16_1x1_A.npy differ diff --git a/tests/fixtures/s8/fourier_c16_1x7_A.npy b/tests/fixtures/s8/fourier_c16_1x7_A.npy new file mode 100644 index 0000000..17fa2d1 Binary files /dev/null and b/tests/fixtures/s8/fourier_c16_1x7_A.npy differ diff --git a/tests/fixtures/s8/fourier_c16_31x17_A.npy b/tests/fixtures/s8/fourier_c16_31x17_A.npy new file mode 100644 index 0000000..d26c9cc Binary files /dev/null and b/tests/fixtures/s8/fourier_c16_31x17_A.npy differ diff --git a/tests/fixtures/s8/fourier_c16_3x5_A.npy b/tests/fixtures/s8/fourier_c16_3x5_A.npy new file mode 100644 index 0000000..b334230 Binary files /dev/null and b/tests/fixtures/s8/fourier_c16_3x5_A.npy differ diff --git a/tests/fixtures/s8/fourier_c16_4x6_A.npy b/tests/fixtures/s8/fourier_c16_4x6_A.npy new file mode 100644 index 0000000..589e397 Binary files /dev/null and b/tests/fixtures/s8/fourier_c16_4x6_A.npy differ diff --git a/tests/fixtures/s8/fourier_c16_7x11_A.npy b/tests/fixtures/s8/fourier_c16_7x11_A.npy new file mode 100644 index 0000000..e4e3233 Binary files /dev/null and b/tests/fixtures/s8/fourier_c16_7x11_A.npy differ diff --git a/tests/fixtures/s8/fourier_c16_8x8_A.npy b/tests/fixtures/s8/fourier_c16_8x8_A.npy new file mode 100644 index 0000000..c8d0f13 Binary files /dev/null and b/tests/fixtures/s8/fourier_c16_8x8_A.npy differ diff --git a/tests/fixtures/s8/fourier_c8_3x5_A.npy b/tests/fixtures/s8/fourier_c8_3x5_A.npy new file mode 100644 index 0000000..6295d36 Binary files /dev/null and b/tests/fixtures/s8/fourier_c8_3x5_A.npy differ diff --git a/tests/fixtures/s8/fourier_c8_8x8_A.npy b/tests/fixtures/s8/fourier_c8_8x8_A.npy new file mode 100644 index 0000000..917c7cb Binary files /dev/null and b/tests/fixtures/s8/fourier_c8_8x8_A.npy differ diff --git a/tests/fixtures/s8/fourier_f4_13x1_A.npy b/tests/fixtures/s8/fourier_f4_13x1_A.npy new file mode 100644 index 0000000..258f834 Binary files /dev/null and b/tests/fixtures/s8/fourier_f4_13x1_A.npy differ diff --git a/tests/fixtures/s8/fourier_f4_16x32_A.npy b/tests/fixtures/s8/fourier_f4_16x32_A.npy new file mode 100644 index 0000000..8af4c61 Binary files /dev/null and b/tests/fixtures/s8/fourier_f4_16x32_A.npy differ diff --git a/tests/fixtures/s8/fourier_f4_1x16_A.npy b/tests/fixtures/s8/fourier_f4_1x16_A.npy new file mode 100644 index 0000000..fe8ef3b Binary files /dev/null and b/tests/fixtures/s8/fourier_f4_1x16_A.npy differ diff --git a/tests/fixtures/s8/fourier_f4_1x2_A.npy b/tests/fixtures/s8/fourier_f4_1x2_A.npy new file mode 100644 index 0000000..14e82fa Binary files /dev/null and b/tests/fixtures/s8/fourier_f4_1x2_A.npy differ diff --git a/tests/fixtures/s8/fourier_f4_3x5_A.npy b/tests/fixtures/s8/fourier_f4_3x5_A.npy new file mode 100644 index 0000000..c7f9886 Binary files /dev/null and b/tests/fixtures/s8/fourier_f4_3x5_A.npy differ diff --git a/tests/fixtures/s8/fourier_f4_7x11_A.npy b/tests/fixtures/s8/fourier_f4_7x11_A.npy new file mode 100644 index 0000000..689d5eb Binary files /dev/null and b/tests/fixtures/s8/fourier_f4_7x11_A.npy differ diff --git a/tests/fixtures/s8/fourier_f8_13x1_A.npy b/tests/fixtures/s8/fourier_f8_13x1_A.npy new file mode 100644 index 0000000..f65d312 Binary files /dev/null and b/tests/fixtures/s8/fourier_f8_13x1_A.npy differ diff --git a/tests/fixtures/s8/fourier_f8_16x32_A.npy b/tests/fixtures/s8/fourier_f8_16x32_A.npy new file mode 100644 index 0000000..a5c326a Binary files /dev/null and b/tests/fixtures/s8/fourier_f8_16x32_A.npy differ diff --git a/tests/fixtures/s8/fourier_f8_1x16_A.npy b/tests/fixtures/s8/fourier_f8_1x16_A.npy new file mode 100644 index 0000000..a793c5e Binary files /dev/null and b/tests/fixtures/s8/fourier_f8_1x16_A.npy differ diff --git a/tests/fixtures/s8/fourier_f8_1x1_A.npy b/tests/fixtures/s8/fourier_f8_1x1_A.npy new file mode 100644 index 0000000..c5d4a5d Binary files /dev/null and b/tests/fixtures/s8/fourier_f8_1x1_A.npy differ diff --git a/tests/fixtures/s8/fourier_f8_1x2_A.npy b/tests/fixtures/s8/fourier_f8_1x2_A.npy new file mode 100644 index 0000000..9995ed9 Binary files /dev/null and b/tests/fixtures/s8/fourier_f8_1x2_A.npy differ diff --git a/tests/fixtures/s8/fourier_f8_1x7_A.npy b/tests/fixtures/s8/fourier_f8_1x7_A.npy new file mode 100644 index 0000000..693780c Binary files /dev/null and b/tests/fixtures/s8/fourier_f8_1x7_A.npy differ diff --git a/tests/fixtures/s8/fourier_f8_31x17_A.npy b/tests/fixtures/s8/fourier_f8_31x17_A.npy new file mode 100644 index 0000000..263c772 Binary files /dev/null and b/tests/fixtures/s8/fourier_f8_31x17_A.npy differ diff --git a/tests/fixtures/s8/fourier_f8_3x5_A.npy b/tests/fixtures/s8/fourier_f8_3x5_A.npy new file mode 100644 index 0000000..1c9c25e Binary files /dev/null and b/tests/fixtures/s8/fourier_f8_3x5_A.npy differ diff --git a/tests/fixtures/s8/fourier_f8_4x6_A.npy b/tests/fixtures/s8/fourier_f8_4x6_A.npy new file mode 100644 index 0000000..c87c3ba Binary files /dev/null and b/tests/fixtures/s8/fourier_f8_4x6_A.npy differ diff --git a/tests/fixtures/s8/fourier_f8_64x64_A.npy b/tests/fixtures/s8/fourier_f8_64x64_A.npy new file mode 100644 index 0000000..0542eb8 Binary files /dev/null and b/tests/fixtures/s8/fourier_f8_64x64_A.npy differ diff --git a/tests/fixtures/s8/fourier_f8_7x11_A.npy b/tests/fixtures/s8/fourier_f8_7x11_A.npy new file mode 100644 index 0000000..f41afae Binary files /dev/null and b/tests/fixtures/s8/fourier_f8_7x11_A.npy differ diff --git a/tests/fixtures/s8/fourier_f8_8x8_A.npy b/tests/fixtures/s8/fourier_f8_8x8_A.npy new file mode 100644 index 0000000..f22762f Binary files /dev/null and b/tests/fixtures/s8/fourier_f8_8x8_A.npy differ diff --git a/tests/fixtures/s8/fourier_i4_1x7_A.npy b/tests/fixtures/s8/fourier_i4_1x7_A.npy new file mode 100644 index 0000000..e2b94dd Binary files /dev/null and b/tests/fixtures/s8/fourier_i4_1x7_A.npy differ diff --git a/tests/fixtures/s8/fourier_i4_4x6_A.npy b/tests/fixtures/s8/fourier_i4_4x6_A.npy new file mode 100644 index 0000000..8b3a1d4 Binary files /dev/null and b/tests/fixtures/s8/fourier_i4_4x6_A.npy differ diff --git a/tests/fixtures/s8/fourier_i4_7x11_A.npy b/tests/fixtures/s8/fourier_i4_7x11_A.npy new file mode 100644 index 0000000..04e19cc Binary files /dev/null and b/tests/fixtures/s8/fourier_i4_7x11_A.npy differ diff --git a/tests/fixtures/s8/fourier_i4_8x8_A.npy b/tests/fixtures/s8/fourier_i4_8x8_A.npy new file mode 100644 index 0000000..00ed9b7 Binary files /dev/null and b/tests/fixtures/s8/fourier_i4_8x8_A.npy differ diff --git a/tests/fixtures/s8/ifft2_c16_1x1_C.npy b/tests/fixtures/s8/ifft2_c16_1x1_C.npy new file mode 100644 index 0000000..4f8c507 Binary files /dev/null and b/tests/fixtures/s8/ifft2_c16_1x1_C.npy differ diff --git a/tests/fixtures/s8/ifft2_c16_1x7_C.npy b/tests/fixtures/s8/ifft2_c16_1x7_C.npy new file mode 100644 index 0000000..6d410bb Binary files /dev/null and b/tests/fixtures/s8/ifft2_c16_1x7_C.npy differ diff --git a/tests/fixtures/s8/ifft2_c16_31x17_C.npy b/tests/fixtures/s8/ifft2_c16_31x17_C.npy new file mode 100644 index 0000000..1961266 Binary files /dev/null and b/tests/fixtures/s8/ifft2_c16_31x17_C.npy differ diff --git a/tests/fixtures/s8/ifft2_c16_3x5_C.npy b/tests/fixtures/s8/ifft2_c16_3x5_C.npy new file mode 100644 index 0000000..b2a71ab Binary files /dev/null and b/tests/fixtures/s8/ifft2_c16_3x5_C.npy differ diff --git a/tests/fixtures/s8/ifft2_c16_4x6_C.npy b/tests/fixtures/s8/ifft2_c16_4x6_C.npy new file mode 100644 index 0000000..ccb9498 Binary files /dev/null and b/tests/fixtures/s8/ifft2_c16_4x6_C.npy differ diff --git a/tests/fixtures/s8/ifft2_c16_7x11_C.npy b/tests/fixtures/s8/ifft2_c16_7x11_C.npy new file mode 100644 index 0000000..66fd552 Binary files /dev/null and b/tests/fixtures/s8/ifft2_c16_7x11_C.npy differ diff --git a/tests/fixtures/s8/ifft2_c16_8x8_C.npy b/tests/fixtures/s8/ifft2_c16_8x8_C.npy new file mode 100644 index 0000000..d0d14fa Binary files /dev/null and b/tests/fixtures/s8/ifft2_c16_8x8_C.npy differ diff --git a/tests/fixtures/s8/ifft2_c8_3x5_C.npy b/tests/fixtures/s8/ifft2_c8_3x5_C.npy new file mode 100644 index 0000000..7f40243 Binary files /dev/null and b/tests/fixtures/s8/ifft2_c8_3x5_C.npy differ diff --git a/tests/fixtures/s8/ifft2_c8_8x8_C.npy b/tests/fixtures/s8/ifft2_c8_8x8_C.npy new file mode 100644 index 0000000..47423be Binary files /dev/null and b/tests/fixtures/s8/ifft2_c8_8x8_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f4_13x1_C.npy b/tests/fixtures/s8/ifft2_f4_13x1_C.npy new file mode 100644 index 0000000..fe39e6c Binary files /dev/null and b/tests/fixtures/s8/ifft2_f4_13x1_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f4_16x32_C.npy b/tests/fixtures/s8/ifft2_f4_16x32_C.npy new file mode 100644 index 0000000..fc2ac53 Binary files /dev/null and b/tests/fixtures/s8/ifft2_f4_16x32_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f4_1x16_C.npy b/tests/fixtures/s8/ifft2_f4_1x16_C.npy new file mode 100644 index 0000000..44fff69 Binary files /dev/null and b/tests/fixtures/s8/ifft2_f4_1x16_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f4_1x2_C.npy b/tests/fixtures/s8/ifft2_f4_1x2_C.npy new file mode 100644 index 0000000..49ec044 Binary files /dev/null and b/tests/fixtures/s8/ifft2_f4_1x2_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f4_3x5_C.npy b/tests/fixtures/s8/ifft2_f4_3x5_C.npy new file mode 100644 index 0000000..321b934 Binary files /dev/null and b/tests/fixtures/s8/ifft2_f4_3x5_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f4_7x11_C.npy b/tests/fixtures/s8/ifft2_f4_7x11_C.npy new file mode 100644 index 0000000..f4a8a93 Binary files /dev/null and b/tests/fixtures/s8/ifft2_f4_7x11_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f8_13x1_C.npy b/tests/fixtures/s8/ifft2_f8_13x1_C.npy new file mode 100644 index 0000000..4dc1121 Binary files /dev/null and b/tests/fixtures/s8/ifft2_f8_13x1_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f8_16x32_C.npy b/tests/fixtures/s8/ifft2_f8_16x32_C.npy new file mode 100644 index 0000000..4cec68c Binary files /dev/null and b/tests/fixtures/s8/ifft2_f8_16x32_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f8_1x16_C.npy b/tests/fixtures/s8/ifft2_f8_1x16_C.npy new file mode 100644 index 0000000..11faac2 Binary files /dev/null and b/tests/fixtures/s8/ifft2_f8_1x16_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f8_1x1_C.npy b/tests/fixtures/s8/ifft2_f8_1x1_C.npy new file mode 100644 index 0000000..6b8ef8f Binary files /dev/null and b/tests/fixtures/s8/ifft2_f8_1x1_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f8_1x2_C.npy b/tests/fixtures/s8/ifft2_f8_1x2_C.npy new file mode 100644 index 0000000..732f29a Binary files /dev/null and b/tests/fixtures/s8/ifft2_f8_1x2_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f8_1x7_C.npy b/tests/fixtures/s8/ifft2_f8_1x7_C.npy new file mode 100644 index 0000000..09a1769 Binary files /dev/null and b/tests/fixtures/s8/ifft2_f8_1x7_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f8_31x17_C.npy b/tests/fixtures/s8/ifft2_f8_31x17_C.npy new file mode 100644 index 0000000..f298f10 Binary files /dev/null and b/tests/fixtures/s8/ifft2_f8_31x17_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f8_3x5_C.npy b/tests/fixtures/s8/ifft2_f8_3x5_C.npy new file mode 100644 index 0000000..67377d0 Binary files /dev/null and b/tests/fixtures/s8/ifft2_f8_3x5_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f8_4x6_C.npy b/tests/fixtures/s8/ifft2_f8_4x6_C.npy new file mode 100644 index 0000000..c0897c5 Binary files /dev/null and b/tests/fixtures/s8/ifft2_f8_4x6_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f8_64x64_C.npy b/tests/fixtures/s8/ifft2_f8_64x64_C.npy new file mode 100644 index 0000000..fe185a3 Binary files /dev/null and b/tests/fixtures/s8/ifft2_f8_64x64_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f8_7x11_C.npy b/tests/fixtures/s8/ifft2_f8_7x11_C.npy new file mode 100644 index 0000000..46d69b7 Binary files /dev/null and b/tests/fixtures/s8/ifft2_f8_7x11_C.npy differ diff --git a/tests/fixtures/s8/ifft2_f8_8x8_C.npy b/tests/fixtures/s8/ifft2_f8_8x8_C.npy new file mode 100644 index 0000000..0a38d26 Binary files /dev/null and b/tests/fixtures/s8/ifft2_f8_8x8_C.npy differ diff --git a/tests/fixtures/s8/ifft2_i4_1x7_C.npy b/tests/fixtures/s8/ifft2_i4_1x7_C.npy new file mode 100644 index 0000000..d822b54 Binary files /dev/null and b/tests/fixtures/s8/ifft2_i4_1x7_C.npy differ diff --git a/tests/fixtures/s8/ifft2_i4_4x6_C.npy b/tests/fixtures/s8/ifft2_i4_4x6_C.npy new file mode 100644 index 0000000..8b25862 Binary files /dev/null and b/tests/fixtures/s8/ifft2_i4_4x6_C.npy differ diff --git a/tests/fixtures/s8/ifft2_i4_7x11_C.npy b/tests/fixtures/s8/ifft2_i4_7x11_C.npy new file mode 100644 index 0000000..fa6ba75 Binary files /dev/null and b/tests/fixtures/s8/ifft2_i4_7x11_C.npy differ diff --git a/tests/fixtures/s8/ifft2_i4_8x8_C.npy b/tests/fixtures/s8/ifft2_i4_8x8_C.npy new file mode 100644 index 0000000..a583406 Binary files /dev/null and b/tests/fixtures/s8/ifft2_i4_8x8_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_c16_1x1_C.npy b/tests/fixtures/s8/ifftshift_c16_1x1_C.npy new file mode 100644 index 0000000..f72c79c Binary files /dev/null and b/tests/fixtures/s8/ifftshift_c16_1x1_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_c16_1x4_C.npy b/tests/fixtures/s8/ifftshift_c16_1x4_C.npy new file mode 100644 index 0000000..fb66318 Binary files /dev/null and b/tests/fixtures/s8/ifftshift_c16_1x4_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_c16_3x5_C.npy b/tests/fixtures/s8/ifftshift_c16_3x5_C.npy new file mode 100644 index 0000000..acf8b4c Binary files /dev/null and b/tests/fixtures/s8/ifftshift_c16_3x5_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_c16_4x6_C.npy b/tests/fixtures/s8/ifftshift_c16_4x6_C.npy new file mode 100644 index 0000000..1c6414c Binary files /dev/null and b/tests/fixtures/s8/ifftshift_c16_4x6_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_c16_5x5_C.npy b/tests/fixtures/s8/ifftshift_c16_5x5_C.npy new file mode 100644 index 0000000..b28ef40 Binary files /dev/null and b/tests/fixtures/s8/ifftshift_c16_5x5_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_c16_6x1_C.npy b/tests/fixtures/s8/ifftshift_c16_6x1_C.npy new file mode 100644 index 0000000..be00df7 Binary files /dev/null and b/tests/fixtures/s8/ifftshift_c16_6x1_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_f8_1x1_C.npy b/tests/fixtures/s8/ifftshift_f8_1x1_C.npy new file mode 100644 index 0000000..24e92d0 Binary files /dev/null and b/tests/fixtures/s8/ifftshift_f8_1x1_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_f8_1x4_C.npy b/tests/fixtures/s8/ifftshift_f8_1x4_C.npy new file mode 100644 index 0000000..5498218 Binary files /dev/null and b/tests/fixtures/s8/ifftshift_f8_1x4_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_f8_3x5_C.npy b/tests/fixtures/s8/ifftshift_f8_3x5_C.npy new file mode 100644 index 0000000..0414aa3 Binary files /dev/null and b/tests/fixtures/s8/ifftshift_f8_3x5_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_f8_4x6_C.npy b/tests/fixtures/s8/ifftshift_f8_4x6_C.npy new file mode 100644 index 0000000..2d65e4e Binary files /dev/null and b/tests/fixtures/s8/ifftshift_f8_4x6_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_f8_5x5_C.npy b/tests/fixtures/s8/ifftshift_f8_5x5_C.npy new file mode 100644 index 0000000..8c73619 Binary files /dev/null and b/tests/fixtures/s8/ifftshift_f8_5x5_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_f8_6x1_C.npy b/tests/fixtures/s8/ifftshift_f8_6x1_C.npy new file mode 100644 index 0000000..c9392af Binary files /dev/null and b/tests/fixtures/s8/ifftshift_f8_6x1_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_i4_1x1_C.npy b/tests/fixtures/s8/ifftshift_i4_1x1_C.npy new file mode 100644 index 0000000..2a7deea Binary files /dev/null and b/tests/fixtures/s8/ifftshift_i4_1x1_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_i4_1x4_C.npy b/tests/fixtures/s8/ifftshift_i4_1x4_C.npy new file mode 100644 index 0000000..f5129a8 Binary files /dev/null and b/tests/fixtures/s8/ifftshift_i4_1x4_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_i4_3x5_C.npy b/tests/fixtures/s8/ifftshift_i4_3x5_C.npy new file mode 100644 index 0000000..19ab052 Binary files /dev/null and b/tests/fixtures/s8/ifftshift_i4_3x5_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_i4_4x6_C.npy b/tests/fixtures/s8/ifftshift_i4_4x6_C.npy new file mode 100644 index 0000000..58e062e Binary files /dev/null and b/tests/fixtures/s8/ifftshift_i4_4x6_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_i4_5x5_C.npy b/tests/fixtures/s8/ifftshift_i4_5x5_C.npy new file mode 100644 index 0000000..2ac1268 Binary files /dev/null and b/tests/fixtures/s8/ifftshift_i4_5x5_C.npy differ diff --git a/tests/fixtures/s8/ifftshift_i4_6x1_C.npy b/tests/fixtures/s8/ifftshift_i4_6x1_C.npy new file mode 100644 index 0000000..a984b40 Binary files /dev/null and b/tests/fixtures/s8/ifftshift_i4_6x1_C.npy differ diff --git a/tests/fixtures/s8/make_fixtures.py b/tests/fixtures/s8/make_fixtures.py new file mode 100644 index 0000000..5fc8734 --- /dev/null +++ b/tests/fixtures/s8/make_fixtures.py @@ -0,0 +1,190 @@ +#!/usr/bin/env python3 +"""S8 (PR-11, PR-14): writes the committed numpy/scipy oracle fixtures for tests/cases/s8_fixtures.hpp. + +Run from anywhere: `python3 tests/fixtures/s8/make_fixtures.py` (numpy 2.5.3, scipy 1.18.1). Every input comes +from a fixed seed, so a second run reproduces every .npy file and manifest.txt byte for byte with the same python, +numpy and scipy versions, which the manifest header records. + +manifest.txt: comment lines start with '#'; every other line is ` `, files relative to +this directory. Case names are `__`; the element type is the field after the op and names the +input's dtype: `f4` (= rb and ca >= cb) or (rb >= ra and cb >= ca): + modes.append("valid") + tol = 0.0 if kind == "i4" else 4 * EPS * float(np.abs(a).sum()) * float(np.abs(b).sum()) + pair = "%s_%s_%s%s" % (kind, shape(a), shape(b), tag) + inputs = [save("conv_" + pair, "A", a, kind), save("conv_" + pair, "B", b, kind)] # shared by the modes + for mode in modes: + c = scipy.signal.convolve2d(a, b, mode) + case = "conv_%s_%s" % (mode, pair) + add(case, "conv_" + mode, tol, kind, [("C", c)], inputs) + + +def conv_cases(): + a = real(101, 7, 9) # asymmetric A + conv(a, real(102, 1, 1), "f8") # 1x1 + conv(a, real(103, 3, 5), "f8") # odd, asymmetric + conv(a, real(104, 2, 4), "f8") # even + conv(a, real(105, 4, 3), "f8") # even rows, odd columns + conv(a, real(106, 3, 12), "f8") # oversized in columns only: no valid + conv(a, real(107, 10, 11), "f8") # oversized in both: valid swaps + conv(real(108, 1, 6), real(109, 1, 3), "f8") + conv(real(110, 5, 1), real(111, 2, 1), "f8") + ai = ints(121, 6, 8) + conv(ai, ints(122, 1, 1), "i4") + conv(ai, ints(123, 3, 5), "i4") + conv(ai, ints(124, 2, 4), "i4") + conv(ai, ints(125, 9, 10), "i4") + + +# ---- fft2, ifft2 (S8-R2) and the shifts (S8-R3) ---- +FFT_SIZES = [(1, 1), (1, 2), (1, 7), (1, 16), (3, 5), (4, 6), (7, 11), (13, 1), (31, 17), (8, 8), (16, 32), (64, 64)] +SHIFT_SIZES = [(1, 1), (1, 4), (3, 5), (4, 6), (5, 5), (6, 1)] +RESULT_KIND = {"f4": "c8", "c8": "c8", "f8": "c16", "i4": "c16", "c16": "c16"} # D-031, numpy's own dtypes + + +def sample(kind, seed, m, n): + if kind == "i4": + return ints(seed, m, n) + if kind in ("f4", "f8"): + return real(seed, m, n).astype(DTYPES[kind]) + return (real(seed, m, n) + 1j * real(seed + 500, m, n)).astype(DTYPES[kind]) + + +def big_l(n): + """L = ceil(log2(2N))""" + l, p = 0, 1 + while p < 2 * n: + p, l = p * 2, l + 1 + return l + + +def fourier(kind, m, n, seed): + a = sample(kind, seed, m, n) + out = RESULT_KIND[kind] + eps = float(np.finfo(np.float32 if out == "c8" else np.float64).eps) + tol = 16 * big_l(m * n) * eps * float(np.abs(a.astype(np.complex128)).sum()) + tag = "%s_%dx%d" % (kind, m, n) + inputs = [save("fourier_" + tag, "A", a, kind)] # shared by fft2 and ifft2 + f, i = np.fft.fft2(a), np.fft.ifft2(a) + assert f.dtype == np.dtype(DTYPES[out]) and i.dtype == np.dtype(DTYPES[out]), (kind, f.dtype, i.dtype) + add("fft2_" + tag, "fft2", tol, out, [("C", f)], inputs) + add("ifft2_" + tag, "ifft2", tol / (m * n), out, [("C", i)], inputs) + + +def fourier_cases(): + for k, (m, n) in enumerate(FFT_SIZES): + fourier("f8", m, n, 201 + k) + for k, (m, n) in enumerate([(1, 1), (1, 7), (3, 5), (4, 6), (7, 11), (8, 8), (31, 17)]): + fourier("c16", m, n, 231 + k) + for k, (m, n) in enumerate([(1, 2), (1, 16), (3, 5), (7, 11), (13, 1), (16, 32)]): + fourier("f4", m, n, 251 + k) + for k, (m, n) in enumerate([(3, 5), (8, 8)]): + fourier("c8", m, n, 271 + k) + for k, (m, n) in enumerate([(1, 7), (4, 6), (7, 11), (8, 8)]): + fourier("i4", m, n, 281 + k) + + +def shift_cases(): + for j, kind in enumerate(["f8", "i4", "c16"]): + for k, (m, n) in enumerate(SHIFT_SIZES): + a = sample(kind, 301 + 10 * j + k, m, n) + tag = "%s_%dx%d" % (kind, m, n) + inputs = [save("shift_" + tag, "A", a, kind)] + add("fftshift_" + tag, "fftshift", 0.0, kind, [("C", np.fft.fftshift(a))], inputs) + add("ifftshift_" + tag, "ifftshift", 0.0, kind, [("C", np.fft.ifftshift(a))], inputs) + + +def main(): + for f in os.listdir(HERE): + if f.endswith(".npy"): + os.remove(os.path.join(HERE, f)) + + conv_cases() + fourier_cases() + shift_cases() + + header = [ + "# S8 oracle fixtures, written by tests/fixtures/s8/make_fixtures.py; do not edit by hand.", + "# python %s, numpy %s, scipy %s" % (sys.version.split()[0], np.__version__, scipy.__version__), + "# ", + ] + with open(os.path.join(HERE, "manifest.txt"), "w", newline="\n") as f: + f.write("\n".join(header + LINES) + "\n") + + +if __name__ == "__main__": + main() diff --git a/tests/fixtures/s8/manifest.txt b/tests/fixtures/s8/manifest.txt new file mode 100644 index 0000000..9be12f9 --- /dev/null +++ b/tests/fixtures/s8/manifest.txt @@ -0,0 +1,136 @@ +# S8 oracle fixtures, written by tests/fixtures/s8/make_fixtures.py; do not edit by hand. +# python 3.14.7, numpy 2.5.3, scipy 1.18.1 +# +conv_full_f8_7x9_1x1 conv_full 2e-14 conv_f8_7x9_1x1_A.npy conv_f8_7x9_1x1_B.npy conv_full_f8_7x9_1x1_C.npy +conv_same_f8_7x9_1x1 conv_same 2e-14 conv_f8_7x9_1x1_A.npy conv_f8_7x9_1x1_B.npy conv_same_f8_7x9_1x1_C.npy +conv_valid_f8_7x9_1x1 conv_valid 2e-14 conv_f8_7x9_1x1_A.npy conv_f8_7x9_1x1_B.npy conv_valid_f8_7x9_1x1_C.npy +conv_full_f8_7x9_3x5 conv_full 2.1e-13 conv_f8_7x9_3x5_A.npy conv_f8_7x9_3x5_B.npy conv_full_f8_7x9_3x5_C.npy +conv_same_f8_7x9_3x5 conv_same 2.1e-13 conv_f8_7x9_3x5_A.npy conv_f8_7x9_3x5_B.npy conv_same_f8_7x9_3x5_C.npy +conv_valid_f8_7x9_3x5 conv_valid 2.1e-13 conv_f8_7x9_3x5_A.npy conv_f8_7x9_3x5_B.npy conv_valid_f8_7x9_3x5_C.npy +conv_full_f8_7x9_2x4 conv_full 1.1e-13 conv_f8_7x9_2x4_A.npy conv_f8_7x9_2x4_B.npy conv_full_f8_7x9_2x4_C.npy +conv_same_f8_7x9_2x4 conv_same 1.1e-13 conv_f8_7x9_2x4_A.npy conv_f8_7x9_2x4_B.npy conv_same_f8_7x9_2x4_C.npy +conv_valid_f8_7x9_2x4 conv_valid 1.1e-13 conv_f8_7x9_2x4_A.npy conv_f8_7x9_2x4_B.npy conv_valid_f8_7x9_2x4_C.npy +conv_full_f8_7x9_4x3 conv_full 1.5e-13 conv_f8_7x9_4x3_A.npy conv_f8_7x9_4x3_B.npy conv_full_f8_7x9_4x3_C.npy +conv_same_f8_7x9_4x3 conv_same 1.5e-13 conv_f8_7x9_4x3_A.npy conv_f8_7x9_4x3_B.npy conv_same_f8_7x9_4x3_C.npy +conv_valid_f8_7x9_4x3 conv_valid 1.5e-13 conv_f8_7x9_4x3_A.npy conv_f8_7x9_4x3_B.npy conv_valid_f8_7x9_4x3_C.npy +conv_full_f8_7x9_3x12 conv_full 6.5e-13 conv_f8_7x9_3x12_A.npy conv_f8_7x9_3x12_B.npy conv_full_f8_7x9_3x12_C.npy +conv_same_f8_7x9_3x12 conv_same 6.5e-13 conv_f8_7x9_3x12_A.npy conv_f8_7x9_3x12_B.npy conv_same_f8_7x9_3x12_C.npy +conv_full_f8_7x9_10x11 conv_full 1.6e-12 conv_f8_7x9_10x11_A.npy conv_f8_7x9_10x11_B.npy conv_full_f8_7x9_10x11_C.npy +conv_same_f8_7x9_10x11 conv_same 1.6e-12 conv_f8_7x9_10x11_A.npy conv_f8_7x9_10x11_B.npy conv_same_f8_7x9_10x11_C.npy +conv_valid_f8_7x9_10x11 conv_valid 1.6e-12 conv_f8_7x9_10x11_A.npy conv_f8_7x9_10x11_B.npy conv_valid_f8_7x9_10x11_C.npy +conv_full_f8_1x6_1x3 conv_full 4e-15 conv_f8_1x6_1x3_A.npy conv_f8_1x6_1x3_B.npy conv_full_f8_1x6_1x3_C.npy +conv_same_f8_1x6_1x3 conv_same 4e-15 conv_f8_1x6_1x3_A.npy conv_f8_1x6_1x3_B.npy conv_same_f8_1x6_1x3_C.npy +conv_valid_f8_1x6_1x3 conv_valid 4e-15 conv_f8_1x6_1x3_A.npy conv_f8_1x6_1x3_B.npy conv_valid_f8_1x6_1x3_C.npy +conv_full_f8_5x1_2x1 conv_full 2.3e-15 conv_f8_5x1_2x1_A.npy conv_f8_5x1_2x1_B.npy conv_full_f8_5x1_2x1_C.npy +conv_same_f8_5x1_2x1 conv_same 2.3e-15 conv_f8_5x1_2x1_A.npy conv_f8_5x1_2x1_B.npy conv_same_f8_5x1_2x1_C.npy +conv_valid_f8_5x1_2x1 conv_valid 2.3e-15 conv_f8_5x1_2x1_A.npy conv_f8_5x1_2x1_B.npy conv_valid_f8_5x1_2x1_C.npy +conv_full_i4_6x8_1x1 conv_full 0.0 conv_i4_6x8_1x1_A.npy conv_i4_6x8_1x1_B.npy conv_full_i4_6x8_1x1_C.npy +conv_same_i4_6x8_1x1 conv_same 0.0 conv_i4_6x8_1x1_A.npy conv_i4_6x8_1x1_B.npy conv_same_i4_6x8_1x1_C.npy +conv_valid_i4_6x8_1x1 conv_valid 0.0 conv_i4_6x8_1x1_A.npy conv_i4_6x8_1x1_B.npy conv_valid_i4_6x8_1x1_C.npy +conv_full_i4_6x8_3x5 conv_full 0.0 conv_i4_6x8_3x5_A.npy conv_i4_6x8_3x5_B.npy conv_full_i4_6x8_3x5_C.npy +conv_same_i4_6x8_3x5 conv_same 0.0 conv_i4_6x8_3x5_A.npy conv_i4_6x8_3x5_B.npy conv_same_i4_6x8_3x5_C.npy +conv_valid_i4_6x8_3x5 conv_valid 0.0 conv_i4_6x8_3x5_A.npy conv_i4_6x8_3x5_B.npy conv_valid_i4_6x8_3x5_C.npy +conv_full_i4_6x8_2x4 conv_full 0.0 conv_i4_6x8_2x4_A.npy conv_i4_6x8_2x4_B.npy conv_full_i4_6x8_2x4_C.npy +conv_same_i4_6x8_2x4 conv_same 0.0 conv_i4_6x8_2x4_A.npy conv_i4_6x8_2x4_B.npy conv_same_i4_6x8_2x4_C.npy +conv_valid_i4_6x8_2x4 conv_valid 0.0 conv_i4_6x8_2x4_A.npy conv_i4_6x8_2x4_B.npy conv_valid_i4_6x8_2x4_C.npy +conv_full_i4_6x8_9x10 conv_full 0.0 conv_i4_6x8_9x10_A.npy conv_i4_6x8_9x10_B.npy conv_full_i4_6x8_9x10_C.npy +conv_same_i4_6x8_9x10 conv_same 0.0 conv_i4_6x8_9x10_A.npy conv_i4_6x8_9x10_B.npy conv_same_i4_6x8_9x10_C.npy +conv_valid_i4_6x8_9x10 conv_valid 0.0 conv_i4_6x8_9x10_A.npy conv_i4_6x8_9x10_B.npy conv_valid_i4_6x8_9x10_C.npy +fft2_f8_1x1 fft2 3.7e-15 fourier_f8_1x1_A.npy fft2_f8_1x1_C.npy +ifft2_f8_1x1 ifft2 3.7e-15 fourier_f8_1x1_A.npy ifft2_f8_1x1_C.npy +fft2_f8_1x2 fft2 3.1e-15 fourier_f8_1x2_A.npy fft2_f8_1x2_C.npy +ifft2_f8_1x2 ifft2 1.6e-15 fourier_f8_1x2_A.npy ifft2_f8_1x2_C.npy +fft2_f8_1x7 fft2 5.3e-14 fourier_f8_1x7_A.npy fft2_f8_1x7_C.npy +ifft2_f8_1x7 ifft2 7.6e-15 fourier_f8_1x7_A.npy ifft2_f8_1x7_C.npy +fft2_f8_1x16 fft2 1.8e-13 fourier_f8_1x16_A.npy fft2_f8_1x16_C.npy +ifft2_f8_1x16 ifft2 1.1e-14 fourier_f8_1x16_A.npy ifft2_f8_1x16_C.npy +fft2_f8_3x5 fft2 1.5e-13 fourier_f8_3x5_A.npy fft2_f8_3x5_C.npy +ifft2_f8_3x5 ifft2 1e-14 fourier_f8_3x5_A.npy ifft2_f8_3x5_C.npy +fft2_f8_4x6 fft2 2.9e-13 fourier_f8_4x6_A.npy fft2_f8_4x6_C.npy +ifft2_f8_4x6 ifft2 1.2e-14 fourier_f8_4x6_A.npy ifft2_f8_4x6_C.npy +fft2_f8_7x11 fft2 1.1e-12 fourier_f8_7x11_A.npy fft2_f8_7x11_C.npy +ifft2_f8_7x11 ifft2 1.4e-14 fourier_f8_7x11_A.npy ifft2_f8_7x11_C.npy +fft2_f8_13x1 fft2 1.5e-13 fourier_f8_13x1_A.npy fft2_f8_13x1_C.npy +ifft2_f8_13x1 ifft2 1.2e-14 fourier_f8_13x1_A.npy ifft2_f8_13x1_C.npy +fft2_f8_31x17 fft2 1.1e-11 fourier_f8_31x17_A.npy fft2_f8_31x17_C.npy +ifft2_f8_31x17 ifft2 2e-14 fourier_f8_31x17_A.npy ifft2_f8_31x17_C.npy +fft2_f8_8x8 fft2 9.4e-13 fourier_f8_8x8_A.npy fft2_f8_8x8_C.npy +ifft2_f8_8x8 ifft2 1.5e-14 fourier_f8_8x8_A.npy ifft2_f8_8x8_C.npy +fft2_f8_16x32 fft2 9.5e-12 fourier_f8_16x32_A.npy fft2_f8_16x32_C.npy +ifft2_f8_16x32 ifft2 1.9e-14 fourier_f8_16x32_A.npy ifft2_f8_16x32_C.npy +fft2_f8_64x64 fft2 1e-10 fourier_f8_64x64_A.npy fft2_f8_64x64_C.npy +ifft2_f8_64x64 ifft2 2.4e-14 fourier_f8_64x64_A.npy ifft2_f8_64x64_C.npy +fft2_c16_1x1 fft2 9.7e-16 fourier_c16_1x1_A.npy fft2_c16_1x1_C.npy +ifft2_c16_1x1 ifft2 9.7e-16 fourier_c16_1x1_A.npy ifft2_c16_1x1_C.npy +fft2_c16_1x7 fft2 7e-14 fourier_c16_1x7_A.npy fft2_c16_1x7_C.npy +ifft2_c16_1x7 ifft2 1e-14 fourier_c16_1x7_A.npy ifft2_c16_1x7_C.npy +fft2_c16_3x5 fft2 1.9e-13 fourier_c16_3x5_A.npy fft2_c16_3x5_C.npy +ifft2_c16_3x5 ifft2 1.3e-14 fourier_c16_3x5_A.npy ifft2_c16_3x5_C.npy +fft2_c16_4x6 fft2 3.8e-13 fourier_c16_4x6_A.npy fft2_c16_4x6_C.npy +ifft2_c16_4x6 ifft2 1.6e-14 fourier_c16_4x6_A.npy ifft2_c16_4x6_C.npy +fft2_c16_7x11 fft2 1.9e-12 fourier_c16_7x11_A.npy fft2_c16_7x11_C.npy +ifft2_c16_7x11 ifft2 2.5e-14 fourier_c16_7x11_A.npy ifft2_c16_7x11_C.npy +fft2_c16_8x8 fft2 1.2e-12 fourier_c16_8x8_A.npy fft2_c16_8x8_C.npy +ifft2_c16_8x8 ifft2 1.9e-14 fourier_c16_8x8_A.npy ifft2_c16_8x8_C.npy +fft2_c16_31x17 fft2 1.6e-11 fourier_c16_31x17_A.npy fft2_c16_31x17_C.npy +ifft2_c16_31x17 ifft2 3.1e-14 fourier_c16_31x17_A.npy ifft2_c16_31x17_C.npy +fft2_f4_1x2 fft2 3.5e-06 fourier_f4_1x2_A.npy fft2_f4_1x2_C.npy +ifft2_f4_1x2 ifft2 1.8e-06 fourier_f4_1x2_A.npy ifft2_f4_1x2_C.npy +fft2_f4_1x16 fft2 7.6e-05 fourier_f4_1x16_A.npy fft2_f4_1x16_C.npy +ifft2_f4_1x16 ifft2 4.8e-06 fourier_f4_1x16_A.npy ifft2_f4_1x16_C.npy +fft2_f4_3x5 fft2 7.7e-05 fourier_f4_3x5_A.npy fft2_f4_3x5_C.npy +ifft2_f4_3x5 ifft2 5.1e-06 fourier_f4_3x5_A.npy ifft2_f4_3x5_C.npy +fft2_f4_7x11 fft2 0.00065 fourier_f4_7x11_A.npy fft2_f4_7x11_C.npy +ifft2_f4_7x11 ifft2 8.4e-06 fourier_f4_7x11_A.npy ifft2_f4_7x11_C.npy +fft2_f4_13x1 fft2 5e-05 fourier_f4_13x1_A.npy fft2_f4_13x1_C.npy +ifft2_f4_13x1 ifft2 3.9e-06 fourier_f4_13x1_A.npy ifft2_f4_13x1_C.npy +fft2_f4_16x32 fft2 0.0053 fourier_f4_16x32_A.npy fft2_f4_16x32_C.npy +ifft2_f4_16x32 ifft2 1e-05 fourier_f4_16x32_A.npy ifft2_f4_16x32_C.npy +fft2_c8_3x5 fft2 0.00011 fourier_c8_3x5_A.npy fft2_c8_3x5_C.npy +ifft2_c8_3x5 ifft2 7.2e-06 fourier_c8_3x5_A.npy ifft2_c8_3x5_C.npy +fft2_c8_8x8 fft2 0.00065 fourier_c8_8x8_A.npy fft2_c8_8x8_C.npy +ifft2_c8_8x8 ifft2 1e-05 fourier_c8_8x8_A.npy ifft2_c8_8x8_C.npy +fft2_i4_1x7 fft2 5.4e-13 fourier_i4_1x7_A.npy fft2_i4_1x7_C.npy +ifft2_i4_1x7 ifft2 7.7e-14 fourier_i4_1x7_A.npy ifft2_i4_1x7_C.npy +fft2_i4_4x6 fft2 2.4e-12 fourier_i4_4x6_A.npy fft2_i4_4x6_C.npy +ifft2_i4_4x6 ifft2 1e-13 fourier_i4_4x6_A.npy ifft2_i4_4x6_C.npy +fft2_i4_7x11 fft2 1e-11 fourier_i4_7x11_A.npy fft2_i4_7x11_C.npy +ifft2_i4_7x11 ifft2 1.4e-13 fourier_i4_7x11_A.npy ifft2_i4_7x11_C.npy +fft2_i4_8x8 fft2 7.7e-12 fourier_i4_8x8_A.npy fft2_i4_8x8_C.npy +ifft2_i4_8x8 ifft2 1.2e-13 fourier_i4_8x8_A.npy ifft2_i4_8x8_C.npy +fftshift_f8_1x1 fftshift 0.0 shift_f8_1x1_A.npy fftshift_f8_1x1_C.npy +ifftshift_f8_1x1 ifftshift 0.0 shift_f8_1x1_A.npy ifftshift_f8_1x1_C.npy +fftshift_f8_1x4 fftshift 0.0 shift_f8_1x4_A.npy fftshift_f8_1x4_C.npy +ifftshift_f8_1x4 ifftshift 0.0 shift_f8_1x4_A.npy ifftshift_f8_1x4_C.npy +fftshift_f8_3x5 fftshift 0.0 shift_f8_3x5_A.npy fftshift_f8_3x5_C.npy +ifftshift_f8_3x5 ifftshift 0.0 shift_f8_3x5_A.npy ifftshift_f8_3x5_C.npy +fftshift_f8_4x6 fftshift 0.0 shift_f8_4x6_A.npy fftshift_f8_4x6_C.npy +ifftshift_f8_4x6 ifftshift 0.0 shift_f8_4x6_A.npy ifftshift_f8_4x6_C.npy +fftshift_f8_5x5 fftshift 0.0 shift_f8_5x5_A.npy fftshift_f8_5x5_C.npy +ifftshift_f8_5x5 ifftshift 0.0 shift_f8_5x5_A.npy ifftshift_f8_5x5_C.npy +fftshift_f8_6x1 fftshift 0.0 shift_f8_6x1_A.npy fftshift_f8_6x1_C.npy +ifftshift_f8_6x1 ifftshift 0.0 shift_f8_6x1_A.npy ifftshift_f8_6x1_C.npy +fftshift_i4_1x1 fftshift 0.0 shift_i4_1x1_A.npy fftshift_i4_1x1_C.npy +ifftshift_i4_1x1 ifftshift 0.0 shift_i4_1x1_A.npy ifftshift_i4_1x1_C.npy +fftshift_i4_1x4 fftshift 0.0 shift_i4_1x4_A.npy fftshift_i4_1x4_C.npy +ifftshift_i4_1x4 ifftshift 0.0 shift_i4_1x4_A.npy ifftshift_i4_1x4_C.npy +fftshift_i4_3x5 fftshift 0.0 shift_i4_3x5_A.npy fftshift_i4_3x5_C.npy +ifftshift_i4_3x5 ifftshift 0.0 shift_i4_3x5_A.npy ifftshift_i4_3x5_C.npy +fftshift_i4_4x6 fftshift 0.0 shift_i4_4x6_A.npy fftshift_i4_4x6_C.npy +ifftshift_i4_4x6 ifftshift 0.0 shift_i4_4x6_A.npy ifftshift_i4_4x6_C.npy +fftshift_i4_5x5 fftshift 0.0 shift_i4_5x5_A.npy fftshift_i4_5x5_C.npy +ifftshift_i4_5x5 ifftshift 0.0 shift_i4_5x5_A.npy ifftshift_i4_5x5_C.npy +fftshift_i4_6x1 fftshift 0.0 shift_i4_6x1_A.npy fftshift_i4_6x1_C.npy +ifftshift_i4_6x1 ifftshift 0.0 shift_i4_6x1_A.npy ifftshift_i4_6x1_C.npy +fftshift_c16_1x1 fftshift 0.0 shift_c16_1x1_A.npy fftshift_c16_1x1_C.npy +ifftshift_c16_1x1 ifftshift 0.0 shift_c16_1x1_A.npy ifftshift_c16_1x1_C.npy +fftshift_c16_1x4 fftshift 0.0 shift_c16_1x4_A.npy fftshift_c16_1x4_C.npy +ifftshift_c16_1x4 ifftshift 0.0 shift_c16_1x4_A.npy ifftshift_c16_1x4_C.npy +fftshift_c16_3x5 fftshift 0.0 shift_c16_3x5_A.npy fftshift_c16_3x5_C.npy +ifftshift_c16_3x5 ifftshift 0.0 shift_c16_3x5_A.npy ifftshift_c16_3x5_C.npy +fftshift_c16_4x6 fftshift 0.0 shift_c16_4x6_A.npy fftshift_c16_4x6_C.npy +ifftshift_c16_4x6 ifftshift 0.0 shift_c16_4x6_A.npy ifftshift_c16_4x6_C.npy +fftshift_c16_5x5 fftshift 0.0 shift_c16_5x5_A.npy fftshift_c16_5x5_C.npy +ifftshift_c16_5x5 ifftshift 0.0 shift_c16_5x5_A.npy ifftshift_c16_5x5_C.npy +fftshift_c16_6x1 fftshift 0.0 shift_c16_6x1_A.npy fftshift_c16_6x1_C.npy +ifftshift_c16_6x1 ifftshift 0.0 shift_c16_6x1_A.npy ifftshift_c16_6x1_C.npy diff --git a/tests/fixtures/s8/shift_c16_1x1_A.npy b/tests/fixtures/s8/shift_c16_1x1_A.npy new file mode 100644 index 0000000..f72c79c Binary files /dev/null and b/tests/fixtures/s8/shift_c16_1x1_A.npy differ diff --git a/tests/fixtures/s8/shift_c16_1x4_A.npy b/tests/fixtures/s8/shift_c16_1x4_A.npy new file mode 100644 index 0000000..e8310a6 Binary files /dev/null and b/tests/fixtures/s8/shift_c16_1x4_A.npy differ diff --git a/tests/fixtures/s8/shift_c16_3x5_A.npy b/tests/fixtures/s8/shift_c16_3x5_A.npy new file mode 100644 index 0000000..c9a7ff6 Binary files /dev/null and b/tests/fixtures/s8/shift_c16_3x5_A.npy differ diff --git a/tests/fixtures/s8/shift_c16_4x6_A.npy b/tests/fixtures/s8/shift_c16_4x6_A.npy new file mode 100644 index 0000000..5ced2fe Binary files /dev/null and b/tests/fixtures/s8/shift_c16_4x6_A.npy differ diff --git a/tests/fixtures/s8/shift_c16_5x5_A.npy b/tests/fixtures/s8/shift_c16_5x5_A.npy new file mode 100644 index 0000000..960d67c Binary files /dev/null and b/tests/fixtures/s8/shift_c16_5x5_A.npy differ diff --git a/tests/fixtures/s8/shift_c16_6x1_A.npy b/tests/fixtures/s8/shift_c16_6x1_A.npy new file mode 100644 index 0000000..b30978f Binary files /dev/null and b/tests/fixtures/s8/shift_c16_6x1_A.npy differ diff --git a/tests/fixtures/s8/shift_f8_1x1_A.npy b/tests/fixtures/s8/shift_f8_1x1_A.npy new file mode 100644 index 0000000..24e92d0 Binary files /dev/null and b/tests/fixtures/s8/shift_f8_1x1_A.npy differ diff --git a/tests/fixtures/s8/shift_f8_1x4_A.npy b/tests/fixtures/s8/shift_f8_1x4_A.npy new file mode 100644 index 0000000..e08219a Binary files /dev/null and b/tests/fixtures/s8/shift_f8_1x4_A.npy differ diff --git a/tests/fixtures/s8/shift_f8_3x5_A.npy b/tests/fixtures/s8/shift_f8_3x5_A.npy new file mode 100644 index 0000000..1b10a0d Binary files /dev/null and b/tests/fixtures/s8/shift_f8_3x5_A.npy differ diff --git a/tests/fixtures/s8/shift_f8_4x6_A.npy b/tests/fixtures/s8/shift_f8_4x6_A.npy new file mode 100644 index 0000000..dcaa08a Binary files /dev/null and b/tests/fixtures/s8/shift_f8_4x6_A.npy differ diff --git a/tests/fixtures/s8/shift_f8_5x5_A.npy b/tests/fixtures/s8/shift_f8_5x5_A.npy new file mode 100644 index 0000000..1e4b191 Binary files /dev/null and b/tests/fixtures/s8/shift_f8_5x5_A.npy differ diff --git a/tests/fixtures/s8/shift_f8_6x1_A.npy b/tests/fixtures/s8/shift_f8_6x1_A.npy new file mode 100644 index 0000000..1fb5de7 Binary files /dev/null and b/tests/fixtures/s8/shift_f8_6x1_A.npy differ diff --git a/tests/fixtures/s8/shift_i4_1x1_A.npy b/tests/fixtures/s8/shift_i4_1x1_A.npy new file mode 100644 index 0000000..2a7deea Binary files /dev/null and b/tests/fixtures/s8/shift_i4_1x1_A.npy differ diff --git a/tests/fixtures/s8/shift_i4_1x4_A.npy b/tests/fixtures/s8/shift_i4_1x4_A.npy new file mode 100644 index 0000000..bf7f96d Binary files /dev/null and b/tests/fixtures/s8/shift_i4_1x4_A.npy differ diff --git a/tests/fixtures/s8/shift_i4_3x5_A.npy b/tests/fixtures/s8/shift_i4_3x5_A.npy new file mode 100644 index 0000000..dbff31f Binary files /dev/null and b/tests/fixtures/s8/shift_i4_3x5_A.npy differ diff --git a/tests/fixtures/s8/shift_i4_4x6_A.npy b/tests/fixtures/s8/shift_i4_4x6_A.npy new file mode 100644 index 0000000..db3f002 Binary files /dev/null and b/tests/fixtures/s8/shift_i4_4x6_A.npy differ diff --git a/tests/fixtures/s8/shift_i4_5x5_A.npy b/tests/fixtures/s8/shift_i4_5x5_A.npy new file mode 100644 index 0000000..4cfa37f Binary files /dev/null and b/tests/fixtures/s8/shift_i4_5x5_A.npy differ diff --git a/tests/fixtures/s8/shift_i4_6x1_A.npy b/tests/fixtures/s8/shift_i4_6x1_A.npy new file mode 100644 index 0000000..ca4dfc5 Binary files /dev/null and b/tests/fixtures/s8/shift_i4_6x1_A.npy differ diff --git a/tests/fuzz/corpus/binary/double_2x3.bin b/tests/fuzz/corpus/binary/double_2x3.bin new file mode 100644 index 0000000..fecdae5 Binary files /dev/null and b/tests/fuzz/corpus/binary/double_2x3.bin differ diff --git a/tests/fuzz/corpus/binary/double_2x3_short.bin b/tests/fuzz/corpus/binary/double_2x3_short.bin new file mode 100644 index 0000000..18a429f Binary files /dev/null and b/tests/fuzz/corpus/binary/double_2x3_short.bin differ diff --git a/tests/fuzz/corpus/binary/empty_0x3.bin b/tests/fuzz/corpus/binary/empty_0x3.bin new file mode 100644 index 0000000..5947748 Binary files /dev/null and b/tests/fuzz/corpus/binary/empty_0x3.bin differ diff --git a/tests/fuzz/corpus/binary/int32_2x2.bin b/tests/fuzz/corpus/binary/int32_2x2.bin new file mode 100644 index 0000000..8bf7e5a Binary files /dev/null and b/tests/fuzz/corpus/binary/int32_2x2.bin differ diff --git a/tests/fuzz/corpus/bmp/rgb24_bottom_up_3x2.bmp b/tests/fuzz/corpus/bmp/rgb24_bottom_up_3x2.bmp new file mode 100644 index 0000000..52cdd1a Binary files /dev/null and b/tests/fuzz/corpus/bmp/rgb24_bottom_up_3x2.bmp differ diff --git a/tests/fuzz/corpus/bmp/rgb24_top_down_3x2.bmp b/tests/fuzz/corpus/bmp/rgb24_top_down_3x2.bmp new file mode 100644 index 0000000..1bdea6b Binary files /dev/null and b/tests/fuzz/corpus/bmp/rgb24_top_down_3x2.bmp differ diff --git a/tests/fuzz/corpus/bmp/rgb24_trailing_1x1.bmp b/tests/fuzz/corpus/bmp/rgb24_trailing_1x1.bmp new file mode 100644 index 0000000..ad5be62 Binary files /dev/null and b/tests/fuzz/corpus/bmp/rgb24_trailing_1x1.bmp differ diff --git a/tests/fuzz/corpus/bmp/rgb24_truncated.bmp b/tests/fuzz/corpus/bmp/rgb24_truncated.bmp new file mode 100644 index 0000000..7823931 Binary files /dev/null and b/tests/fuzz/corpus/bmp/rgb24_truncated.bmp differ diff --git a/tests/fuzz/corpus/bmp/rgb32_2x2.bmp b/tests/fuzz/corpus/bmp/rgb32_2x2.bmp new file mode 100644 index 0000000..6747cf2 Binary files /dev/null and b/tests/fuzz/corpus/bmp/rgb32_2x2.bmp differ diff --git a/tests/fuzz/corpus/npy/c16_le.npy b/tests/fuzz/corpus/npy/c16_le.npy new file mode 100644 index 0000000..472b934 Binary files /dev/null and b/tests/fuzz/corpus/npy/c16_le.npy differ diff --git a/tests/fuzz/corpus/npy/f8_be.npy b/tests/fuzz/corpus/npy/f8_be.npy new file mode 100644 index 0000000..8e2a382 Binary files /dev/null and b/tests/fuzz/corpus/npy/f8_be.npy differ diff --git a/tests/fuzz/corpus/npy/f8_le.npy b/tests/fuzz/corpus/npy/f8_le.npy new file mode 100644 index 0000000..568e6d4 Binary files /dev/null and b/tests/fuzz/corpus/npy/f8_le.npy differ diff --git a/tests/fuzz/corpus/npy/f8_le_truncated.npy b/tests/fuzz/corpus/npy/f8_le_truncated.npy new file mode 100644 index 0000000..6696718 Binary files /dev/null and b/tests/fuzz/corpus/npy/f8_le_truncated.npy differ diff --git a/tests/fuzz/corpus/npy/fortran_2x3.npy b/tests/fuzz/corpus/npy/fortran_2x3.npy new file mode 100644 index 0000000..099bf2f Binary files /dev/null and b/tests/fuzz/corpus/npy/fortran_2x3.npy differ diff --git a/tests/fuzz/corpus/npy/i4_be.npy b/tests/fuzz/corpus/npy/i4_be.npy new file mode 100644 index 0000000..9db497e Binary files /dev/null and b/tests/fuzz/corpus/npy/i4_be.npy differ diff --git a/tests/fuzz/corpus/npy/one_d.npy b/tests/fuzz/corpus/npy/one_d.npy new file mode 100644 index 0000000..aa15719 Binary files /dev/null and b/tests/fuzz/corpus/npy/one_d.npy differ diff --git a/tests/fuzz/corpus/npy/structured.npy b/tests/fuzz/corpus/npy/structured.npy new file mode 100644 index 0000000..270fbd8 Binary files /dev/null and b/tests/fuzz/corpus/npy/structured.npy differ diff --git a/tests/fuzz/corpus/npy/u1.npy b/tests/fuzz/corpus/npy/u1.npy new file mode 100644 index 0000000..8f31e56 Binary files /dev/null and b/tests/fuzz/corpus/npy/u1.npy differ diff --git a/tests/fuzz/corpus/text/complex_2x2.txt b/tests/fuzz/corpus/text/complex_2x2.txt new file mode 100644 index 0000000..c5692ba --- /dev/null +++ b/tests/fuzz/corpus/text/complex_2x2.txt @@ -0,0 +1,2 @@ +(1,2) (3,-4) +(0.5,0) 7 diff --git a/tests/fuzz/corpus/text/float_exp.txt b/tests/fuzz/corpus/text/float_exp.txt new file mode 100644 index 0000000..3708f89 --- /dev/null +++ b/tests/fuzz/corpus/text/float_exp.txt @@ -0,0 +1,2 @@ +1.5e3 -2.25e-2 ++3 0x10 diff --git a/tests/fuzz/corpus/text/int_tabs_crlf.txt b/tests/fuzz/corpus/text/int_tabs_crlf.txt new file mode 100644 index 0000000..78ed961 --- /dev/null +++ b/tests/fuzz/corpus/text/int_tabs_crlf.txt @@ -0,0 +1,2 @@ +1 -2 +3 4 diff --git a/tests/fuzz/corpus/text/ragged.txt b/tests/fuzz/corpus/text/ragged.txt new file mode 100644 index 0000000..0177db2 --- /dev/null +++ b/tests/fuzz/corpus/text/ragged.txt @@ -0,0 +1,2 @@ +1 2 +3 diff --git a/tests/fuzz/corpus/text/real_2x3.txt b/tests/fuzz/corpus/text/real_2x3.txt new file mode 100644 index 0000000..b36d133 --- /dev/null +++ b/tests/fuzz/corpus/text/real_2x3.txt @@ -0,0 +1,2 @@ +1 2 3 +4 5 6 diff --git a/tests/fuzz/corpus/text/unclosed.txt b/tests/fuzz/corpus/text/unclosed.txt new file mode 100644 index 0000000..efd6400 --- /dev/null +++ b/tests/fuzz/corpus/text/unclosed.txt @@ -0,0 +1 @@ +(1,2 3 diff --git a/tests/fuzz/fuzz_binary.cc b/tests/fuzz/fuzz_binary.cc new file mode 100644 index 0000000..8ce98d4 --- /dev/null +++ b/tests/fuzz/fuzz_binary.cc @@ -0,0 +1,10 @@ +// S5-R5 (PR-7): libFuzzer harness for matrix_details::parse_binary; traps when a failed parse changed a pre-filled +// destination or a successful one is inconsistent (checks in fuzz_checks.hpp). No file I/O. +#include "../../matrix.hpp" +#include "fuzz_checks.hpp" + +extern "C" int LLVMFuzzerTestOneInput( std::uint8_t const* data, std::size_t size ) +{ + if ( fuzz_checks::check_binary( data, size ) != nullptr ) __builtin_trap(); + return 0; +} diff --git a/tests/fuzz/fuzz_bmp.cc b/tests/fuzz/fuzz_bmp.cc new file mode 100644 index 0000000..694a30d --- /dev/null +++ b/tests/fuzz/fuzz_bmp.cc @@ -0,0 +1,10 @@ +// S5-R5 (PR-7): libFuzzer harness for matrix_details::parse_bmp; traps when a failed parse changed a pre-filled +// destination or a successful one is inconsistent (checks in fuzz_checks.hpp). No file I/O. +#include "../../matrix.hpp" +#include "fuzz_checks.hpp" + +extern "C" int LLVMFuzzerTestOneInput( std::uint8_t const* data, std::size_t size ) +{ + if ( fuzz_checks::check_bmp( data, size ) != nullptr ) __builtin_trap(); + return 0; +} diff --git a/tests/fuzz/fuzz_checks.hpp b/tests/fuzz/fuzz_checks.hpp new file mode 100644 index 0000000..00f9728 --- /dev/null +++ b/tests/fuzz/fuzz_checks.hpp @@ -0,0 +1,136 @@ +// S5-R5 (PR-7): the checks shared by the libFuzzer harnesses (which trap on a failure) and the +// `[S5][S5-R5]` replay test (which REQUIREs them). Each check parses a byte buffer with the matrix_details +// parsers into pre-filled destinations of several element types and returns nullptr when every parse was +// transactional and consistent, otherwise a static message. No file I/O. +#ifndef FENG_TESTS_FUZZ_CHECKS_HPP_INCLUDED +#define FENG_TESTS_FUZZ_CHECKS_HPP_INCLUDED + +#include +#include +#include +#include +#include + +namespace fuzz_checks +{ + enum class format { npy, text, binary }; + + template < typename T > + struct is_complex : std::false_type {}; + template < typename F > + struct is_complex< std::complex< F > > : std::true_type {}; + + template < typename T > + T fill_value( std::size_t i ) noexcept + { + if constexpr ( is_complex< T >::value ) + { + using F = typename T::value_type; + return T{ static_cast< F >( i ) + F( 0.5 ), -static_cast< F >( i ) }; + } + else + return static_cast< T >( i * 7 + 3 ); + } + + template < typename T > + feng::matrix< T > prefilled( std::size_t r, std::size_t c ) + { + feng::matrix< T > m( r, c ); + for ( std::size_t i = 0; i != m.size(); ++i ) m.data()[i] = fill_value< T >( i ); + return m; + } + + template < typename T > + bool consistent( feng::matrix< T > const& m ) noexcept + { + if ( m.size() != m.row() * m.col() ) return false; + if ( m.size() != 0 && m.data() == nullptr ) return false; + return true; + } + + // Compares element values, not bytes: long double (and std::complex) carry padding bytes whose + // contents are unspecified after an element assignment, so a byte comparison would be unreliable. The + // pre-filled values are finite, so operator== is exact. + template < typename T > + bool unchanged( feng::matrix< T > const& m, std::size_t r, std::size_t c, std::vector< T > const& before ) noexcept + { + if ( m.row() != r || m.col() != c || m.size() != r * c || m.size() != before.size() ) return false; + if ( before.empty() ) return true; + if ( m.data() == nullptr ) return false; + for ( std::size_t i = 0; i != before.size(); ++i ) + if ( !( m.data()[i] == before[i] ) ) return false; + return true; + } + + template < format F, typename T > + char const* check_shape( std::uint8_t const* data, std::size_t size, std::size_t r, std::size_t c ) + { + feng::matrix< T > m = prefilled< T >( r, c ); + std::vector< T > const before( m.data(), m.data() + m.size() ); + bool ok = false; + if constexpr ( F == format::npy ) + ok = feng::matrix_details::parse_npy< T >( data, size, m ); + else if constexpr ( F == format::text ) + ok = feng::matrix_details::parse_text< T >( reinterpret_cast< char const* >( data ), size, m ); + else + ok = feng::matrix_details::parse_binary< T >( data, size, m ); + if ( ok ) + return consistent( m ) ? nullptr : "a successful parse left an inconsistent matrix (size != rows*cols or null data)"; + return unchanged( m, r, c, before ) ? nullptr : "a failed parse changed the destination"; + } + + template < format F, typename T > + char const* check_type( std::uint8_t const* data, std::size_t size ) + { + if ( char const* e = check_shape< F, T >( data, size, 2, 3 ) ) return e; + return check_shape< F, T >( data, size, 0, 0 ); + } + + template < format F > + char const* check_all( std::uint8_t const* data, std::size_t size ) + { + char const* results[] = { + check_type< F, double >( data, size ), + check_type< F, float >( data, size ), + check_type< F, std::uint8_t >( data, size ), + check_type< F, std::int8_t >( data, size ), + check_type< F, std::int16_t >( data, size ), + check_type< F, std::uint16_t >( data, size ), + check_type< F, std::int32_t >( data, size ), + check_type< F, std::uint32_t >( data, size ), + check_type< F, std::int64_t >( data, size ), + check_type< F, std::uint64_t >( data, size ), + check_type< F, std::complex< float > >( data, size ), + check_type< F, std::complex< double > >( data, size ), + }; + for ( char const* e : results ) + if ( e ) return e; + // The native binary format also supports long double and std::complex (is_binary_element_v); + // the npy and text parsers do not take them. + if constexpr ( F == format::binary ) + { + if ( char const* e = check_type< F, long double >( data, size ) ) return e; + if ( char const* e = check_type< F, std::complex< long double > >( data, size ) ) return e; + } + return nullptr; + } + + inline char const* check_npy( std::uint8_t const* data, std::size_t size ) { return check_all< format::npy >( data, size ); } + inline char const* check_text( std::uint8_t const* data, std::size_t size ) { return check_all< format::text >( data, size ); } + inline char const* check_binary( std::uint8_t const* data, std::size_t size ) { return check_all< format::binary >( data, size ); } + + // BMP has one result type: three equally shaped, non-empty, consistent channels on success. + inline char const* check_bmp( std::uint8_t const* data, std::size_t size ) + { + auto const ans = feng::matrix_details::parse_bmp( data, size ); + if ( !ans ) return nullptr; + auto const& [r, g, b] = *ans; + if ( !consistent( r ) || !consistent( g ) || !consistent( b ) ) return "a successful BMP parse left an inconsistent channel"; + if ( r.row() != g.row() || r.row() != b.row() || r.col() != g.col() || r.col() != b.col() ) + return "a successful BMP parse returned channels of different shapes"; + if ( r.size() == 0 ) return "a successful BMP parse returned empty channels"; + return nullptr; + } +} + +#endif diff --git a/tests/fuzz/fuzz_npy.cc b/tests/fuzz/fuzz_npy.cc new file mode 100644 index 0000000..b811b45 --- /dev/null +++ b/tests/fuzz/fuzz_npy.cc @@ -0,0 +1,10 @@ +// S5-R5 (PR-7): libFuzzer harness for matrix_details::parse_npy; traps when a failed parse changed a pre-filled +// destination or a successful one is inconsistent (checks in fuzz_checks.hpp). No file I/O. +#include "../../matrix.hpp" +#include "fuzz_checks.hpp" + +extern "C" int LLVMFuzzerTestOneInput( std::uint8_t const* data, std::size_t size ) +{ + if ( fuzz_checks::check_npy( data, size ) != nullptr ) __builtin_trap(); + return 0; +} diff --git a/tests/fuzz/fuzz_text.cc b/tests/fuzz/fuzz_text.cc new file mode 100644 index 0000000..1823eef --- /dev/null +++ b/tests/fuzz/fuzz_text.cc @@ -0,0 +1,10 @@ +// S5-R5 (PR-7): libFuzzer harness for matrix_details::parse_text; traps when a failed parse changed a pre-filled +// destination or a successful one is inconsistent (checks in fuzz_checks.hpp). No file I/O. +#include "../../matrix.hpp" +#include "fuzz_checks.hpp" + +extern "C" int LLVMFuzzerTestOneInput( std::uint8_t const* data, std::size_t size ) +{ + if ( fuzz_checks::check_text( data, size ) != nullptr ) __builtin_trap(); + return 0; +} diff --git a/tests/fuzz/make_corpus.py b/tests/fuzz/make_corpus.py new file mode 100644 index 0000000..5de93e9 --- /dev/null +++ b/tests/fuzz/make_corpus.py @@ -0,0 +1,62 @@ +#!/usr/bin/env python3 +# S5-R5: writes the seed corpora under tests/fuzz/corpus// (run from the repo root). NPY seeds are copies of +# tests/fixtures/s5/*.npy plus a truncated one; the others are built here. Valid and near-valid inputs, all small. +import os +import shutil +import struct + +ROOT = os.path.join("tests", "fuzz", "corpus") + + +def put(target, name, data): + d = os.path.join(ROOT, target) + os.makedirs(d, exist_ok=True) + with open(os.path.join(d, name), "wb") as f: + f.write(data) + + +# NPY +for fx in ["f8_le", "f8_be", "c16_le", "i4_be", "fortran_2x3", "one_d", "u1", "structured"]: + os.makedirs(os.path.join(ROOT, "npy"), exist_ok=True) + shutil.copy(os.path.join("tests", "fixtures", "s5", fx + ".npy"), os.path.join(ROOT, "npy", fx + ".npy")) +with open(os.path.join("tests", "fixtures", "s5", "f8_le.npy"), "rb") as f: + put("npy", "f8_le_truncated.npy", f.read()[:-5]) + + +# BMP: 14-byte file header + 40-byte info header, BI_RGB. +def bmp(width, height, bpp, trailing=b""): + rows = abs(height) + stride = (width * bpp + 31) // 32 * 4 + pixels = bytearray() + for r in range(rows): + row = bytearray() + for c in range(width): + px = bytes([(r * 40 + c) & 255, (c * 60) & 255, (r * 90 + 7) & 255]) + row += px + (b"\xff" if bpp == 32 else b"") + row += b"\0" * (stride - len(row)) + pixels += row + offset = 54 + header = b"BM" + struct.pack("): det (PartialPivLU), solve, inverse, singular values (JacobiSVD), pinv (JacobiSVD with +// feng's cutoff max(m, n)·ε·s_1) and Cholesky (LLT). Built and run by `tools/check.sh oracle S7` +// (-std=c++20 -O2 -isystem /usr/include/eigen3). Prints the Eigen version, one line per failing comparison and +// `EIGEN PASS comparisons`; exit status 0 only when every comparison passes. +#include "../../matrix.hpp" + +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace +{ + template< typename T > struct real_of { using type = T; }; + template< typename T > struct real_of< std::complex< T > > { using type = T; }; + template< typename T > using real_t = typename real_of< T >::type; + template< typename T > using emat = Eigen::Matrix< T, Eigen::Dynamic, Eigen::Dynamic >; + + double const eps = std::numeric_limits< double >::epsilon(); + std::size_t passed = 0, failed = 0; + + void check( bool ok, std::string const& what, double err, double tol ) + { + if ( std::getenv( "S7_RATIOS" ) ) std::cout << what << " ratio " << err / tol << "\n"; + if ( ok ) { ++passed; return; } + ++failed; + std::cout << "EIGEN FAIL " << what << ": error " << err << " > tol " << tol << "\n"; + } + + template< typename T > + T conj_( T const& x ) + { + if constexpr ( std::is_same_v< T, double > ) return x; + else return std::conj( x ); + } + + template< typename T > + T draw( std::mt19937_64& g ) + { + std::uniform_real_distribution< double > u( -1.0, 1.0 ); + if constexpr ( std::is_same_v< T, double > ) return u( g ); + else { double const re = u( g ); return T( re, u( g ) ); } + } + + template< typename T > + feng::matrix< T > random( std::size_t m, std::size_t n, std::mt19937_64& g ) + { + feng::matrix< T > a{ m, n }; + for ( std::size_t r = 0; r != m; ++r ) + for ( std::size_t c = 0; c != n; ++c ) a[r][c] = draw< T >( g ); + return a; + } + + template< typename T > + emat< T > to_eigen( feng::matrix< T > const& a ) + { + emat< T > e( a.row(), a.col() ); + for ( std::size_t r = 0; r != a.row(); ++r ) + for ( std::size_t c = 0; c != a.col(); ++c ) e( r, c ) = a[r][c]; + return e; + } + + // ||f - e||_F / ||e||_F; +inf on a shape mismatch + template< typename T > + double rel( feng::matrix< T > const& f, emat< T > const& e ) + { + if ( f.row() != std::size_t( e.rows() ) || f.col() != std::size_t( e.cols() ) ) return HUGE_VAL; + double d = 0; + for ( std::size_t r = 0; r != f.row(); ++r ) + for ( std::size_t c = 0; c != f.col(); ++c ) d += std::norm( f[r][c] - e( r, c ) ); + return std::sqrt( d ) / e.norm(); + } + + template< typename T > + emat< T > eigen_pinv( emat< T > const& a ) + { + Eigen::JacobiSVD< emat< T >, Eigen::ComputeThinU | Eigen::ComputeThinV > svd( a ); + auto const& s = svd.singularValues(); + double const cut = double( std::max( a.rows(), a.cols() ) ) * eps * ( s.size() ? s( 0 ) : 0.0 ); + Eigen::Matrix< T, Eigen::Dynamic, 1 > inv_s( s.size() ); + for ( Eigen::Index i = 0; i != s.size(); ++i ) inv_s( i ) = s( i ) > cut ? T( 1.0 / s( i ) ) : T( 0 ); + return svd.matrixV() * inv_s.asDiagonal() * svd.matrixU().adjoint(); + } + + // κ over the singular values above the pinv cutoff + template< typename T > + double kappa( emat< T > const& a ) + { + Eigen::JacobiSVD< emat< T > > svd( a ); + auto const& s = svd.singularValues(); + double const cut = double( std::max( a.rows(), a.cols() ) ) * eps * s( 0 ); + double lo = s( 0 ); + for ( Eigen::Index i = 0; i != s.size(); ++i ) if ( s( i ) > cut ) lo = s( i ); + return s( 0 ) / lo; + } + + template< typename T > + void svd_and_pinv( feng::matrix< T > const& a, std::string const& tag ) + { + std::size_t const p = std::max( a.row(), a.col() ); + emat< T > const e = to_eigen( a ); + Eigen::JacobiSVD< emat< T > > esvd( e ); + auto const& es = esvd.singularValues(); + auto const f = feng::svd_factor( a ); + double err = HUGE_VAL; + if ( f.ok() && f.s().size() == std::size_t( es.size() ) ) + { + err = 0; + for ( std::size_t i = 0; i != f.s().size(); ++i ) err = std::max( err, std::abs( f.s()[i] - es( i ) ) ); + } + double tol = 32.0 * p * eps * es( 0 ); + check( err <= tol, "svdvals " + tag, err, tol ); + + double const k = kappa( e ); + feng::matrix< T > x; + err = feng::pinverse( a, x ) == feng::linalg_status::ok ? rel( x, eigen_pinv( e ) ) : HUGE_VAL; + tol = 64.0 * p * eps * k * k; + check( err <= tol, "pinv " + tag, err, tol ); + } + + template< typename T > + void square( std::size_t n, std::mt19937_64& g, std::string const& tname ) + { + std::string const tag = tname + " " + std::to_string( n ) + "x" + std::to_string( n ); + auto const a = random< T >( n, n, g ); + emat< T > const e = to_eigen( a ); + double const k = kappa( e ); + double const tol = 64.0 * n * eps * k; + + Eigen::PartialPivLU< emat< T > > lu( e ); + T const de = lu.determinant(); + T const df = feng::det( a ); + double err = std::abs( df - de ) / std::abs( de ); + check( err <= tol, "det " + tag, err, tol ); + + auto const b = random< T >( n, 3, g ); + auto const xs = feng::solve( a, b ); + err = xs.ok() ? rel( xs.value, emat< T >( lu.solve( to_eigen( b ) ) ) ) : HUGE_VAL; + check( err <= tol, "solve " + tag, err, tol ); + + auto const xi = feng::try_inverse( a ); + err = xi.ok() ? rel( xi.value, emat< T >( lu.inverse() ) ) : HUGE_VAL; + check( err <= tol, "inverse " + tag, err, tol ); + + // Hermitian positive definite: A·Aᴴ + n·I, symmetrized exactly + feng::matrix< T > h{ n, n }; + for ( std::size_t r = 0; r != n; ++r ) + for ( std::size_t c = 0; c != n; ++c ) + { + T s = r == c ? T( double( n ) ) : T( 0 ); + for ( std::size_t j = 0; j != n; ++j ) s += a[r][j] * conj_( a[c][j] ); + h[r][c] = s; + } + for ( std::size_t r = 0; r != n; ++r ) + { + h[r][r] = T( std::real( h[r][r] ) ); + for ( std::size_t c = 0; c != r; ++c ) h[r][c] = conj_( h[c][r] ); + } + emat< T > const eh = to_eigen( h ); + Eigen::LLT< emat< T > > llt( eh ); + auto const ch = feng::cholesky_factor( h ); + double const tolc = 64.0 * n * eps * kappa( eh ); + err = ( ch.ok() && llt.info() == Eigen::Success ) ? rel( ch.l(), emat< T >( llt.matrixL() ) ) : HUGE_VAL; + check( err <= tolc, "cholesky " + tag, err, tolc ); + + svd_and_pinv( a, tag ); + } + + template< typename T > + void run( std::string const& tname, std::uint64_t seed ) + { + std::mt19937_64 g{ seed }; + for ( std::size_t n : { 1, 4, 8 } ) square< T >( n, g, tname ); + svd_and_pinv( random< T >( 6, 3, g ), tname + " 6x3" ); + svd_and_pinv( random< T >( 3, 7, g ), tname + " 3x7" ); + // rank 2, 5x5: a 5x2 times a 2x5 product + auto const b = random< T >( 5, 2, g ); + auto const c = random< T >( 2, 5, g ); + svd_and_pinv( feng::matrix< T >( b * c ), tname + " 5x5 rank 2" ); + } +} + +int main() +{ + std::cout << "Eigen " << EIGEN_WORLD_VERSION << "." << EIGEN_MAJOR_VERSION << "." << EIGEN_MINOR_VERSION; +#ifdef EIGEN_VERSION_STRING + std::cout << " (" << EIGEN_VERSION_STRING << ")"; +#endif + std::cout << "\n"; + run< double >( "double", 20261001 ); + run< std::complex< double > >( "complex", 20261002 ); + if ( failed != 0 ) + { + std::cout << "EIGEN FAIL " << failed << " of " << ( passed + failed ) << " comparisons\n"; + return 1; + } + std::cout << "EIGEN PASS " << passed << " comparisons\n"; + return 0; +} diff --git a/tests/test.cc b/tests/test.cc index 8dca80d..8de2cee 100644 --- a/tests/test.cc +++ b/tests/test.cc @@ -1,7 +1,7 @@ #include "../matrix.hpp" #define CATCH_CONFIG_MAIN -#include "./catch.hpp" +#include #include "./cases/abs.hpp" #include "./cases/acosh.hpp" @@ -61,5 +61,44 @@ #include "./cases/trunc.hpp" #include "./cases/view_bracket.hpp" #include "./cases/zeros.hpp" +#include "./cases/s1_f16.hpp" +#include "./cases/s2_r1.hpp" +#include "./cases/s2_r2.hpp" +#include "./cases/s2_r3.hpp" +#include "./cases/s2_r4.hpp" +#include "./cases/s3_r1.hpp" +#include "./cases/s3_r4.hpp" +#include "./cases/s3_r2.hpp" +#include "./cases/s3_r3.hpp" +#include "./cases/s3_r5.hpp" +#include "./cases/s4_r1.hpp" +#include "./cases/s4_r2.hpp" +#include "./cases/s4_r3.hpp" +#include "./cases/s4_r4.hpp" +#include "./cases/s4_r5.hpp" +#include "./cases/s5_r1.hpp" +#include "./cases/s5_r2.hpp" +#include "./cases/s5_r3.hpp" +#include "./cases/s5_r4.hpp" +#include "./cases/s5_r5.hpp" +#include "./cases/s6_r1.hpp" +#include "./cases/s6_r3.hpp" +#include "./cases/s6_r4.hpp" +#include "./cases/s6_r5.hpp" +#include "./cases/s7_r1.hpp" +#include "./cases/s7_r2.hpp" +#include "./cases/s7_r3.hpp" +#include "./cases/s7_r4.hpp" +#include "./cases/s7_r5.hpp" +#include "./cases/s7_r6.hpp" +#include "./cases/s8_r1.hpp" +#include "./cases/s8_r2.hpp" +#include "./cases/s8_r3.hpp" +#include "./cases/s8_fixtures.hpp" +#include "./cases/s9_r1.hpp" +#include "./cases/s9_r4.hpp" +#include "./cases/s9_r5.hpp" +#include "./cases/s9_r6.hpp" +#include "./cases/s10_r3.hpp" #include "./cases/opencv.hpp" diff --git a/tools/bench/fft.cc b/tools/bench/fft.cc new file mode 100644 index 0000000..f355c08 --- /dev/null +++ b/tools/bench/fft.cc @@ -0,0 +1,78 @@ +// tools/bench/fft.cc: the S8-R4 (D-031) complexity benchmark of feng::fft, run by `tools/check.sh bench fft`. +// Times feng::fft on fixed-seed complex inputs: 128x128, 256x256, 512x512 (radix-2) and, for information, +// 125x125, 250x250, 500x500 (Bluestein). Per size one warmup, then 15 timed samples (5 with --smoke) on steady_clock. +// Each sample is a batch of `reps` transforms of the same input, so every sample lasts at least ~20 ms and timer and +// scheduler noise stay small next to it; the warmup picks `reps` (it doubles a batch from one transform until the +// batch lasts 20 ms). The per-transform median (sample time / reps) is printed as `BENCH fft x median s`, +// with `RATIO / ` and `INFO bluestein ...` lines. +#include "matrix.hpp" + +#include +#include +#include +#include +#include +#include +#include + +namespace +{ + double sink = 0.0; // keeps the transforms observable + + double median_seconds( std::size_t n, int samples ) + { + std::mt19937_64 gen{ 20261001ULL + n }; + std::uniform_real_distribution< double > dist{ -1.0, 1.0 }; + feng::matrix< std::complex< double > > x( n, n ); + for ( auto& v : x ) v = std::complex< double >{ dist( gen ), dist( gen ) }; + + auto const run = [&]() + { + auto const X = feng::fft( x ); + sink += X[n / 2][n / 3].real(); + }; + auto const batch = [&]( long reps ) + { + auto const t0 = std::chrono::steady_clock::now(); + for ( long r = 0; r != reps; ++r ) run(); + auto const t1 = std::chrono::steady_clock::now(); + return std::chrono::duration< double >( t1 - t0 ).count(); + }; + // warmup: grows the batch until it lasts at least 20 ms; its final size is the repeat count + long reps = 1; + while ( batch( reps ) < 0.020 ) reps *= 2; + std::vector< double > t; + t.reserve( static_cast< std::size_t >( samples ) ); + for ( int s = 0; s != samples; ++s ) t.push_back( batch( reps ) / static_cast< double >( reps ) ); + std::sort( t.begin(), t.end() ); + std::size_t const m = t.size() / 2; + return t.size() % 2 ? t[m] : 0.5 * ( t[m - 1] + t[m] ); + } +} + +int main( int argc, char** argv ) +{ + int samples = 15; + for ( int i = 1; i < argc; ++i ) + if ( std::strcmp( argv[i], "--smoke" ) == 0 ) samples = 5; + + double const m128 = median_seconds( 128, samples ); + double const m256 = median_seconds( 256, samples ); + double const m512 = median_seconds( 512, samples ); + std::printf( "BENCH fft 128x128 median %.6f s\n", m128 ); + std::printf( "BENCH fft 256x256 median %.6f s\n", m256 ); + std::printf( "BENCH fft 512x512 median %.6f s\n", m512 ); + std::printf( "RATIO 256/128 %.3f\n", m256 / m128 ); + std::printf( "RATIO 512/256 %.3f\n", m512 / m256 ); + + double const b125 = median_seconds( 125, samples ); + double const b250 = median_seconds( 250, samples ); + double const b500 = median_seconds( 500, samples ); + std::printf( "INFO bluestein 125x125 median %.6f s\n", b125 ); + std::printf( "INFO bluestein 250x250 median %.6f s\n", b250 ); + std::printf( "INFO bluestein 500x500 median %.6f s\n", b500 ); + std::printf( "INFO bluestein ratio 250/125 %.3f\n", b250 / b125 ); + std::printf( "INFO bluestein ratio 500/250 %.3f\n", b500 / b250 ); + std::printf( "INFO samples %d checksum %.6g\n", samples, sink ); + return 0; +} diff --git a/tools/bench_compare.py b/tools/bench_compare.py new file mode 100644 index 0000000..029076b --- /dev/null +++ b/tools/bench_compare.py @@ -0,0 +1,222 @@ +#!/usr/bin/env python3 +"""tools/bench_compare.py: the S10-R1/R3/R4 (D-008) judge behind `tools/check.sh bench compare`. + +usage: bench_compare.py --base --head [--base-par --head-par ] --kept --log + [--smoke] + +For each workload (head `--list`, or its `--list --smoke` subset plus the kept workloads with --smoke; env +BENCH_ONLY= restricts the set for experiments) it runs the processes in ORDER_FULL = ABBA BAAB ABBA BAAB +(A = base, B = head; ORDER_SMOKE = A B B A with --smoke) so drift and process-to-process noise cancel, each +`--only --samples ` with s = SAMPLES_FULL = 5 (SAMPLES_SMOKE = 3 with --smoke), pinned with +`taskset -c ${BENCH_CPU:-2}` when taskset exists. The runs of a build are pooled (8 x 5 = 40, smoke 2 x 3 = 6 +samples); median, rel. MAD (median absolute deviation / median) and +change = head median / base median - 1 are printed per workload. Stable: rel. MAD <= 5% in both builds. + +Parallel variant (S10-T6): --base-par/--head-par are the same harness built with -DFENG_MATRIX_PARALLEL; their +`--list` names the par/ set (prefixed `par/`). Those workloads run after the serial ones with the same order and +sample counts but are NOT pinned (they need many cores); they share the table, BENCH_ONLY, kept.txt and the rules. + +Judgement: bench/kept.txt (` ...`, `#` comments, blank lines) must list at least one optimization, +only known workloads, and each listed workload must have change <= -10% (KEPT-ok, else KEPT-miss). Full runs also +fail on any stable workload with change > +5% (REGRESSION); smoke runs print that flag but do not judge it. +The last line is `LANE bench PASS: kept, workloads, unstable` or `LANE bench FAIL: `. +Standard library only. +""" +import argparse +import os +import re +import shutil +import statistics +import subprocess +import sys + +STABLE_MAD = 0.05 +KEPT_GAIN = -0.10 +REGRESSION = 0.05 +ORDER_FULL = "ABBABAABABBABAAB" # A = base, B = head; ABBA BAAB ABBA BAAB, 8 processes per build +ORDER_SMOKE = "ABBA" # 2 processes per build +SAMPLES_FULL = 5 # samples per process +SAMPLES_SMOKE = 3 + + +def out(line=""): + print(line, flush=True) + + +def fail(reason): + out(f"LANE bench FAIL: {reason}") + sys.exit(1) + + +def run_bench(cmd, log): + proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True) + log.write(f"$ {' '.join(cmd)}\n{proc.stdout}") + log.flush() + if proc.returncode != 0: + fail(f"'{' '.join(cmd)}' exited {proc.returncode}, see {log.name}") + return proc.stdout + + +def parse_samples(text, name): + for line in text.splitlines(): + f = line.split() + if len(f) >= 4 and f[0] == "SAMPLES" and f[1] == name and f[2] == "reps": + return [float(x) for x in f[4:]] + return None + + +def info_lines(text): + return [l for l in text.splitlines() if l.startswith("INFO ")] + + +def read_kept(path, known): + """Returns [(opt_id, [workloads])]; fails on an unknown workload or a line without one.""" + kept = [] + try: + with open(path, encoding="utf-8") as f: + lines = f.readlines() + except OSError as e: + fail(f"cannot read {path}: {e.strerror}") + for no, raw in enumerate(lines, 1): + line = raw.split("#", 1)[0].strip() + if not line: + continue + f = line.split() + if len(f) < 2: + fail(f"{path}:{no}: optimization '{f[0]}' names no workload") + for w in f[1:]: + if w not in known: + fail(f"{path}:{no}: unknown workload '{w}' (not in bench --list)") + kept.append((f[0], f[1:])) + return kept + + +def stats(xs): + med = statistics.median(xs) + mad = statistics.median(abs(x - med) for x in xs) + return med, (mad / med if med > 0 else float("inf")) + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--base", required=True) + ap.add_argument("--head", required=True) + ap.add_argument("--base-par") + ap.add_argument("--head-par") + ap.add_argument("--kept", required=True) + ap.add_argument("--log", required=True) + ap.add_argument("--smoke", action="store_true") + a = ap.parse_args() + if bool(a.base_par) != bool(a.head_par): + ap.error("--base-par and --head-par go together") + + samples = SAMPLES_SMOKE if a.smoke else SAMPLES_FULL + order = [{"A": "base", "B": "head"}[c] for c in (ORDER_SMOKE if a.smoke else ORDER_FULL)] + runs_per_build = order.count("base") + assert runs_per_build == order.count("head") + smoke = ["--smoke"] if a.smoke else [] + cpu = os.environ.get("BENCH_CPU", "2") + if shutil.which("taskset"): + pin = ["taskset", "-c", cpu] + pinned = f"cpu {cpu} (taskset -c {cpu})" + else: + pin = [] + pinned = "none (taskset not found)" + + # (label, base binary, head binary, pin prefix): the serial pair pinned, the parallel pair unpinned + pairs = [("", a.base, a.head, pin)] + if a.base_par: + pairs.append(("-par", a.base_par, a.head_par, [])) + + with open(a.log, "w", encoding="utf-8") as log: + lists = [] + for label, base_exe, head_exe, _ in pairs: + names = run_bench([head_exe, "--list"], log).split() + if names != run_bench([base_exe, "--list"], log).split(): + fail(f"base{label} and head{label} bench --list differ (the harness must be the same source)") + lists.append(names) + known = [w for names in lists for w in names] + if len(set(known)) != len(known): + fail("a workload name is listed by both the serial and the parallel build") + kept = read_kept(a.kept, set(known)) + kept_w = {w for _, ws in kept for w in ws} + only = os.environ.get("BENCH_ONLY") + rx = re.compile(only) if only else None + jobs = [] # (workload, base binary, head binary, pin prefix) in harness order, serial pair first + for (label, base_exe, head_exe, pair_pin), names in zip(pairs, lists): + chosen = run_bench([head_exe, "--list", "--smoke"], log).split() if a.smoke else list(names) + chosen += [w for w in names if w in kept_w and w not in chosen] + if rx: + chosen = [w for w in chosen if rx.search(w)] + jobs += [(w, base_exe, head_exe, pair_pin) for w in names if w in chosen] + if only and not jobs: + fail(f"BENCH_ONLY='{only}' matches no workload") + + infos = {} + rows = [] + for w, base_exe, head_exe, pair_pin in jobs: + pool = {"base": [], "head": []} + label = "-par" if w.startswith("par/") else "" + for build in order: + exe = base_exe if build == "base" else head_exe + text = run_bench(pair_pin + [exe, "--only", w, "--samples", str(samples)] + smoke, log) + xs = parse_samples(text, w) + if not xs or len(xs) != samples: + fail(f"{build}{label} printed no SAMPLES line with {samples} samples for {w}, see {a.log}") + pool[build] += xs + infos.setdefault(build + label, info_lines(text)) + bm, bd = stats(pool["base"]) + hm, hd = stats(pool["head"]) + change = hm / bm - 1.0 + stable = bd <= STABLE_MAD and hd <= STABLE_MAD + if w in kept_w: + flag = "KEPT-ok" if change <= KEPT_GAIN else "KEPT-miss" + elif stable and change > REGRESSION: + flag = "REGRESSION" + else: + flag = "stable" if stable else "unstable" + rows.append((w, bm, hm, change, bd, hd, len(pool["base"]), len(pool["head"]), stable, flag)) + + for build in ("head", "base", "head-par", "base-par"): + for l in infos.get(build, []): + key = l.split()[1] if len(l.split()) > 1 else "" + if build == "head" or key == "flags" or key == "compiler": + out(f"INFO {build} {l[5:]}") + out(f"INFO pinned {pinned}") + if a.base_par: + out("INFO pinned par/ none (par/ workloads run unpinned: the parallel build needs many cores)") + seq = ORDER_SMOKE if a.smoke else ORDER_FULL + seq = " ".join(seq[i:i + 4] for i in range(0, len(seq), 4)) + out(f"INFO mode {'smoke' if a.smoke else 'full'}, order {seq} " + f"(A base, B head), {runs_per_build} runs x {samples} samples = {runs_per_build * samples} pooled per build") + if only: + out(f"INFO BENCH_ONLY {only}") + out(f"{'workload':<24} {'base_med_s':>12} {'head_med_s':>12} {'change%':>8} {'base_mad%':>9} " + f"{'head_mad%':>9} {'n':>7} flag") + for w, bm, hm, ch, bd, hd, nb, nh, _, flag in rows: + out(f"{w:<24} {bm:12.4e} {hm:12.4e} {100 * ch:+8.2f} {100 * bd:9.2f} {100 * hd:9.2f} " + f"{f'{nb}/{nh}':>7} {flag}") + + reasons = [] + ran = {r[0]: r for r in rows} + if not kept: + reasons.append(f"no kept optimization ({a.kept} lists none)") + for opt, ws in kept: + for w in ws: + if w not in ran: + reasons.append(f"{opt}: kept workload {w} not run (BENCH_ONLY)") + elif ran[w][9] == "KEPT-miss": + reasons.append(f"{opt}: {w} change {100 * ran[w][3]:+.2f}% is not <= -10%") + regress = [r for r in rows if r[9] == "REGRESSION"] + if regress and a.smoke: + out(f"INFO smoke: {len(regress)} REGRESSION flag(s) printed, not judged") + elif regress: + reasons.append("stable regression > +5%: " + ", ".join(f"{r[0]} {100 * r[3]:+.2f}%" for r in regress)) + if reasons: + fail("; ".join(reasons)) + unstable = sum(1 for r in rows if not r[8]) + out(f"LANE bench PASS: {len(kept)} kept, {len(rows)} workloads, {unstable} unstable") + + +if __name__ == "__main__": + main() diff --git a/tools/check.sh b/tools/check.sh new file mode 100755 index 0000000..8a2ba80 --- /dev/null +++ b/tools/check.sh @@ -0,0 +1,933 @@ +#!/usr/bin/env bash +# tools/check.sh [args] [--smoke] (D-013) +# Builds only under build//, logs under build/logs//, TMPDIR=build/tmp, runs from the repo root. +# Each suite run gets its own fresh TMPDIR=build/tmp/run.XXXXXX, removed after the run (concurrent runs never share one). +# Lanes: gcc clang sanitize tagged '' warnings api examples make ci docs compile-fail fuzz tsan +# oracle bench bench compare config all. +# `fuzz` (S5-R5), `tsan` (S6-R2), `oracle` (S7-R6), `bench` (S8-R4; `bench compare` S10-R1) and `config` (S9-R3) +# are not part of `all` (cost, optional Eigen, timing, link-failure pairs). +# BENCH_CPU (default 2) is the CPU `bench compare` pins to; BENCH_ONLY= restricts its workloads +# (experiments only: a restricted run cannot pass while a kept workload is left out). +# Last line: `LANE PASS[: detail]` (exit 0) or `LANE FAIL: ` (exit non-zero). +# Environment: CHECK_GCCS (default "g++ g++-15"), CHECK_CLANG (default clang++), CHECK_JOBS (default 8), +# CHECK_REBUILD=1 forces rebuilding binaries that are otherwise reused when up to date, +# CHECK_API_FAMILIES="a b ..." replaces the api lane's family list (default: the backticked first +# column of the `## API families (S1-R4)` table in docs/stages/S1/spec.md). +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +cd "$ROOT" +SELF="$ROOT/tools/check.sh" + +export TMPDIR="$ROOT/build/tmp" +mkdir -p "$TMPDIR" + +CHECK_GCCS="${CHECK_GCCS:-g++ g++-15}" +CHECK_CLANG="${CHECK_CLANG:-clang++}" +CHECK_JOBS="${CHECK_JOBS:-8}" +SAN_FLAGS="-fsanitize=address,undefined -fno-sanitize-recover=all -O1 -g -DFENG_MATRIX_CHECKED_ITERATORS" +export ASAN_OPTIONS="${ASAN_OPTIONS:-detect_leaks=1}" +export UBSAN_OPTIONS="${UBSAN_OPTIONS:-print_stacktrace=1}" + +die_lane() { # lane reason + echo "LANE $1 FAIL: $2" + exit 1 +} + +# True when $1 exists, its flag stamp equals $2 and it is newer than every input. +up_to_date() { + local bin="$1" flags="$2" f + [[ "${CHECK_REBUILD:-0}" == 1 ]] && return 1 + [[ -x "$bin" && -f "$bin.flags" ]] || return 1 + [[ "$(cat "$bin.flags")" == "$flags" ]] || return 1 + for f in matrix.hpp tests/test.cc tests/catch.hpp tests/cases/* tests/fuzz/*.hpp; do + [[ "$f" -nt "$bin" ]] && return 1 + done + return 0 +} + +# build ; returns the compiler's status. +build_suite() { + local cxx="$1" flags="$2" bin="$3" log="$4" + mkdir -p "$(dirname "$bin")" "$(dirname "$log")" + if up_to_date "$bin" "$cxx $flags"; then + echo "reused $bin (up to date)" > "$log" + return 0 + fi + rm -f "$bin" "$bin.flags" + # shellcheck disable=SC2086 + if "$cxx" $flags -pthread -isystem tests tests/test.cc -o "$bin" > "$log" 2>&1; then + printf '%s' "$cxx $flags" > "$bin.flags" + return 0 + fi + return 1 +} + +# run_suite [catch spec]; succeeds only if it exits 0, at least one test case ran and nothing failed. +run_suite() { + local bin="$1" log="$2" tmp rc=0; shift 2 + tmp="$(mktemp -d "$TMPDIR/run.XXXXXX")" || return 1 + TMPDIR="$tmp" "$bin" "$@" > "$log" 2>&1 || rc=$? + rm -rf "$tmp" + [[ $rc -eq 0 ]] || return 1 + grep -q '^No tests ran' "$log" && return 2 + grep -Eq '^All tests passed \([0-9]+ assertions? in [1-9][0-9]* test cases?\)' "$log" || return 2 + return 0 +} + +# Internal: one config of a build matrix. __config +config_job() { + local lane="$1" cxx="$2" std="$3" mode="$4" par="$5" + local flags="-std=c++$std" + if [[ "$mode" == debug ]]; then flags+=" -O0 -g"; else flags+=" -O2 -DNDEBUG"; fi + [[ "$par" == parallel ]] && flags+=" -DFENG_MATRIX_PARALLEL" + local id="${cxx}-c++${std}-${mode}-${par}" + local bin="build/$lane/$id/test" logd="build/logs/$lane" + local status=PASS why="" + if ! build_suite "$cxx" "$flags" "$bin" "$logd/$id.build.log"; then + status=FAIL; why=" (build, see $logd/$id.build.log)" + else + local rc=0 + run_suite "$bin" "$logd/$id.run.log" || rc=$? + if [[ $rc == 2 ]]; then status=FAIL; why=" (no test ran, see $logd/$id.run.log)" + elif [[ $rc != 0 ]]; then status=FAIL; why=" (tests, see $logd/$id.run.log)"; fi + fi + echo "CONFIG $cxx c++$std $mode $par $status$why" + echo "$status" > "build/$lane/$id.result" +} + +# matrix_lane +matrix_lane() { + local lane="$1" + mkdir -p "build/$lane" "build/logs/$lane" + rm -f "build/$lane"/*.result + local jobs; jobs="$(cat)" + local n; n="$(grep -c . <<< "$jobs" || true)" + [[ "$n" -gt 0 ]] || die_lane "$lane" "no configurations" + # xargs exits non-zero if a job fails; results are judged from the result files. + xargs -P "$CHECK_JOBS" -L 1 "$SELF" __config "$lane" <<< "$jobs" || true + local pass fail + pass="$(cat "build/$lane"/*.result 2>/dev/null | grep -c '^PASS$' || true)" + fail=$(( n - pass )) + if [[ "$fail" -ne 0 ]]; then + die_lane "$lane" "$fail of $n configurations failed" + fi + echo "LANE $lane PASS: $n configurations" +} + +lane_gcc() { + local cxx std mode par + { + if [[ "$SMOKE" == 1 ]]; then + echo "g++ 20 debug parallel" + echo "g++-15 26 ndebug serial" + else + for cxx in $CHECK_GCCS; do for std in 20 23 26; do for mode in debug ndebug; do for par in serial parallel; do + echo "$cxx $std $mode $par" + done; done; done; done + fi + } | matrix_lane gcc +} + +lane_clang() { + local cxx std mode par + { + if [[ "$SMOKE" == 1 ]]; then + echo "$CHECK_CLANG 23 ndebug parallel" + else + for cxx in $CHECK_CLANG; do for std in 20 23 26; do for mode in debug ndebug; do for par in serial parallel; do + echo "$cxx $std $mode $par" + done; done; done; done + fi + } | matrix_lane clang +} + +first_gcc() { set -- $CHECK_GCCS; echo "$1"; } + +# Internal: one sanitizer build. __sanbuild ; builds build/sanitize/-/test. +san_bin() { echo "build/sanitize/$1-$2/test"; } +san_flags() { + local f="-std=c++20 $SAN_FLAGS -DFENG_MATRIX_PARALLEL" + [[ "$1" == ndebug ]] && f+=" -DNDEBUG" + echo "$f" +} +san_build_job() { + local cxx="$1" mode="$2" + mkdir -p build/logs/sanitize + if build_suite "$cxx" "$(san_flags "$mode")" "$(san_bin "$cxx" "$mode")" "build/logs/sanitize/$cxx-$mode.build.log"; then + echo "BUILD $cxx $mode asan+ubsan PASS" + else + echo "BUILD $cxx $mode asan+ubsan FAIL (see build/logs/sanitize/$cxx-$mode.build.log)" + fi +} + +san_variants() { # prints " " lines + local g; g="$(first_gcc)" + if [[ "$SMOKE" == 1 && "$1" == sanitize ]]; then + echo "$g debug"; echo "$CHECK_CLANG ndebug" + else + echo "$g debug"; echo "$g ndebug"; echo "$CHECK_CLANG debug"; echo "$CHECK_CLANG ndebug" + fi +} + +# san_runs [spec]; builds every variant, runs each, checks logs. +san_runs() { + local lane="$1"; shift + mkdir -p build/sanitize "build/logs/$lane" + local variants; variants="$(san_variants "$lane")" + xargs -P "$CHECK_JOBS" -L 1 "$SELF" __sanbuild <<< "$variants" || true + local n=0 fail=0 nomatch=0 cxx mode bin log rc + while read -r cxx mode; do + n=$((n + 1)) + bin="$(san_bin "$cxx" "$mode")" + log="build/logs/$lane/$cxx-$mode.run.log" + if [[ ! -x "$bin" ]]; then + echo "CONFIG $cxx c++20 $mode asan+ubsan FAIL (build)"; fail=$((fail + 1)); continue + fi + rc=0 + run_suite "$bin" "$log" "$@" || rc=$? + if grep -Eq 'runtime error:|AddressSanitizer' "$log"; then rc=3; fi + case $rc in + 0) echo "CONFIG $cxx c++20 $mode asan+ubsan PASS" ;; + 2) echo "CONFIG $cxx c++20 $mode asan+ubsan FAIL (no test matched, see $log)"; nomatch=$((nomatch + 1)) ;; + 3) echo "CONFIG $cxx c++20 $mode asan+ubsan FAIL (sanitizer diagnostic, see $log)"; fail=$((fail + 1)) ;; + *) echo "CONFIG $cxx c++20 $mode asan+ubsan FAIL (tests, see $log)"; fail=$((fail + 1)) ;; + esac + done <<< "$variants" + if [[ $nomatch -ne 0 ]]; then die_lane "$lane" "$nomatch of $n runs matched no test"; fi + if [[ $fail -ne 0 ]]; then die_lane "$lane" "$fail of $n runs failed"; fi + echo "LANE $lane PASS: $n runs" +} + +lane_sanitize() { san_runs sanitize; } + +lane_tagged() { + [[ ${#ARGS[@]} -ge 1 && -n "${ARGS[0]}" ]] || die_lane tagged "usage: tools/check.sh tagged '' [--smoke]" + san_runs tagged "${ARGS[@]}" +} + +# ---- warnings lane (S1-R3) ---- +WARN_FLAGS="-Wall -Wextra -Werror -DFENG_MATRIX_PARALLEL -pthread -isystem tests -fsyntax-only" + +# Internal: one warnings config. __warn ; any diagnostic, or any literal-operator deprecation, fails it. +warn_job() { + local cxx="$1" std="$2" id="$1-c++$2" log="build/logs/warnings/$1-c++$2.log" status=PASS why="" + mkdir -p build/warnings build/logs/warnings + # shellcheck disable=SC2086 + if ! "$cxx" -std=c++$std $WARN_FLAGS tests/test.cc > "$log" 2>&1; then + status=FAIL; why=" (compile, see $log)" + elif [[ -s "$log" ]]; then + status=FAIL; why=" (diagnostics, see $log)" + fi + if grep -q 'deprecated-literal-operator' "$log"; then status=FAIL; why=" (deprecated-literal-operator, see $log)"; fi + echo "CONFIG $cxx c++$std warnings $status$why" + echo "$status" > "build/warnings/$id.result" +} + +lane_warnings() { + local cxx std jobs n pass + if [[ "$SMOKE" == 1 ]]; then + jobs="$(first_gcc) 26"$'\n'"$CHECK_CLANG 26" + else + jobs="$(for cxx in $CHECK_GCCS $CHECK_CLANG; do for std in 20 26; do echo "$cxx $std"; done; done)" + fi + mkdir -p build/warnings build/logs/warnings + rm -f build/warnings/*.result + n="$(grep -c . <<< "$jobs")" + xargs -P "$CHECK_JOBS" -L 1 "$SELF" __warn <<< "$jobs" || true + pass="$(cat build/warnings/*.result 2>/dev/null | grep -c '^PASS$' || true)" + [[ $(( n - pass )) -eq 0 ]] || die_lane warnings "$(( n - pass )) of $n configurations failed" + echo "LANE warnings PASS: $n configurations" +} + +# ---- api lane (S1-R4) ---- +API_FLAGS="-std=c++20 -Wall -Wextra -Werror -DFENG_MATRIX_PARALLEL" + +# The families: CHECK_API_FAMILIES if set, else the backticked first column of the spec's API families table. +api_families() { + if [[ -n "${CHECK_API_FAMILIES:-}" ]]; then + tr ' ' '\n' <<< "$CHECK_API_FAMILIES" | grep . + return 0 + fi + awk '/^## API families \(S1-R4\)/ { on = 1; next } + on && /^## / { exit } + on && /^\| *`[^`]+` *\|/ { s = $0; sub(/^\| *`/, "", s); sub(/`.*/, "", s); print s }' docs/stages/S1/spec.md +} + +api_exflag() { [[ "$1" == noexc ]] && echo "-fno-exceptions" || true; } + +# Internal: compile one family TU alone. __api +api_job() { + local cxx="$1" mode="$2" fam="$3" id="$1-$2-$3" + local obj="build/api/$1-$2/$3.o" log="build/logs/api/$id.log" status=PASS why="" + mkdir -p "build/api/$1-$2" build/logs/api + rm -f "$obj" + if [[ ! -f "tests/api/$fam.cc" ]]; then + status=FAIL; why=" (missing tests/api/$fam.cc)"; : > "$log" + else + # shellcheck disable=SC2046,SC2086 + "$cxx" $API_FLAGS $(api_exflag "$mode") -c "tests/api/$fam.cc" -o "$obj" > "$log" 2>&1 \ + || { status=FAIL; why=" (compile, see $log)"; } + fi + echo "API $fam $cxx $mode $status$why" + echo "$status" > "build/api/$id.result" +} + +# Internal: build link_a.cc + link_b.cc with one configuration, link and run. __apilink +api_link_job() { + local cxx="$1" d="build/api/link-$1" log="build/logs/api/link-$1.log" status=PASS why="" + mkdir -p "$d" build/logs/api + rm -f "$d"/*.o "$d/link" + { + # shellcheck disable=SC2086 + "$cxx" $API_FLAGS -O2 -pthread -c tests/api/link_a.cc -o "$d/link_a.o" && + "$cxx" $API_FLAGS -O2 -pthread -c tests/api/link_b.cc -o "$d/link_b.o" && + "$cxx" -pthread "$d/link_a.o" "$d/link_b.o" -o "$d/link" + } > "$log" 2>&1 || { status=FAIL; why=" (build or link, see $log)"; } + if [[ "$status" == PASS ]]; then + local rc=0 + "$d/link" >> "$log" 2>&1 || rc=$? + [[ $rc -eq 0 ]] || { status=FAIL; why=" (program exited $rc, see $log)"; } + fi + echo "LINK $cxx link_a+link_b $status$why" + echo "$status" > "build/api/link-$cxx.result" +} + +lane_api() { + local fams cxx mode f n jobs links compilers + fams="$(api_families)" + n="$(grep -c . <<< "$fams" || true)" + [[ "$n" -gt 0 ]] || die_lane api "no API families found" + if [[ "$SMOKE" == 1 ]]; then compilers="$(first_gcc)"; else compilers="$CHECK_GCCS $CHECK_CLANG"; fi + jobs="$(for cxx in $compilers; do for mode in exc noexc; do for f in $fams; do echo "$cxx $mode $f"; done; done; done)" + links="$(for cxx in $compilers; do echo "$cxx"; done)" + mkdir -p build/api build/logs/api + rm -f build/api/*.result + xargs -P "$CHECK_JOBS" -L 1 "$SELF" __api <<< "$jobs" || true + xargs -P "$CHECK_JOBS" -L 1 "$SELF" __apilink <<< "$links" || true + local total pass fail + total=$(( $(grep -c . <<< "$jobs") + $(grep -c . <<< "$links") )) + pass="$(cat build/api/*.result 2>/dev/null | grep -c '^PASS$' || true)" + fail=$(( total - pass )) + [[ "$fail" -eq 0 ]] || die_lane api "$n families, $fail failures" + echo "LANE api PASS: $n families, 0 failures" +} + +# ---- examples lane (S1-R5) ---- +# Builds examples/example.cc under build/examples/ and runs it in a fresh scratch copy of images/ (the program +# rewrites ./images/*); the repo's images/ must stay untouched. --smoke is the same run (one binary). +EX_FLAGS="-std=c++20 -O2 -DFENG_MATRIX_PARALLEL" +lane_examples() { + local cxx bin=build/examples/example log=build/logs/examples run=build/examples/run stamp + cxx="$(first_gcc)" + mkdir -p build/examples "$log" + stamp="build/examples/images.stamp" + : > "$stamp" + local stale=1 f + if [[ "${CHECK_REBUILD:-0}" != 1 && -x "$bin" && -f "$bin.flags" && "$(cat "$bin.flags")" == "$cxx $EX_FLAGS" ]]; then + stale=0 + for f in matrix.hpp examples/example.cc examples/cases/*; do [[ "$f" -nt "$bin" ]] && stale=1; done + fi + if [[ $stale == 1 ]]; then + rm -f "$bin" "$bin.flags" + # shellcheck disable=SC2086 + "$cxx" $EX_FLAGS -pthread examples/example.cc -o "$bin" > "$log/build.log" 2>&1 \ + || die_lane examples "build failed, see $log/build.log" + printf '%s' "$cxx $EX_FLAGS" > "$bin.flags" + else + echo "reused $bin (up to date)" > "$log/build.log" + fi + rm -rf "$run" + mkdir -p "$run" + cp -r images "$run/images" + local rc=0 + ( cd "$run" && "$ROOT/$bin" ) > "$log/run.log" 2>&1 || rc=$? + if [[ -n "$(find images -newer "$stamp" -print -quit)" ]]; then + die_lane examples "the repo's images/ was modified" + fi + [[ $rc -eq 0 ]] || die_lane examples "example exited $rc, see $log/run.log" + echo "LANE examples PASS: example ran in $run" +} + +# ---- make lane (S1-R6) ---- +# Asserts the default and FAST=1 flag sets from `make -n`, then runs one real default build into build/make/. +lane_make() { + local log=build/logs/make dry fast before after + mkdir -p "$log" + dry="$(env -u BUILD_DIR -u FAST make -n -B test example 2>&1)" || die_lane make "make -n failed" + printf '%s\n' "$dry" > "$log/dry-default.log" + grep -q -- '-O2' <<< "$dry" || die_lane make "default flags lack -O2" + if grep -Eq -- '-Ofast|-march=native|-ffast-math|-flto' <<< "$dry"; then + die_lane make "default flags contain -Ofast, -march=native, -ffast-math or -flto" + fi + grep -q -- '-o build/test_test' <<< "$dry" || die_lane make "default output is not build/test_test" + grep -q -- '-o build/test_example' <<< "$dry" || die_lane make "default output is not build/test_example" + if grep -oE -- '-o [^ ]+' <<< "$dry" | grep -qv -- '^-o build/'; then + die_lane make "an output path is outside build/" + fi + fast="$(env -u BUILD_DIR make -n -B FAST=1 test example 2>&1)" || die_lane make "make -n FAST=1 failed" + printf '%s\n' "$fast" > "$log/dry-fast.log" + grep -q -- '-Ofast' <<< "$fast" || die_lane make "FAST=1 lacks -Ofast" + grep -q -- '-march=native' <<< "$fast" || die_lane make "FAST=1 lacks -march=native" + before="$(ls -A)" + local bflag="" + [[ "${CHECK_REBUILD:-0}" == 1 ]] && bflag="-B" + # shellcheck disable=SC2086 + env -u FAST make $bflag -j2 BUILD_DIR=build/make test example > "$log/build.log" 2>&1 \ + || die_lane make "make test example failed, see $log/build.log" + after="$(ls -A)" + [[ "$before" == "$after" ]] || die_lane make "make wrote at the repo root" + [[ -x build/make/test_test && -x build/make/test_example ]] || die_lane make "binaries missing in build/make/" + echo "LANE make PASS: -O2 default, FAST=1 -Ofast -march=native, build/make/test_test and test_example" +} + +# ---- ci lane (S1-R6) ---- +lane_ci() { + local log=build/logs/ci out rc=0 + mkdir -p "$log" + out="$(python3 - .github/workflows/ci.yml <<'EOF' 2>&1 +import re, sys +try: + import yaml +except ImportError: + print("FAIL: PyYAML missing"); sys.exit(1) +path = sys.argv[1] +text = open(path).read() +doc = yaml.safe_load(text) +jobs = (doc or {}).get("jobs") or {} +if not jobs: + print("FAIL: no jobs"); sys.exit(1) +if re.search(r"(gcc|g\+\+)[-: ]?12\b|version:\s*['\"]?12\b", text, re.I): + print("FAIL: GCC 12 appears"); sys.exit(1) +gcc_jobs, clang_jobs = [], [] +for name, job in jobs.items(): + blob = yaml.safe_dump(job) + runs = " ".join(str(s.get("run", "")) for s in (job.get("steps") or [])) + if "tools/check.sh" not in runs: + continue + gv = [int(v) for v in re.findall(r"(?:gcc|g\+\+)[-: ](\d+)", blob, re.I)] + cv = [int(v) for v in re.findall(r"clang(?:\+\+)?[-: ](\d+)", blob, re.I)] + if gv and min(gv) >= 15 and not cv: + gcc_jobs.append(f"{name} (GCC {min(gv)})") + if cv and min(cv) >= 22: + clang_jobs.append(f"{name} (Clang {min(cv)})") +if not gcc_jobs: + print("FAIL: no GCC >= 15 job running tools/check.sh"); sys.exit(1) +if not clang_jobs: + print("FAIL: no Clang >= 22 job running tools/check.sh"); sys.exit(1) +print("OK: gcc " + ", ".join(gcc_jobs) + "; clang " + ", ".join(clang_jobs)) +EOF +)" || rc=$? + printf '%s\n' "$out" > "$log/ci.log" + [[ $rc -eq 0 ]] || die_lane ci "$(tail -n 1 <<< "$out" | sed 's/^FAIL: //')" + echo "LANE ci PASS: ${out#OK: }" +} + +# ---- docs lane (S1-R6, D-013) ---- +lane_docs() { + [[ ${#ARGS[@]} -ge 1 && -n "${ARGS[0]}" ]] || die_lane docs "usage: tools/check.sh docs " + local id="${ARGS[0]}" f=docs/migration.md section bad n + if [[ "$id" == S9 ]]; then + # S9-R1: tools/const_returns.py (clang AST, D-034) counts top-level const value returns in matrix.hpp. + # S9-R6: matrix.hpp line counts at the stage start (the parent of the commit whose subject is exactly + # `S9: spec and tasks`; the oldest one if there are several) and at head. + local scan cr start_commit lines_start lines_head + scan="$(python3 tools/const_returns.py)" || die_lane docs "tools/const_returns.py failed: $(tail -n 1 <<< "$scan")" + [[ -n "$scan" ]] && grep -v '^CONST_RETURNS ' <<< "$scan" || true + cr="$(sed -n 's/^CONST_RETURNS \([0-9][0-9]*\)$/\1/p' <<< "$scan")" + [[ -n "$cr" ]] || die_lane docs "tools/const_returns.py printed no CONST_RETURNS count" + echo "CONST_RETURNS $cr" + start_commit="$(git log --format='%H %s' --grep='^S9: spec and tasks$' \ + | awk '{ h = $1; $1 = ""; if ( substr( $0, 2 ) == "S9: spec and tasks" ) last = h } END { print last }')" + [[ -n "$start_commit" ]] || die_lane docs "no commit with subject 'S9: spec and tasks'" + lines_start="$(git show "$start_commit^:matrix.hpp" | wc -l)" + lines_head="$(wc -l < matrix.hpp)" + echo "LINES start $lines_start head $lines_head" + [[ "$cr" -eq 0 ]] || die_lane docs "$cr top-level const value returns in matrix.hpp" + [[ "$lines_head" -lt "$lines_start" ]] \ + || die_lane docs "matrix.hpp has $lines_head lines at head, not fewer than $lines_start at the stage start" + fi + [[ -f "$f" ]] || die_lane docs "$f missing" + grep -qx "## $id" "$f" || die_lane docs "no '## $id' section in $f" + section="$(awk -v h="## $id" '$0 == h { on = 1; next } on && /^## / { exit } on' "$f")" + n="$(grep -c '^- ' <<< "$section" || true)" + [[ "$n" -gt 0 ]] || die_lane docs "'## $id' has no entries" + bad="$(grep '^- ' <<< "$section" \ + | grep -Ev '^- none$' \ + | grep -Ev '^- [^ ].*: old: .+; new: .+; finding: (F[0-9][0-9]|none)$' || true)" + [[ -z "$bad" ]] || die_lane docs "malformed entry: $(head -n 1 <<< "$bad")" + if [[ "$id" == S1 ]]; then + grep -q 'tools/check.sh' ReadMe.md || die_lane docs "ReadMe.md does not name tools/check.sh" + grep -q 'FAST=1' ReadMe.md || die_lane docs "ReadMe.md does not name FAST=1" + grep -q 'make test example' ReadMe.md || die_lane docs "ReadMe.md does not name make test example" + fi + if [[ "$id" == S2 ]]; then + grep -q 'std::abort' ReadMe.md || die_lane docs "ReadMe.md does not name std::abort" + grep -q 'fliplr' ReadMe.md || die_lane docs "ReadMe.md does not name fliplr" + grep -qF 'at(' ReadMe.md || die_lane docs "ReadMe.md does not name at(" + fi + if [[ "$id" == S3 ]]; then + grep -q 'matrix_element' ReadMe.md || die_lane docs "ReadMe.md does not name matrix_element" + grep -q 'must not throw' ReadMe.md || die_lane docs "ReadMe.md does not say callbacks must not throw" + fi + if [[ "$id" == S4 ]]; then + grep -q 'invalidat' ReadMe.md || die_lane docs "ReadMe.md does not describe iterator invalidation" + grep -q 'make_mutable_view' ReadMe.md || die_lane docs "ReadMe.md does not name make_mutable_view" + grep -q 'temporary' ReadMe.md || die_lane docs "ReadMe.md does not say a temporary owner is rejected" + fi + if [[ "$id" == S5 ]]; then + grep -q 'save_as_npy' ReadMe.md || die_lane docs "ReadMe.md does not name save_as_npy" + grep -q 'fortran_order' ReadMe.md || die_lane docs "ReadMe.md does not name fortran_order" + fi + if [[ "$id" == S6 ]]; then + grep -q 'uniform_random_bit_generator' ReadMe.md || die_lane docs "ReadMe.md does not name uniform_random_bit_generator" + grep -q 'ddof' ReadMe.md || die_lane docs "ReadMe.md does not name ddof" + grep -q 'common_element_t' ReadMe.md || die_lane docs "ReadMe.md does not name common_element_t" + grep -q 'overflow' ReadMe.md || die_lane docs "ReadMe.md does not state the overflow precondition" + grep -q 'must not throw' ReadMe.md || die_lane docs "ReadMe.md does not say callbacks must not throw" + fi + if [[ "$id" == S7 ]]; then + grep -q 'lu_factor' ReadMe.md || die_lane docs "ReadMe.md does not name lu_factor" + grep -q 'linalg_status' ReadMe.md || die_lane docs "ReadMe.md does not name linalg_status" + grep -q 'experimental' ReadMe.md || die_lane docs "ReadMe.md does not name the experimental routines" + fi + if [[ "$id" == S9 ]]; then + grep -q 'FENG_MATRIX_PARALLEL' ReadMe.md || die_lane docs "ReadMe.md does not name FENG_MATRIX_PARALLEL" + grep -q 'nodiscard' ReadMe.md || die_lane docs "ReadMe.md does not name nodiscard" + grep -q 'to_mdspan' ReadMe.md || die_lane docs "ReadMe.md does not name to_mdspan" + grep -q 'to_expected' ReadMe.md || die_lane docs "ReadMe.md does not name to_expected" + fi + echo "LANE docs PASS: $id, $n entries" +} + +# ---- compile-fail lane (S3-R3) ---- +# For each tests/compile_fail//*.cc, compiles with the first GCC and $CHECK_CLANG (-std=c++20 -fsyntax-only +# -isystem tests, under LC_ALL=C so GCC quotes with ASCII '); a case passes when every compile fails and, per +# compiler, the diagnostic contains the case's `// expect:` text (required: a case without it fails) and, when the +# case has one, that compiler's `// expect-gcc:` (GCC compile) or `// expect-clang:` (Clang compile) text. Lines +# quoting a `// expect` comment itself do not count. A case's optional `// flags:` line appends its text to the +# compile flags of both compilers (S9-R1: `-Werror=unused-result`); a case without it compiles with $CF_FLAGS only. +# Logs under build/logs/compile-fail//; --smoke is the same run; zero cases is a failure. +CF_FLAGS="-std=c++20 -fsyntax-only -isystem tests" +# cf_expect : the text after the first `// :` in , trailing blanks trimmed. +cf_expect() { sed -n "s|^.*// $2: *||p" "$1" | head -n 1 | sed 's/[[:space:]]*$//'; } +lane_compile_fail() { + [[ ${#ARGS[@]} -ge 1 && -n "${ARGS[0]}" ]] || die_lane compile-fail "usage: tools/check.sh compile-fail " + local id="${ARGS[0]}" dir="tests/compile_fail/${ARGS[0]}" logd="build/logs/compile-fail/${ARGS[0]}" + local f name expect tight cxx kind log n=0 bad=0 ok miss extra + [[ -d "$dir" ]] || die_lane compile-fail "no directory $dir" + mkdir -p "$logd" + local cases=() + for f in "$dir"/*.cc; do [[ -f "$f" ]] && cases+=("$f"); done + [[ ${#cases[@]} -gt 0 ]] || die_lane compile-fail "no cases in $dir" + for f in "${cases[@]}"; do + n=$((n + 1)) + name="$(basename "$f" .cc)" + expect="$(cf_expect "$f" expect)" + if [[ -z "$expect" ]]; then + echo "CASE $id/$name FAIL (no // expect: line)"; bad=$((bad + 1)); continue + fi + extra="$(cf_expect "$f" flags)" + ok=1 + for kind in gcc clang; do + if [[ $kind == gcc ]]; then cxx="$(first_gcc)"; else cxx="$CHECK_CLANG"; fi + tight="$(cf_expect "$f" "expect-$kind")" + log="$logd/$name.$cxx.log" + # shellcheck disable=SC2086 + if LC_ALL=C "$cxx" $CF_FLAGS $extra "$f" > "$log" 2>&1; then + echo "CASE $id/$name $cxx FAIL (compiled, see $log)"; ok=0 + else + miss="" + grep -vF -- '// expect' "$log" | grep -qF -- "$expect" || miss="'$expect'" + if [[ -n "$tight" ]] && ! grep -vF -- '// expect' "$log" | grep -qF -- "$tight"; then + miss="${miss:+$miss and }'$tight'" + fi + if [[ -z "$miss" ]]; then + echo "CASE $id/$name $cxx PASS (expect: $expect${tight:+; expect-$kind: $tight}${extra:+; flags: $extra})" + else + echo "CASE $id/$name $cxx FAIL (diagnostic lacks $miss, see $log)"; ok=0 + fi + fi + done + [[ $ok == 1 ]] || bad=$((bad + 1)) + done + [[ $bad -eq 0 ]] || die_lane compile-fail "$bad of $n cases failed" + echo "LANE compile-fail PASS: $n cases" +} + +# ---- fuzz lane (S5-R5) ---- +# Builds tests/fuzz/fuzz_.cc for t in npy bmp binary text with $CHECK_CLANG and libFuzzer+ASan+UBSan into +# build/fuzz// (logs build/logs/fuzz/), copies tests/fuzz/corpus// into a fresh build/fuzz//corpus, replays +# every file in tests/fuzz/regressions// (if it exists), then runs the four targets in parallel for 300 s each +# (--smoke: 20 s) with a fixed seed per target and input, time, memory and leak limits. Passes only when every run +# exited 0, no crash-/leak-/timeout-/oom-/slow-unit- artifact exists and `git status --short` is unchanged. +FUZZ_FLAGS="-std=c++20 -g -O1 -fsanitize=fuzzer,address,undefined -fno-sanitize-recover=all -DFENG_MATRIX_CHECKED_ITERATORS" +FUZZ_TARGETS="npy bmp binary text" +fuzz_seed() { case "$1" in npy) echo 1001 ;; bmp) echo 1002 ;; binary) echo 1003 ;; text) echo 1004 ;; esac; } +fuzz_max_len() { [[ "$1" == text ]] && echo 16384 || echo 65536; } + +# Internal: one target. __fuzz ; prints `TARGET PASS|FAIL (...)` and writes build/fuzz/.result. +fuzz_job() { + local t="$1" secs="$2" d="build/fuzz/$1" logd="build/logs/fuzz" status=PASS why="" + local bin="build/fuzz/$1/fuzz_$1" art="build/fuzz/$1/artifacts/" + mkdir -p "$logd" + rm -rf "$d" + mkdir -p "$d/corpus" "$art" + local opts=( -seed="$(fuzz_seed "$t")" -max_len="$(fuzz_max_len "$t")" -timeout=10 -rss_limit_mb=2048 + -malloc_limit_mb=512 -detect_leaks=1 -artifact_prefix="$art" ) + # shellcheck disable=SC2086 + if ! "$CHECK_CLANG" $FUZZ_FLAGS -pthread "tests/fuzz/fuzz_$t.cc" -o "$bin" > "$logd/$t.build.log" 2>&1; then + echo "TARGET $t FAIL (build, see $logd/$t.build.log)"; echo FAIL > "build/fuzz/$t.result"; return 0 + fi + [[ -d "tests/fuzz/corpus/$t" ]] && cp -r "tests/fuzz/corpus/$t/." "$d/corpus/" + local regs=() f nreg=0 + if [[ -d "tests/fuzz/regressions/$t" ]]; then + for f in "tests/fuzz/regressions/$t"/*; do [[ -f "$f" ]] && regs+=("$f"); done + fi + nreg=${#regs[@]} + : > "$logd/$t.replay.log" + if [[ $nreg -gt 0 ]] && ! "$bin" "${opts[@]}" "${regs[@]}" > "$logd/$t.replay.log" 2>&1; then + status=FAIL; why="regression replay failed, see $logd/$t.replay.log" + fi + local rc=0 start=$SECONDS + if [[ $status == PASS ]]; then + "$bin" "${opts[@]}" -max_total_time="$secs" -print_final_stats=1 "$d/corpus" > "$logd/$t.run.log" 2>&1 || rc=$? + [[ $rc -eq 0 ]] || { status=FAIL; why="exit $rc, see $logd/$t.run.log"; } + fi + local found + found="$(find "$art" -maxdepth 1 -type f \( -name 'crash-*' -o -name 'leak-*' -o -name 'timeout-*' -o -name 'oom-*' -o -name 'slow-unit-*' \) | head -n 1)" + if [[ -n "$found" ]]; then status=FAIL; why="${why:+$why; }artifact $found"; fi + local runs; runs="$(sed -n 's/^stat::number_of_executed_units: *//p' "$logd/$t.run.log" 2>/dev/null | tail -n 1)" + if [[ $status == PASS ]]; then + why="$((SECONDS - start)) s, ${runs:-?} runs, $nreg regressions, seed $(fuzz_seed "$t")" + fi + echo "TARGET $t $status ($why)" + echo "$status" > "build/fuzz/$t.result" +} + +lane_fuzz() { + local secs=300 t gs_before="" gs_after="" git_ok=0 pass n + [[ "$SMOKE" == 1 ]] && secs=20 + unset DEBUGINFOD_URLS # the symbolizer must not reach the network + command -v "$CHECK_CLANG" > /dev/null || die_lane fuzz "$CHECK_CLANG not found" + if gs_before="$(git status --short 2>/dev/null)"; then git_ok=1; fi + mkdir -p build/fuzz build/logs/fuzz + rm -f build/fuzz/*.result + n="$(wc -w <<< "$FUZZ_TARGETS")" + tr ' ' '\n' <<< "$FUZZ_TARGETS" | sed "s/\$/ $secs/" | xargs -P 4 -L 1 "$SELF" __fuzz || true + pass="$(cat build/fuzz/*.result 2>/dev/null | grep -c '^PASS$' || true)" + [[ $(( n - pass )) -eq 0 ]] || die_lane fuzz "$(( n - pass )) of $n targets failed" + if [[ $git_ok == 1 ]]; then + gs_after="$(git status --short 2>/dev/null || true)" + [[ "$gs_before" == "$gs_after" ]] || die_lane fuzz "git status changed" + fi + echo "LANE fuzz PASS: $n targets, $secs s each" +} + +# ---- tsan lane (S6-R2) ---- +# Builds the suite with -fsanitize=thread -O1 -g -DFENG_MATRIX_PARALLEL into build/tsan//test with the first GCC and +# $CHECK_CLANG (--smoke: $CHECK_CLANG only) and runs TSAN_SPEC in each; logs in build/logs/tsan/. Fails if a build +# fails, a run matches no test, a test fails, or any log holds 'WARNING: ThreadSanitizer'. +TSAN_FLAGS="-std=c++20 -fsanitize=thread -O1 -g -DFENG_MATRIX_PARALLEL" +TSAN_SPEC="[S6-R1],[S6-R3]" +lane_tsan() { + export TSAN_OPTIONS="${TSAN_OPTIONS:-halt_on_error=1}" + unset DEBUGINFOD_URLS # the symbolizer must not reach the network + local compilers cxx bin log n=0 fail=0 nomatch=0 rc + if [[ "$SMOKE" == 1 ]]; then compilers="$CHECK_CLANG"; else compilers="$(first_gcc) $CHECK_CLANG"; fi + mkdir -p build/tsan build/logs/tsan + rm -f build/logs/tsan/*.log + for cxx in $compilers; do + n=$((n + 1)) + bin="build/tsan/$cxx/test" + log="build/logs/tsan/$cxx.run.log" + if ! build_suite "$cxx" "$TSAN_FLAGS" "$bin" "build/logs/tsan/$cxx.build.log"; then + echo "CONFIG $cxx c++20 tsan FAIL (build, see build/logs/tsan/$cxx.build.log)"; fail=$((fail + 1)); continue + fi + rc=0 + run_suite "$bin" "$log" "$TSAN_SPEC" || rc=$? + if grep -q 'WARNING: ThreadSanitizer' "$log"; then rc=3; fi + case $rc in + 0) echo "CONFIG $cxx c++20 tsan PASS" ;; + 2) echo "CONFIG $cxx c++20 tsan FAIL (no test matched, see $log)"; nomatch=$((nomatch + 1)) ;; + 3) echo "CONFIG $cxx c++20 tsan FAIL (ThreadSanitizer warning, see $log)"; fail=$((fail + 1)) ;; + *) echo "CONFIG $cxx c++20 tsan FAIL (tests, see $log)"; fail=$((fail + 1)) ;; + esac + done + if grep -lq 'WARNING: ThreadSanitizer' build/logs/tsan/*.log 2>/dev/null; then + die_lane tsan "ThreadSanitizer warning in build/logs/tsan/" + fi + if [[ $nomatch -ne 0 ]]; then die_lane tsan "$nomatch of $n runs matched no test"; fi + if [[ $fail -ne 0 ]]; then die_lane tsan "$fail of $n runs failed"; fi + echo "LANE tsan PASS: $n runs" +} + +# ---- oracle lane (S7-R6) ---- +# Builds the suite with the first GCC at -std=c++20 -O2 -DFENG_MATRIX_PARALLEL into build/oracle/test, runs "[oracle][]" +# (the committed fixtures under tests/fixtures//, each test printing `ORACLE FIXTURES `), then, when +# /usr/include/eigen3/Eigen/Core and tests/oracle/_eigen.cc both exist, builds that program under build/oracle/ +# with -isystem /usr/include/eigen3 and runs it (it must exit 0 and print `EIGEN PASS`); otherwise prints +# `oracle: Eigen absent`. Logs under build/logs/oracle/; --smoke is the same run; `git status --short` must not change. +ORACLE_FLAGS="-std=c++20 -O2 -DFENG_MATRIX_PARALLEL" +ORACLE_EIGEN_FLAGS="-std=c++20 -O2 -isystem /usr/include/eigen3" +lane_oracle() { + [[ ${#ARGS[@]} -ge 1 && -n "${ARGS[0]}" ]] || die_lane oracle "usage: tools/check.sh oracle " + local id="${ARGS[0]}" lid logd=build/logs/oracle bin=build/oracle/test rc=0 n eigen=absent + local gs_before="" gs_after="" git_ok=0 + lid="$(tr '[:upper:]' '[:lower:]' <<< "$id")" + mkdir -p build/oracle "$logd" + if gs_before="$(git status --short 2>/dev/null)"; then git_ok=1; fi + build_suite "$(first_gcc)" "$ORACLE_FLAGS" "$bin" "$logd/build.log" || die_lane oracle "build failed, see $logd/build.log" + run_suite "$bin" "$logd/run.log" "[oracle][$id]" || rc=$? + [[ $rc != 2 ]] || die_lane oracle "no [oracle][$id] test ran, see $logd/run.log" + [[ $rc == 0 ]] || die_lane oracle "fixture tests failed, see $logd/run.log" + n="$(sed -n "s/^ORACLE $id FIXTURES \([0-9][0-9]*\)\$/\1/p" "$logd/run.log" | awk '{ s += $1 } END { print s + 0 }')" + grep -q "^ORACLE $id FIXTURES " "$logd/run.log" || die_lane oracle "no 'ORACLE $id FIXTURES' line, see $logd/run.log" + echo "fixtures: $n checked ($logd/run.log)" + local src="tests/oracle/${lid}_eigen.cc" ebin="build/oracle/${lid}_eigen" + if [[ -f /usr/include/eigen3/Eigen/Core && -f "$src" ]]; then + rm -f "$ebin" + # shellcheck disable=SC2086 + "$(first_gcc)" $ORACLE_EIGEN_FLAGS -pthread "$src" -o "$ebin" > "$logd/eigen.build.log" 2>&1 \ + || die_lane oracle "Eigen program build failed, see $logd/eigen.build.log" + rc=0 + "$ebin" > "$logd/eigen.run.log" 2>&1 || rc=$? + [[ $rc == 0 ]] || die_lane oracle "Eigen comparison exited $rc, see $logd/eigen.run.log" + grep -q '^EIGEN PASS ' "$logd/eigen.run.log" || die_lane oracle "no EIGEN PASS line, see $logd/eigen.run.log" + head -n 1 "$logd/eigen.run.log" + grep '^EIGEN PASS ' "$logd/eigen.run.log" + eigen=ran + else + echo "oracle: Eigen absent" + fi + if [[ $git_ok == 1 ]]; then + gs_after="$(git status --short 2>/dev/null || true)" + [[ "$gs_before" == "$gs_after" ]] || die_lane oracle "git status changed" + fi + echo "LANE oracle PASS: $n fixtures, eigen $eigen" +} + +# ---- bench lane (S8-R4, D-031) ---- +# Builds tools/bench/.cc with the first GCC at -std=c++20 -O2 -pthread (no -march=native, -Ofast or +# fast-math, D-006) into build/bench//, logs under build/logs/bench/, runs it (passing --smoke through) and +# judges its output. Workload `fft`: passes when the `BENCH fft 256x256` median is under 1 s and both `RATIO` values +# are below 6.75 (the medians are per transform; fft.cc batches repeats so each sample lasts at least ~20 ms); the +# BENCH, RATIO and INFO lines are echoed. `git status --short` must not change. +BENCH_FLAGS="-std=c++20 -O2 -pthread" + +# `bench compare` (S10-R1, S10-R3, S10-R4, D-008): the baseline is matrix.hpp at the parent of the oldest commit whose +# subject is exactly `S10: spec and tasks`, extracted with git show to build/bench/compare/base/matrix.hpp. bench/bench.cc +# is built by the first GCC with exactly $BENCH_FLAGS -DBENCH_FLAGS="..." twice: base/bench (-I the extracted header) +# and head/bench (-I the repo root); bench/ holds no matrix.hpp, so each build sees only its own header. The parallel +# pair (S10-T6) base-par/bench and head-par/bench is the same two builds with $BENCH_FLAGS -DFENG_MATRIX_PARALLEL (the +# define is part of their recorded INFO flags) against the same two headers; its workloads are named par/. Logs go +# to build/logs/bench/. tools/bench_compare.py runs per workload the processes ABBA BAAB ABBA BAAB (A base, B head; +# smoke A B B A), each with 5 samples (smoke 3), pools 40 (smoke 6) samples per build (serial pinned, par/ unpinned), +# prints the INFO lines and the one table and judges bench/kept.txt; its LANE line is printed last, after the +# `git status --short` check. +lane_bench_compare() { + local d=build/bench/compare logd=build/logs/bench gs_before="" gs_after="" git_ok=0 rc=0 start cxx line + [[ ! -e bench/matrix.hpp ]] || die_lane bench "bench/matrix.hpp exists; the base build would include it" + if gs_before="$(git status --short 2>/dev/null)"; then git_ok=1; fi + start="$(git log --format='%H %s' --grep='^S10: spec and tasks$' \ + | awk '{ h = $1; $1 = ""; if ( substr( $0, 2 ) == "S10: spec and tasks" ) last = h } END { print last }')" + [[ -n "$start" ]] || die_lane bench "no commit with subject 'S10: spec and tasks'" + mkdir -p "$d/base" "$d/head" "$d/base-par" "$d/head-par" "$logd" + rm -f "$d/base/bench" "$d/head/bench" "$d/base-par/bench" "$d/head-par/bench" + git show "$start^:matrix.hpp" > "$d/base/matrix.hpp" || die_lane bench "git show $start^:matrix.hpp failed" + echo "INFO baseline $(git rev-parse --short "$start^") (parent of $(git rev-parse --short "$start"))" + cxx="$(first_gcc)" + # shellcheck disable=SC2086 + "$cxx" $BENCH_FLAGS -DBENCH_FLAGS="\"$BENCH_FLAGS\"" -I "$d/base" bench/bench.cc -o "$d/base/bench" \ + > "$logd/compare.base.build.log" 2>&1 & + local pb=$! + # shellcheck disable=SC2086 + "$cxx" $BENCH_FLAGS -DBENCH_FLAGS="\"$BENCH_FLAGS\"" -I . bench/bench.cc -o "$d/head/bench" \ + > "$logd/compare.head.build.log" 2>&1 & + local ph=$! + local par_flags="$BENCH_FLAGS -DFENG_MATRIX_PARALLEL" + # shellcheck disable=SC2086 + "$cxx" $par_flags -DBENCH_FLAGS="\"$par_flags\"" -I "$d/base" bench/bench.cc -o "$d/base-par/bench" \ + > "$logd/compare.base-par.build.log" 2>&1 & + local pbp=$! + # shellcheck disable=SC2086 + "$cxx" $par_flags -DBENCH_FLAGS="\"$par_flags\"" -I . bench/bench.cc -o "$d/head-par/bench" \ + > "$logd/compare.head-par.build.log" 2>&1 & + local php=$! + wait "$pb" || die_lane bench "base build failed, see $logd/compare.base.build.log" + wait "$ph" || die_lane bench "head build failed, see $logd/compare.head.build.log" + wait "$pbp" || die_lane bench "base-par build failed, see $logd/compare.base-par.build.log" + wait "$php" || die_lane bench "head-par build failed, see $logd/compare.head-par.build.log" + echo "INFO build $cxx $BENCH_FLAGS" + echo "INFO build-par $cxx $par_flags" + local smoke=() + [[ "$SMOKE" == 1 ]] && smoke=(--smoke) + rm -f "$d/rc" + { python3 tools/bench_compare.py --base "$d/base/bench" --head "$d/head/bench" \ + --base-par "$d/base-par/bench" --head-par "$d/head-par/bench" --kept bench/kept.txt \ + --log "$logd/compare.run.log" "${smoke[@]}" 2>&1 || echo "$?" > "$d/rc"; } \ + | tee "$logd/compare.log" | { grep --line-buffered -v '^LANE ' || true; } + [[ -f "$d/rc" ]] && rc="$(cat "$d/rc")" + line="$(grep '^LANE bench ' "$logd/compare.log" | tail -n 1 || true)" + if [[ $git_ok == 1 ]]; then + gs_after="$(git status --short 2>/dev/null || true)" + [[ "$gs_before" == "$gs_after" ]] || die_lane bench "git status changed" + fi + [[ -n "$line" ]] || die_lane bench "tools/bench_compare.py exited $rc without a LANE line, see $logd/compare.log" + echo "$line" + [[ $rc -eq 0 && "$line" == "LANE bench PASS"* ]] || exit 1 +} + +lane_bench() { + [[ ${#ARGS[@]} -ge 1 && -n "${ARGS[0]}" ]] || die_lane bench "usage: tools/check.sh bench |compare [--smoke]" + if [[ "${ARGS[0]}" == compare ]]; then lane_bench_compare; return; fi + local w="${ARGS[0]}" logd=build/logs/bench gs_before="" gs_after="" git_ok=0 rc=0 out + local src="tools/bench/${ARGS[0]}.cc" bin="build/bench/${ARGS[0]}/${ARGS[0]}" + [[ "$w" =~ ^[A-Za-z0-9_-]+$ && -f "$src" ]] || die_lane bench "unknown workload '$w' (no tools/bench/$w.cc)" + case "$w" in + fft) ;; + *) die_lane bench "unknown workload '$w' (no pass rule)" ;; + esac + mkdir -p "build/bench/$w" "$logd" + if gs_before="$(git status --short 2>/dev/null)"; then git_ok=1; fi + rm -f "$bin" + # shellcheck disable=SC2086 + "$(first_gcc)" $BENCH_FLAGS -I . "$src" -o "$bin" > "$logd/$w.build.log" 2>&1 \ + || die_lane bench "build failed, see $logd/$w.build.log" + local smoke=() + [[ "$SMOKE" == 1 ]] && smoke=(--smoke) + "$bin" "${smoke[@]}" > "$logd/$w.run.log" 2>&1 || rc=$? + [[ $rc -eq 0 ]] || die_lane bench "$w exited $rc, see $logd/$w.run.log" + grep -E '^(BENCH|RATIO|INFO) ' "$logd/$w.run.log" || true + if [[ $git_ok == 1 ]]; then + gs_after="$(git status --short 2>/dev/null || true)" + [[ "$gs_before" == "$gs_after" ]] || die_lane bench "git status changed" + fi + local med r1 r2 + med="$(sed -n 's/^BENCH fft 256x256 median \([0-9.eE+-]*\) s$/\1/p' "$logd/$w.run.log" | tail -n 1)" + r1="$(sed -n 's|^RATIO 256/128 \([0-9.eE+-]*\)$|\1|p' "$logd/$w.run.log" | tail -n 1)" + r2="$(sed -n 's|^RATIO 512/256 \([0-9.eE+-]*\)$|\1|p' "$logd/$w.run.log" | tail -n 1)" + [[ -n "$med" && -n "$r1" && -n "$r2" ]] || die_lane bench "missing BENCH 256x256 or RATIO line, see $logd/$w.run.log" + awk -v m="$med" 'BEGIN { exit !(m + 0 < 1) }' || die_lane bench "fft 256x256 median $med s is not under 1 s" + awk -v a="$r1" 'BEGIN { exit !(a + 0 < 6.75) }' || die_lane bench "RATIO 256/128 $r1 is not below 6.75" + awk -v a="$r2" 'BEGIN { exit !(a + 0 < 6.75) }' || die_lane bench "RATIO 512/256 $r2 is not below 6.75" + echo "LANE bench PASS: fft 256x256 median $med s, ratios $r1 $r2" +} + +# ---- config lane (S9-R3, D-033) ---- +# Builds tests/config/tu_a.cc and tu_b.cc into build/config/[-lto]// (logs build/logs/config/) with the +# first GCC and $CHECK_CLANG at -std=c++20 -O2, in four pairs per compiler: same-new (both -DFENG_MATRIX_PARALLEL) +# and old-vs-new (-DPARALLEL vs -DFENG_MATRIX_PARALLEL) must link and run with exit 0; mixed-parallel and +# mixed-checked-iterators must fail to link with a log naming feng_matrix_configuration_mismatch. The first GCC with +# -flto adds same-new and mixed-parallel. One line per pair; --smoke is the same run. Not part of `all`. +CONFIG_FLAGS="-std=c++20 -O2 -pthread" +# config_pair ; prints one PAIR line, returns 1 on failure. +config_pair() { + local cxx="$1" lto="$2" pair="$3" want="$4" fa="$5" fb="$6" extra="" tag="$1" + [[ "$lto" == lto ]] && { extra="-flto"; tag="$1-lto"; } + local d="build/config/$tag/$pair" log="build/logs/config/$tag-$pair.log" rc=0 got + mkdir -p "$d" build/logs/config + rm -f "$d"/*.o "$d/prog" + # shellcheck disable=SC2086 + "$cxx" $CONFIG_FLAGS $extra $fa -c tests/config/tu_a.cc -o "$d/tu_a.o" > "$log" 2>&1 || rc=1 + # shellcheck disable=SC2086 + [[ $rc == 0 ]] && { "$cxx" $CONFIG_FLAGS $extra $fb -c tests/config/tu_b.cc -o "$d/tu_b.o" >> "$log" 2>&1 || rc=1; } + if [[ $rc != 0 ]]; then echo "PAIR $tag $pair FAIL (compile, see $log)"; return 1; fi + # shellcheck disable=SC2086 + if "$cxx" $CONFIG_FLAGS $extra "$d/tu_a.o" "$d/tu_b.o" -o "$d/prog" >> "$log" 2>&1; then got=link; else got=nolink; fi + if [[ "$want" == link ]]; then + [[ $got == link ]] || { echo "PAIR $tag $pair FAIL (link failed, see $log)"; return 1; } + "$d/prog" >> "$log" 2>&1 || rc=$? + [[ $rc == 0 ]] || { echo "PAIR $tag $pair FAIL (program exited $rc, see $log)"; return 1; } + echo "PAIR $tag $pair PASS (links, runs, exit 0)" + else + [[ $got == nolink ]] || { echo "PAIR $tag $pair FAIL (mixed configuration linked, see $log)"; return 1; } + grep -q feng_matrix_configuration_mismatch "$log" \ + || { echo "PAIR $tag $pair FAIL (link failed without feng_matrix_configuration_mismatch, see $log)"; return 1; } + echo "PAIR $tag $pair PASS (link fails: feng_matrix_configuration_mismatch)" + fi +} +lane_config() { + local cxx g n=0 bad=0 + g="$(first_gcc)" + command -v "$g" > /dev/null || die_lane config "$g not found" + command -v "$CHECK_CLANG" > /dev/null || die_lane config "$CHECK_CLANG not found" + local p1="-DFENG_MATRIX_PARALLEL -DCONFIG_EXPECT_PARALLEL=1" p0="-DCONFIG_EXPECT_PARALLEL=0" + for cxx in "$g" "$CHECK_CLANG"; do + n=$((n + 4)) + config_pair "$cxx" plain same-new link "$p1" "-DFENG_MATRIX_PARALLEL" || bad=$((bad + 1)) + config_pair "$cxx" plain old-vs-new link "-DPARALLEL -DCONFIG_EXPECT_PARALLEL=1" "-DFENG_MATRIX_PARALLEL" || bad=$((bad + 1)) + config_pair "$cxx" plain mixed-parallel nolink "$p1" "" || bad=$((bad + 1)) + config_pair "$cxx" plain mixed-checked-iterators nolink "$p0 -DFENG_MATRIX_CHECKED_ITERATORS" "" || bad=$((bad + 1)) + done + n=$((n + 2)) + config_pair "$g" lto same-new link "$p1" "-DFENG_MATRIX_PARALLEL" || bad=$((bad + 1)) + config_pair "$g" lto mixed-parallel nolink "$p1" "" || bad=$((bad + 1)) + [[ $bad -eq 0 ]] || die_lane config "$bad of $n pairs failed" + echo "LANE config PASS: $n pairs" +} + +# ---- all lane (S1-R5) ---- +# Runs every regression lane in sequence even after a failure; full output in build/logs/all/.log. +lane_all() { + local log=build/logs/all lane line failed=() smoke=() gs_before="" gs_after="" git_ok=0 + mkdir -p "$log" + [[ "$SMOKE" == 1 ]] && smoke=(--smoke) + if gs_before="$(git status --short 2>/dev/null)"; then git_ok=1; fi + for lane in gcc clang sanitize warnings api examples make ci "docs S1"; do + local name="${lane%% *}" rc=0 + # shellcheck disable=SC2086 + case "$name" in + make|ci|docs) "$SELF" $lane > "$log/$name.log" 2>&1 || rc=$? ;; + *) "$SELF" $lane "${smoke[@]}" > "$log/$name.log" 2>&1 || rc=$? ;; + esac + line="$(grep '^LANE ' "$log/$name.log" | tail -n 1)" + [[ -n "$line" ]] || line="LANE $name FAIL: no LANE line (exit $rc), see $log/$name.log" + echo "$line" + if [[ $rc -ne 0 || "$line" != "LANE $name PASS"* ]]; then failed+=("$name"); fi + done + if [[ $git_ok == 1 ]]; then + gs_after="$(git status --short 2>/dev/null || true)" + [[ "$gs_before" == "$gs_after" ]] || failed+=("git-status") + fi + [[ ${#failed[@]} -eq 0 ]] || die_lane all "${failed[*]}" + echo "LANE all PASS" +} + +# ---- dispatch ---- +if [[ "${1:-}" == __config ]]; then shift; config_job "$@"; exit 0; fi +if [[ "${1:-}" == __sanbuild ]]; then shift; san_build_job "$@"; exit 0; fi +if [[ "${1:-}" == __warn ]]; then shift; warn_job "$@"; exit 0; fi +if [[ "${1:-}" == __api ]]; then shift; api_job "$@"; exit 0; fi +if [[ "${1:-}" == __apilink ]]; then shift; api_link_job "$@"; exit 0; fi +if [[ "${1:-}" == __fuzz ]]; then shift; fuzz_job "$@"; exit 0; fi + +LANE="${1:-}" +[[ -n "$LANE" ]] || { echo "usage: tools/check.sh [args] [--smoke]"; echo "LANE none FAIL: no lane given"; exit 2; } +shift +SMOKE=0 +ARGS=() +for a in "$@"; do + if [[ "$a" == --smoke ]]; then SMOKE=1; else ARGS+=("$a"); fi +done + +case "$LANE" in + gcc) lane_gcc ;; + clang) lane_clang ;; + sanitize) lane_sanitize ;; + tagged) lane_tagged ;; + warnings) lane_warnings ;; + api) lane_api ;; + examples) lane_examples ;; + make) lane_make ;; + ci) lane_ci ;; + docs) lane_docs ;; + compile-fail) lane_compile_fail ;; + fuzz) lane_fuzz ;; + tsan) lane_tsan ;; + oracle) lane_oracle ;; + bench) lane_bench ;; + config) lane_config ;; + all) lane_all ;; + *) die_lane "$LANE" "unknown lane" ;; +esac diff --git a/tools/const_returns.py b/tools/const_returns.py new file mode 100644 index 0000000..c5d0591 --- /dev/null +++ b/tools/const_returns.py @@ -0,0 +1,244 @@ +#!/usr/bin/env python3 +"""tools/const_returns.py [header] (S9-R1, D-034) + +Counts the functions declared in matrix.hpp whose return type is a top-level const value (`T const f()`, +`const T f()`, `auto const f()`, `auto f() -> T const`, `T* const f()`); references and pointers to const +(`T const&`, `T const*`) do not count. Route 1, the clang AST: it runs +`clang++ -fsyntax-only -Xclang -ast-dump=json -Xclang -ast-dump-filter=feng` on a TU that includes the +header, once per configuration in CONFIGS (-std=c++20; -std=c++26 with FENG_MATRIX_CHECKED_ITERATORS and +FENG_MATRIX_PARALLEL, so the __cpp_lib_mdspan / submdspan / expected and checked-iterator blocks are parsed), +walks FunctionDecl / CXXMethodDecl / CXXConversionDecl nodes located in the header (multi-line declarations are +one node), reads the return type from the function type's qualType and unions the hits. Route 2, text: OpenCV is +not installed, so the `#ifdef FENG_MATRIX_OPENCV` regions are scanned as text for a declaration line with a +top-level const before the function name (`T const f(`, `const T f(`) or after a trailing `->`. Prints one +line per hit, `matrix.hpp:: : `, then `CONST_RETURNS `. Exit 0 when the scan ran +(whatever n is), 2 when clang++ fails. Environment: CHECK_CLANG (default clang++). Stdlib only. +""" +import json +import os +import re +import subprocess +import sys +import tempfile + +FUNC_KINDS = {"FunctionDecl", "CXXMethodDecl", "CXXConversionDecl"} +CONFIGS = [["-std=c++20"], + ["-std=c++26", "-DFENG_MATRIX_CHECKED_ITERATORS", "-DFENG_MATRIX_PARALLEL"]] +GATE = "FENG_MATRIX_OPENCV" +# A declaration line: optional specifiers, then a return type holding a top-level const, then `name(`. +TYPE = r"[A-Za-z_][\w:]*(?:\s*<[^;{}]*>)?(?:\s*\*)*" +SPEC = r"(?:(?:template\s*<[^;{}]*>|\[\[[^\]]*\]\]|static|inline|constexpr|consteval|virtual|friend|explicit)\s+)*" +DECL_CONST = [re.compile(r"^\s*" + SPEC + r"const\s+" + TYPE + r"\s+(?P[A-Za-z_]\w*|operator\s*\S+?)\s*\("), + re.compile(r"^\s*" + SPEC + TYPE + r"\s+const\s+(?P[A-Za-z_]\w*|operator\s*\S+?)\s*\("), + re.compile(r"^\s*" + SPEC + r"auto\s+(?P[A-Za-z_]\w*)\s*\(.*\)[^;{]*->\s*[^;{&]*\bconst\s*(?:\{|$)")] + + +def split_function_type(qual): + """Return type text of a clang function qualType such as `const T (int) const noexcept` or + `auto () -> const T`; None when it is not recognisable.""" + depth_angle = depth_paren = 0 + start = -1 + for i, ch in enumerate(qual): + if ch == "<": + depth_angle += 1 + elif ch == ">": + depth_angle -= 1 + elif ch == "(": + # The parameter list is the first top-level '(' preceded by a space (decltype( is not). + if depth_angle == 0 and depth_paren == 0 and i > 0 and qual[i - 1] == " ": + start = i + break + depth_paren += 1 + elif ch == ")": + depth_paren -= 1 + if start < 0: + return None + ret = qual[:start].strip() + depth = 0 + end = -1 + for j in range(start, len(qual)): + if qual[j] == "(": + depth += 1 + elif qual[j] == ")": + depth -= 1 + if depth == 0: + end = j + break + if end >= 0: + rest = qual[end + 1:] + arrow = rest.find("->") + if arrow >= 0: + ret = rest[arrow + 2:].strip() + return ret + + +def top_level_const(ret): + """True when the type text is const-qualified at top level (not a reference, not a pointer to const).""" + t = ret.strip() + # Drop template arguments so `const std::vector ` and `foo` read correctly. + flat, depth = [], 0 + for ch in t: + if ch in "<(": + depth += 1 + elif ch in ">)": + depth -= 1 + elif depth == 0: + flat.append(ch) + f = "".join(flat).strip() + if f.endswith("&"): + return False + if f.endswith("const") and (f == "const" or not f[-6].isalnum() and f[-6] != "_"): + return True # `T *const`, `T const` + if "*" in f: + return False # pointer whose pointee is const (`const T *`) + words = f.replace("*", " ").split() + return "const" in words + + +class Walker: + def __init__(self, header): + self.header = os.path.realpath(header) + self.file = None + self.line = None + self.hits = {} + + def track(self, loc): + """Clang's JSON omits `file` and `line` when unchanged since the last printed location (document order).""" + if not isinstance(loc, dict): + return + for key in ("spellingLoc", "expansionLoc"): + if key in loc: + self.track(loc[key]) + if "file" in loc: + self.file = loc["file"] + if "line" in loc: + self.line = loc["line"] + + def loc_of(self, loc): + if "expansionLoc" in loc: + self.track(loc.get("spellingLoc", {})) + self.track(loc["expansionLoc"]) + else: + self.track(loc) + return self.file, self.line + + def walk(self, node): + if isinstance(node, list): + for x in node: + self.walk(x) + return + if not isinstance(node, dict): + return + kind = node.get("kind") + here = None + for key, val in node.items(): + if key == "loc": + here = self.loc_of(val) + elif key == "range": + self.track(val.get("begin", {})) + self.track(val.get("end", {})) + elif isinstance(val, (dict, list)) and key != "type": + self.walk(val) + if kind in FUNC_KINDS: + self.check(node, here) + + def check(self, node, here): + if node.get("isImplicit") or not here or here[0] is None: + return + if os.path.realpath(here[0]) != self.header: + return + qual = node.get("type", {}).get("qualType", "") + ret = split_function_type(qual) + if ret is None or not top_level_const(ret): + return + self.hits.setdefault((here[1], node.get("name", "?")), ret) + + +def gated_regions(lines, macro): + """1-based line numbers inside `#ifdef macro` / `#if defined( macro )` blocks, nested #if counted.""" + opener = re.compile(r"^\s*#\s*(?:ifdef\s+%s\b|if\s+defined\s*\(?\s*%s\b)" % (macro, macro)) + inside, depth = set(), 0 + for no, text in enumerate(lines, 1): + if depth == 0: + if opener.match(text) and "!" not in text: + depth = 1 + continue + if re.match(r"^\s*#\s*if", text): + depth += 1 + elif re.match(r"^\s*#\s*endif", text): + depth -= 1 + continue + if depth > 0: + inside.add(no) + return inside + + +def text_scan(header, macro): + """Hits {(line, name): line text} for const-returning declarations inside the macro's regions.""" + with open(header) as fh: + lines = fh.read().split("\n") + hits = {} + for no in sorted(gated_regions(lines, macro)): + text = lines[no - 1].split("//")[0] + if text.rstrip().endswith(";"): + continue # a local variable such as `T const x( a );`, not a definition + for rx in DECL_CONST: + m = rx.match(text) + if m: + hits[(no, m.group("name"))] = text.strip() + break + return hits + + +def ast_scan(clang, header, flags): + """Hits {(line, name): return type} of one clang AST pass; None when clang fails.""" + with tempfile.TemporaryDirectory() as tmp: + tu = os.path.join(tmp, "const_returns.cc") + with open(tu, "w") as fh: + fh.write('#include "%s"\n' % os.path.realpath(header)) + proc = subprocess.run([clang] + flags + ["-fsyntax-only", "-Xclang", "-ast-dump=json", + "-Xclang", "-ast-dump-filter=feng", tu], + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + if proc.returncode != 0: + sys.stderr.write(proc.stderr) + return None + walker = Walker(header) + dec = json.JSONDecoder() + text, pos = proc.stdout, 0 + while True: + while pos < len(text) and text[pos] != "{": + pos = text.find("\n", pos) + if pos < 0: + pos = len(text) + break + pos += 1 + if pos >= len(text): + break + obj, pos = dec.raw_decode(text, pos) + walker.walk(obj) + return walker.hits + + +def main(): + root = os.path.dirname(os.path.dirname(os.path.realpath(__file__))) + header = sys.argv[1] if len(sys.argv) > 1 else os.path.join(root, "matrix.hpp") + clang = os.environ.get("CHECK_CLANG", "clang++") + hits = {} + for flags in CONFIGS: + found = ast_scan(clang, header, flags) + if found is None: + print("CONST_RETURNS error: %s %s failed" % (clang, " ".join(flags))) + return 2 + for key, ret in found.items(): + hits.setdefault(key, ret) + for key, ret in text_scan(header, GATE).items(): + hits.setdefault(key, ret) + base = os.path.basename(header) + for (line, name), ret in sorted(hits.items()): + print("%s:%s: %s: %s" % (base, line, name, ret)) + print("CONST_RETURNS %d" % len(hits)) + return 0 + + +if __name__ == "__main__": + sys.exit(main())