From 5e3a8da84c04d4c48596066b2d95024716dcaa39 Mon Sep 17 00:00:00 2001 From: Arthur031221 Date: Thu, 1 Oct 2026 22:32:23 +0800 Subject: [PATCH 1/3] Return a finite value for single-element windows, linspace and logspace --- include/matx/generators/bartlett.h | 4 +- include/matx/generators/blackman.h | 4 +- include/matx/generators/flattop.h | 2 + include/matx/generators/hamming.h | 4 +- include/matx/generators/hanning.h | 4 +- include/matx/generators/linspace.h | 2 +- include/matx/generators/logspace.h | 8 ++-- test/00_operators/GeneratorTests.cu | 61 +++++++++++++++++++++++++++++ 8 files changed, 76 insertions(+), 13 deletions(-) diff --git a/include/matx/generators/bartlett.h b/include/matx/generators/bartlett.h index d42dcaea7..7c551d802 100644 --- a/include/matx/generators/bartlett.h +++ b/include/matx/generators/bartlett.h @@ -72,7 +72,7 @@ namespace matx " template \n" + " __MATX_INLINE__ __MATX_DEVICE__ auto operator()(index_t i) const\n" + " {\n" + - " return detail::ApplyGeneratorVecFunc([](index_t idx) { return 1 - cuda::std::abs(((2*T(idx))/(T(size_ - 1))) - 1); }, i);\n" + + " return detail::ApplyGeneratorVecFunc([](index_t idx) { if (size_ == 1) { return T(1); } return 1 - cuda::std::abs(((2*T(idx))/(T(size_ - 1))) - 1); }, i);\n" + " }\n" + " static __MATX_INLINE__ constexpr __MATX_DEVICE__ int32_t Rank() { return 1; }\n" + " constexpr __MATX_INLINE__ __MATX_DEVICE__ index_t Size([[maybe_unused]] int dim) const { return size_; }\n" + @@ -92,7 +92,7 @@ namespace matx template inline __MATX_HOST__ __MATX_DEVICE__ auto operator()(index_t i) const { - return detail::ApplyGeneratorVecFunc([this](index_t idx) { return 1 - cuda::std::abs(((2*T(idx))/(T(size_ - 1))) - 1); }, i); + return detail::ApplyGeneratorVecFunc([this](index_t idx) { if (size_ == 1) { return T(1); } return 1 - cuda::std::abs(((2*T(idx))/(T(size_ - 1))) - 1); }, i); } inline __MATX_HOST__ __MATX_DEVICE__ auto operator()(index_t i) const diff --git a/include/matx/generators/blackman.h b/include/matx/generators/blackman.h index a76740baf..3bec58745 100644 --- a/include/matx/generators/blackman.h +++ b/include/matx/generators/blackman.h @@ -72,7 +72,7 @@ namespace matx " template \n" + " __MATX_INLINE__ __MATX_DEVICE__ auto operator()(index_t i) const\n" + " {\n" + - " return detail::ApplyGeneratorVecFunc([](index_t idx) { return T(.42) - T(.5) * cuda::std::cos(T(2 * cuda::std::numbers::pi) * T(idx) / T(size_ - 1)) + T(.08) * cuda::std::cos(T(4 * cuda::std::numbers::pi) * T(idx) / T(size_ - 1)); }, i);\n" + + " return detail::ApplyGeneratorVecFunc([](index_t idx) { if (size_ == 1) { return T(1); } return T(.42) - T(.5) * cuda::std::cos(T(2 * cuda::std::numbers::pi) * T(idx) / T(size_ - 1)) + T(.08) * cuda::std::cos(T(4 * cuda::std::numbers::pi) * T(idx) / T(size_ - 1)); }, i);\n" + " }\n" + " static __MATX_INLINE__ constexpr __MATX_DEVICE__ int32_t Rank() { return 1; }\n" + " constexpr __MATX_INLINE__ __MATX_DEVICE__ index_t Size([[maybe_unused]] int dim) const { return size_; }\n" + @@ -92,7 +92,7 @@ namespace matx template __MATX_INLINE__ __MATX_HOST__ __MATX_DEVICE__ auto operator()(index_t i) const { - return detail::ApplyGeneratorVecFunc([this](index_t idx) { return T(.42) - T(.5) * cuda::std::cos(T(2 * M_PI) * T(idx) / T(size_ - 1)) + return detail::ApplyGeneratorVecFunc([this](index_t idx) { if (size_ == 1) { return T(1); } return T(.42) - T(.5) * cuda::std::cos(T(2 * M_PI) * T(idx) / T(size_ - 1)) + T(.08) * cuda::std::cos(T(4 * M_PI) * T(idx) / T(size_ - 1)); }, i); } diff --git a/include/matx/generators/flattop.h b/include/matx/generators/flattop.h index 84bba59d7..f1d6c021f 100644 --- a/include/matx/generators/flattop.h +++ b/include/matx/generators/flattop.h @@ -85,6 +85,7 @@ namespace matx " __MATX_INLINE__ __MATX_DEVICE__ auto operator()(index_t i) const\n" + " {\n" + " return detail::ApplyGeneratorVecFunc([](index_t idx) {\n" + + " if (size_ == 1) { return T(1); }\n" + " static constexpr T tmp_pi = cuda::std::numbers::pi;\n" + " return a0 - a1 * cuda::std::cos(static_cast(2) * tmp_pi * idx / (size_ - 1)) + a2 * cuda::std::cos(static_cast(4) * tmp_pi * idx / (size_ - 1)) - a3 * cuda::std::cos(static_cast(6) * tmp_pi * idx / (size_ - 1)) + a4 * cuda::std::cos(static_cast(8) * tmp_pi * idx / (size_ - 1));\n" + " }, i);\n" + @@ -108,6 +109,7 @@ namespace matx inline __MATX_HOST__ __MATX_DEVICE__ auto operator()(index_t i) const { return detail::ApplyGeneratorVecFunc([this](index_t idx) { + if (size_ == 1) { return T(1); } static constexpr T tmp_pi = cuda::std::numbers::pi; return a0 - a1 * cuda::std::cos(static_cast(2)*tmp_pi*idx / (size_ - 1)) diff --git a/include/matx/generators/hamming.h b/include/matx/generators/hamming.h index 653d2705e..4c1002860 100644 --- a/include/matx/generators/hamming.h +++ b/include/matx/generators/hamming.h @@ -73,7 +73,7 @@ namespace matx " template \n" + " __MATX_INLINE__ __MATX_DEVICE__ auto operator()(index_t i) const\n" + " {\n" + - " return detail::ApplyGeneratorVecFunc([](index_t idx) { return T(.54) - T(.46) * cuda::std::cos(T(2 * cuda::std::numbers::pi) * T(idx) / T(size_ - 1)); }, i);\n" + + " return detail::ApplyGeneratorVecFunc([](index_t idx) { if (size_ == 1) { return T(1); } return T(.54) - T(.46) * cuda::std::cos(T(2 * cuda::std::numbers::pi) * T(idx) / T(size_ - 1)); }, i);\n" + " }\n" + " static __MATX_INLINE__ constexpr __MATX_DEVICE__ int32_t Rank() { return 1; }\n" + " constexpr __MATX_INLINE__ __MATX_DEVICE__ index_t Size([[maybe_unused]] int dim) const { return size_; }\n" + @@ -93,7 +93,7 @@ namespace matx template inline __MATX_HOST__ __MATX_DEVICE__ auto operator()(index_t i) const { - return detail::ApplyGeneratorVecFunc([this](index_t idx) { return T(.54) - T(.46) * cuda::std::cos(T(2 * M_PI) * T(idx) / T(size_ - 1)); }, i); + return detail::ApplyGeneratorVecFunc([this](index_t idx) { if (size_ == 1) { return T(1); } return T(.54) - T(.46) * cuda::std::cos(T(2 * M_PI) * T(idx) / T(size_ - 1)); }, i); } inline __MATX_HOST__ __MATX_DEVICE__ auto operator()(index_t i) const diff --git a/include/matx/generators/hanning.h b/include/matx/generators/hanning.h index a7aac5b41..f7244293c 100644 --- a/include/matx/generators/hanning.h +++ b/include/matx/generators/hanning.h @@ -73,7 +73,7 @@ namespace matx " template \n" + " __MATX_INLINE__ __MATX_DEVICE__ auto operator()(index_t i) const\n" + " {\n" + - " return detail::ApplyGeneratorVecFunc([](index_t idx) { return T(0.5) * (1 - cuda::std::cos(T(2 * cuda::std::numbers::pi) * T(idx) / T(size_ - 1))); }, i);\n" + + " return detail::ApplyGeneratorVecFunc([](index_t idx) { if (size_ == 1) { return T(1); } return T(0.5) * (1 - cuda::std::cos(T(2 * cuda::std::numbers::pi) * T(idx) / T(size_ - 1))); }, i);\n" + " }\n" + " static __MATX_INLINE__ constexpr __MATX_DEVICE__ int32_t Rank() { return 1; }\n" + " constexpr __MATX_INLINE__ __MATX_DEVICE__ index_t Size([[maybe_unused]] int dim) const { return size_; }\n" + @@ -93,7 +93,7 @@ namespace matx template inline __MATX_HOST__ __MATX_DEVICE__ auto operator()(index_t i) const { - return detail::ApplyGeneratorVecFunc([this](index_t idx) { return T(0.5) * (1 - cuda::std::cos(T(2 * M_PI) * T(idx) / T(size_ - 1))); }, i); + return detail::ApplyGeneratorVecFunc([this](index_t idx) { if (size_ == 1) { return T(1); } return T(0.5) * (1 - cuda::std::cos(T(2 * M_PI) * T(idx) / T(size_ - 1))); }, i); } inline __MATX_HOST__ __MATX_DEVICE__ T operator()(index_t i) const diff --git a/include/matx/generators/linspace.h b/include/matx/generators/linspace.h index 3c3c13dcc..500a26a84 100644 --- a/include/matx/generators/linspace.h +++ b/include/matx/generators/linspace.h @@ -131,7 +131,7 @@ namespace matx count_ = count; for (int i = 0; i < NUM_RC; ++i) { firsts_[i] = firsts[i]; - steps_[i] = (lasts[i] - firsts[i]) / static_cast(count - 1); + steps_[i] = (count > 1) ? (lasts[i] - firsts[i]) / static_cast(count - 1) : static_cast(0); } } diff --git a/include/matx/generators/logspace.h b/include/matx/generators/logspace.h index 475475d27..45cfc318a 100644 --- a/include/matx/generators/logspace.h +++ b/include/matx/generators/logspace.h @@ -96,20 +96,20 @@ namespace matx { #ifdef __CUDA_ARCH__ if constexpr (is_matx_half_v) { - range_ = Range{first, (last - first) / static_cast(count - 1.0f)}; + range_ = Range{first, (last - first) / static_cast(count > 1 ? count - 1.0f : 1.0f)}; } else { - range_ = Range{first, (last - first) / static_cast(count - 1)}; + range_ = Range{first, (last - first) / static_cast(count > 1 ? count - 1 : 1)}; } #else // Host has no support for most half precision operators/intrinsics if constexpr (is_matx_half_v) { range_ = Range{static_cast(first), (static_cast(last) - static_cast(first)) / - static_cast(count - 1)}; + static_cast(count > 1 ? count - 1 : 1)}; } else { - range_ = Range{first, (last - first) / static_cast(count - 1)}; + range_ = Range{first, (last - first) / static_cast(count > 1 ? count - 1 : 1)}; } MATX_LOG_TRACE("Logspace constructor: first={}, last={}, count={}", first, last, count); #endif diff --git a/test/00_operators/GeneratorTests.cu b/test/00_operators/GeneratorTests.cu index 720b95978..3a52a8bc4 100644 --- a/test/00_operators/GeneratorTests.cu +++ b/test/00_operators/GeneratorTests.cu @@ -137,6 +137,67 @@ TYPED_TEST(BasicGeneratorTestsFloatNonComplex, Windows) MATX_EXIT_HANDLER(); } +TYPED_TEST(BasicGeneratorTestsFloatNonComplex, SingleElementWindows) +{ + MATX_ENTER_HANDLER(); + + using TestType = cuda::std::tuple_element_t<0, TypeParam>; + using ExecType = cuda::std::tuple_element_t<1, TypeParam>; + ExecType exec{}; + + auto ov = make_tensor({1}); + + // A one-point window is 1, matching numpy and scipy. + (ov = hanning<0>({1})).run(exec); + exec.sync(); + EXPECT_EQ(static_cast(ov(0)), 1.0f); + + (ov = hamming<0>({1})).run(exec); + exec.sync(); + EXPECT_EQ(static_cast(ov(0)), 1.0f); + + (ov = bartlett<0>({1})).run(exec); + exec.sync(); + EXPECT_EQ(static_cast(ov(0)), 1.0f); + + (ov = blackman<0>({1})).run(exec); + exec.sync(); + EXPECT_EQ(static_cast(ov(0)), 1.0f); + + (ov = flattop<0>({1})).run(exec); + exec.sync(); + EXPECT_EQ(static_cast(ov(0)), 1.0f); + + MATX_EXIT_HANDLER(); +} + +TEST(OperatorTests, SingleElementLinspaceLogspace) +{ + MATX_ENTER_HANDLER(); + cudaExecutor exec{}; + + auto ov = make_tensor({1}); + + // A single point is the start value, matching numpy. + (ov = linspace(2.0f, 5.0f, 1)).run(exec); + exec.sync(); + EXPECT_EQ(ov(0), 2.0f); + + (ov = logspace<0>({1}, 1.0f, 3.0f)).run(exec); + exec.sync(); + EXPECT_NEAR(ov(0), 10.0f, 1e-4f); + + const float firsts[] = {2.0f, 3.0f}; + const float lasts[] = {5.0f, 9.0f}; + auto ov2 = make_tensor({1, 2}); + (ov2 = linspace(firsts, lasts, 1, 0)).run(exec); + exec.sync(); + EXPECT_EQ(ov2(0, 0), 2.0f); + EXPECT_EQ(ov2(0, 1), 3.0f); + + MATX_EXIT_HANDLER(); +} + TYPED_TEST(BasicGeneratorTestsAll, Diag) { using TestType = cuda::std::tuple_element_t<0, TypeParam>; From 2d026880a2b642f766a3975a5d8deabd709dd040 Mon Sep 17 00:00:00 2001 From: Arthur031221 Date: Fri, 2 Oct 2026 01:06:58 +0800 Subject: [PATCH 2/3] Fix singleton logspace and half-precision bartlett, test one-point JIT windows A one-point logspace still divided (last - first) by one, so an endpoint difference that overflows gave an infinite step and NaN at index zero. Use a zero step when count is one. The one-point early return added to bartlett was typed T while the general expression was not, so bartlett with a half or bfloat16 generator type stopped compiling. Convert the general expression to T. Add a one-point window test for the JIT executor and build the one-point bartlett test with the output type. --- include/matx/generators/bartlett.h | 4 +-- include/matx/generators/logspace.h | 10 ++++---- test/00_operators/GeneratorTests.cu | 40 ++++++++++++++++++++++++++++- 3 files changed, 46 insertions(+), 8 deletions(-) diff --git a/include/matx/generators/bartlett.h b/include/matx/generators/bartlett.h index 7c551d802..dade03ca9 100644 --- a/include/matx/generators/bartlett.h +++ b/include/matx/generators/bartlett.h @@ -72,7 +72,7 @@ namespace matx " template \n" + " __MATX_INLINE__ __MATX_DEVICE__ auto operator()(index_t i) const\n" + " {\n" + - " return detail::ApplyGeneratorVecFunc([](index_t idx) { if (size_ == 1) { return T(1); } return 1 - cuda::std::abs(((2*T(idx))/(T(size_ - 1))) - 1); }, i);\n" + + " return detail::ApplyGeneratorVecFunc([](index_t idx) { if (size_ == 1) { return T(1); } return T(1 - cuda::std::abs(((2*T(idx))/(T(size_ - 1))) - 1)); }, i);\n" + " }\n" + " static __MATX_INLINE__ constexpr __MATX_DEVICE__ int32_t Rank() { return 1; }\n" + " constexpr __MATX_INLINE__ __MATX_DEVICE__ index_t Size([[maybe_unused]] int dim) const { return size_; }\n" + @@ -92,7 +92,7 @@ namespace matx template inline __MATX_HOST__ __MATX_DEVICE__ auto operator()(index_t i) const { - return detail::ApplyGeneratorVecFunc([this](index_t idx) { if (size_ == 1) { return T(1); } return 1 - cuda::std::abs(((2*T(idx))/(T(size_ - 1))) - 1); }, i); + return detail::ApplyGeneratorVecFunc([this](index_t idx) { if (size_ == 1) { return T(1); } return T(1 - cuda::std::abs(((2*T(idx))/(T(size_ - 1))) - 1)); }, i); } inline __MATX_HOST__ __MATX_DEVICE__ auto operator()(index_t i) const diff --git a/include/matx/generators/logspace.h b/include/matx/generators/logspace.h index 45cfc318a..b1c5f3662 100644 --- a/include/matx/generators/logspace.h +++ b/include/matx/generators/logspace.h @@ -96,20 +96,20 @@ namespace matx { #ifdef __CUDA_ARCH__ if constexpr (is_matx_half_v) { - range_ = Range{first, (last - first) / static_cast(count > 1 ? count - 1.0f : 1.0f)}; + range_ = Range{first, count > 1 ? (last - first) / static_cast(count - 1.0f) : static_cast(0.0f)}; } else { - range_ = Range{first, (last - first) / static_cast(count > 1 ? count - 1 : 1)}; + range_ = Range{first, count > 1 ? (last - first) / static_cast(count - 1) : static_cast(0)}; } #else // Host has no support for most half precision operators/intrinsics if constexpr (is_matx_half_v) { range_ = Range{static_cast(first), - (static_cast(last) - static_cast(first)) / - static_cast(count > 1 ? count - 1 : 1)}; + count > 1 ? (static_cast(last) - static_cast(first)) / + static_cast(count - 1) : 0.0f}; } else { - range_ = Range{first, (last - first) / static_cast(count > 1 ? count - 1 : 1)}; + range_ = Range{first, count > 1 ? (last - first) / static_cast(count - 1) : static_cast(0)}; } MATX_LOG_TRACE("Logspace constructor: first={}, last={}, count={}", first, last, count); #endif diff --git a/test/00_operators/GeneratorTests.cu b/test/00_operators/GeneratorTests.cu index 3a52a8bc4..b7c1c521f 100644 --- a/test/00_operators/GeneratorTests.cu +++ b/test/00_operators/GeneratorTests.cu @@ -156,7 +156,7 @@ TYPED_TEST(BasicGeneratorTestsFloatNonComplex, SingleElementWindows) exec.sync(); EXPECT_EQ(static_cast(ov(0)), 1.0f); - (ov = bartlett<0>({1})).run(exec); + (ov = bartlett<0, cuda::std::array, TestType>({1})).run(exec); exec.sync(); EXPECT_EQ(static_cast(ov(0)), 1.0f); @@ -187,6 +187,12 @@ TEST(OperatorTests, SingleElementLinspaceLogspace) exec.sync(); EXPECT_NEAR(ov(0), 10.0f, 1e-4f); + // The endpoint must not matter for a single point, even when last - first overflows. + (ov = logspace<0>({1}, -1e38f, 3e38f)).run(exec); + exec.sync(); + EXPECT_FALSE(std::isnan(ov(0))); + EXPECT_EQ(ov(0), 0.0f); + const float firsts[] = {2.0f, 3.0f}; const float lasts[] = {5.0f, 9.0f}; auto ov2 = make_tensor({1, 2}); @@ -885,6 +891,38 @@ TEST(OperatorTests, RandomJITCapabilityLimits) MATX_EXIT_HANDLER(); } +#ifdef MATX_EN_JIT +TEST(OperatorTests, SingleElementWindowsJIT) +{ + MATX_ENTER_HANDLER(); + CUDAJITExecutor exec{}; + + auto ov = make_tensor({1}); + + (ov = hanning<0>({1})).run(exec); + exec.sync(); + EXPECT_EQ(ov(0), 1.0f); + + (ov = hamming<0>({1})).run(exec); + exec.sync(); + EXPECT_EQ(ov(0), 1.0f); + + (ov = bartlett<0>({1})).run(exec); + exec.sync(); + EXPECT_EQ(ov(0), 1.0f); + + (ov = blackman<0>({1})).run(exec); + exec.sync(); + EXPECT_EQ(ov(0), 1.0f); + + (ov = flattop<0>({1})).run(exec); + exec.sync(); + EXPECT_EQ(ov(0), 1.0f); + + MATX_EXIT_HANDLER(); +} +#endif + #if defined(MATX_EN_MATHDX) && defined(MATX_EN_JIT) TEST(OperatorTests, RandomJITFusedUniform) { From 2d1e6057d698d92ae178fe23cc6a739a7873b1d6 Mon Sep 17 00:00:00 2001 From: Arthur031221 Date: Fri, 2 Oct 2026 04:51:38 +0800 Subject: [PATCH 3/3] Test singleton windows with explicit float and double generators --- test/00_operators/GeneratorTests.cu | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/test/00_operators/GeneratorTests.cu b/test/00_operators/GeneratorTests.cu index b7c1c521f..6a623159c 100644 --- a/test/00_operators/GeneratorTests.cu +++ b/test/00_operators/GeneratorTests.cu @@ -146,25 +146,28 @@ TYPED_TEST(BasicGeneratorTestsFloatNonComplex, SingleElementWindows) ExecType exec{}; auto ov = make_tensor({1}); + using Shape = cuda::std::array; + // These windows cannot currently be instantiated with half or bfloat16 output types. + using WindowType = cuda::std::conditional_t, float, TestType>; // A one-point window is 1, matching numpy and scipy. - (ov = hanning<0>({1})).run(exec); + (ov = hanning<0, Shape, WindowType>({1})).run(exec); exec.sync(); EXPECT_EQ(static_cast(ov(0)), 1.0f); - (ov = hamming<0>({1})).run(exec); + (ov = hamming<0, Shape, WindowType>({1})).run(exec); exec.sync(); EXPECT_EQ(static_cast(ov(0)), 1.0f); - (ov = bartlett<0, cuda::std::array, TestType>({1})).run(exec); + (ov = bartlett<0, Shape, TestType>({1})).run(exec); exec.sync(); EXPECT_EQ(static_cast(ov(0)), 1.0f); - (ov = blackman<0>({1})).run(exec); + (ov = blackman<0, Shape, WindowType>({1})).run(exec); exec.sync(); EXPECT_EQ(static_cast(ov(0)), 1.0f); - (ov = flattop<0>({1})).run(exec); + (ov = flattop<0, Shape, WindowType>({1})).run(exec); exec.sync(); EXPECT_EQ(static_cast(ov(0)), 1.0f);