From 7419945115860636aaa56fe5caaad2a117e597cf Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:13:55 -0400 Subject: [PATCH 01/20] Evaluate the beta and gamma densities in log space beta_pdf_scalar, gamma_pdf_scalar and chi_square_pdf_scalar returned nan or inf once a shape parameter passed ~171: Beta(200, 200).pdf(0.5) was nan against a true 15.95, Gamma(200, 1).pdf(200) nan against 0.0282, and every chi-square with k above ~342 likewise, since it delegates to the gamma density. Same cause as the negative binomial PMF fixed earlier: the normalising constants were formed from raw tgamma calls (and theta^alpha), which overflow a double long before the density itself does. Both densities are now evaluated in log space through lgamma. A unit exponent is special-cased to contribute zero, so the endpoint values that pow(0, 0) == 1 used to produce survive the rewrite -- Beta(1, b) at x = 0 is b, and Gamma(1, theta) at x = 0 is 1/theta. Tests pin those alongside the large-shape cases. Co-Authored-By: Claude Opus 5 --- src/math/beta.cpp | 11 ++++++++--- src/math/gamma.cpp | 6 +++++- tests/python/test_beta.py | 30 ++++++++++++++++++++++++++++++ tests/python/test_chi_square.py | 16 ++++++++++++++++ tests/python/test_gamma.py | 25 +++++++++++++++++++++++++ 5 files changed, 84 insertions(+), 4 deletions(-) diff --git a/src/math/beta.cpp b/src/math/beta.cpp index c3c6d60..31404b9 100644 --- a/src/math/beta.cpp +++ b/src/math/beta.cpp @@ -23,9 +23,14 @@ namespace fastdist::math { return 0.0; } - const double B = std::tgamma(alpha) * std::tgamma(beta) / std::tgamma(alpha + beta); - - return std::pow(x, alpha - 1.0) * std::pow(1.0 - x, beta - 1.0) / B; + // Evaluated in log space: Gamma(alpha) and Gamma(beta) overflow a double + // once either shape passes ~171, long before the density does. A unit + // exponent contributes nothing, matching pow(0, 0) == 1 at the endpoints. + const double log_beta = std::lgamma(alpha) + std::lgamma(beta) - std::lgamma(alpha + beta); + const double log_x_term = (alpha == 1.0) ? 0.0 : (alpha - 1.0) * std::log(x); + const double log_1mx_term = (beta == 1.0) ? 0.0 : (beta - 1.0) * std::log1p(-x); + + return std::exp(log_x_term + log_1mx_term - log_beta); } // Forward declarations for internal functions diff --git a/src/math/gamma.cpp b/src/math/gamma.cpp index d379075..bfdfc5a 100644 --- a/src/math/gamma.cpp +++ b/src/math/gamma.cpp @@ -19,7 +19,11 @@ namespace fastdist::math { if (x < 0.0) return 0.0; - return std::pow(x, alpha - 1.0) * std::exp(-x / theta) / (std::tgamma(alpha) * std::pow(theta, alpha)); + // Evaluated in log space: Gamma(alpha) and theta^alpha overflow a double + // once alpha passes ~171, long before the density does. A unit exponent + // contributes nothing, matching pow(0, 0) == 1 at x = 0. + const double log_x_term = (alpha == 1.0) ? 0.0 : (alpha - 1.0) * std::log(x); + return std::exp(log_x_term - x / theta - std::lgamma(alpha) - alpha * std::log(theta)); } // Forward declarations for internal functions diff --git a/tests/python/test_beta.py b/tests/python/test_beta.py index 4b12932..a5c77c4 100644 --- a/tests/python/test_beta.py +++ b/tests/python/test_beta.py @@ -269,3 +269,33 @@ def test_sample_lies_within_the_unit_interval(alpha, beta): def test_slots_prevent_dynamic_attributes(): with pytest.raises(AttributeError): Beta(2.0, 3.0).extra = 123 + + +# --------------------------------------------------------------------------- +# Large shapes +# +# The density used to be formed from raw tgamma calls, which overflow once a +# shape passes ~171 and turned Beta(200, 200).pdf(0.5) into nan. It is now +# evaluated in log space; these cases hold that. +# --------------------------------------------------------------------------- + +def _beta_pdf_reference(x, a, b): + log_b = math.lgamma(a) + math.lgamma(b) - math.lgamma(a + b) + return math.exp((a - 1) * math.log(x) + (b - 1) * math.log1p(-x) - log_b) + + +@pytest.mark.parametrize("alpha, beta, x", [(150.0, 150.0, 0.5), (200.0, 200.0, 0.5), (500.0, 300.0, 0.6)]) +def test_pdf_is_finite_and_correct_for_large_shapes(alpha, beta, x): + value = Beta(alpha, beta).pdf_scalar(x) + assert math.isfinite(value) + assert value == pytest.approx(_beta_pdf_reference(x, alpha, beta), rel=1e-10) + + +@pytest.mark.parametrize("alpha, beta, x, expected", [ + (1.0, 1.0, 0.0, 1.0), # uniform: density 1 at both endpoints + (1.0, 1.0, 1.0, 1.0), + (1.0, 3.0, 0.0, 3.0), # pow(0, 0) == 1 must survive the log rewrite + (2.0, 3.0, 0.0, 0.0), +]) +def test_pdf_endpoint_values(alpha, beta, x, expected): + assert Beta(alpha, beta).pdf_scalar(x) == pytest.approx(expected, **EXACT) diff --git a/tests/python/test_chi_square.py b/tests/python/test_chi_square.py index 72ae818..a404ffb 100644 --- a/tests/python/test_chi_square.py +++ b/tests/python/test_chi_square.py @@ -250,3 +250,19 @@ def test_sample_is_positive_and_finite(k): def test_slots_prevent_dynamic_attributes(): with pytest.raises(AttributeError): ChiSquare(k=5.0).extra = 123 + + +# --------------------------------------------------------------------------- +# Large degrees of freedom +# +# The chi-square density delegates to the gamma density, which overflowed for +# shapes past ~171 -- so any k above ~342 returned nan. +# --------------------------------------------------------------------------- + +@pytest.mark.parametrize("k", [400.0, 1000.0]) +def test_pdf_is_finite_for_large_degrees_of_freedom(k): + x = k + expected = math.exp((k / 2 - 1) * math.log(x) - x / 2 - math.lgamma(k / 2) - (k / 2) * math.log(2.0)) + value = ChiSquare(k).pdf(x) + assert math.isfinite(value) + assert value == pytest.approx(expected, rel=1e-10) diff --git a/tests/python/test_gamma.py b/tests/python/test_gamma.py index 14a75a1..90f19ff 100644 --- a/tests/python/test_gamma.py +++ b/tests/python/test_gamma.py @@ -371,3 +371,28 @@ def test_sample_is_positive_and_finite(alpha, theta): def test_slots_prevent_dynamic_attributes(): with pytest.raises(AttributeError): Gamma(2.0, 3.0).extra = 123 + + +# --------------------------------------------------------------------------- +# Large shapes +# +# The density used to divide by tgamma(alpha) * theta**alpha, both of which +# overflow once alpha passes ~171; Gamma(200, 1).pdf(200) was nan. It is now +# evaluated in log space; these cases hold that. +# --------------------------------------------------------------------------- + +def _gamma_pdf_reference(x, alpha, theta): + return math.exp((alpha - 1) * math.log(x) - x / theta - math.lgamma(alpha) - alpha * math.log(theta)) + + +@pytest.mark.parametrize("alpha, theta, x", [(150.0, 1.0, 150.0), (200.0, 1.0, 200.0), (1000.0, 2.0, 2000.0)]) +def test_pdf_is_finite_and_correct_for_large_shapes(alpha, theta, x): + value = Gamma(alpha, theta).pmf_scalar(x) + assert math.isfinite(value) + assert value == pytest.approx(_gamma_pdf_reference(x, alpha, theta), rel=1e-10) + + +@pytest.mark.parametrize("theta", [0.5, 2.0]) +def test_pdf_at_zero_with_unit_shape_is_the_rate(theta): + """pow(0, 0) == 1 must survive the log rewrite: Gamma(1, th).pdf(0) = 1/th.""" + assert Gamma(1.0, theta).pmf_scalar(0.0) == pytest.approx(1.0 / theta, **EXACT) From f9918cf904ecd16cee75ae93ddea20674cde3b16 Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:13:56 -0400 Subject: [PATCH 02/20] Tighten comments in the Python package Comments only; no behaviour change. mypy over python/fastdist stays clean. - Header comments in beta.py and negative_binomial.py named the wrong module (bernoulli.py, poisson.py), and every distribution module omitted the fastdist package directory from its path. - A three-line note on why the scalar/array branch tests np.ndarray rather than numbers.Real was pasted at every call site, around ten per file. One short copy per file remains, at the first site. - The setter and sigmoid comments described how the code used to fail rather than what it guarantees now. That history is in the commits that fixed it. - config.py described a delayed import that does not exist (the registry is simply built on first use) and narrated one-line helpers. Co-Authored-By: Claude Opus 5 --- python/fastdist/config.py | 6 ++--- python/fastdist/distributions/bernoulli.py | 16 +++----------- python/fastdist/distributions/beta.py | 2 +- python/fastdist/distributions/binomial.py | 2 +- python/fastdist/distributions/chi_square.py | 2 +- .../distributions/discrete_uniform.py | 8 +++---- python/fastdist/distributions/exponential.py | 16 +++----------- python/fastdist/distributions/gamma.py | 2 +- python/fastdist/distributions/geometric.py | 2 +- .../distributions/negative_binomial.py | 2 +- python/fastdist/distributions/normal.py | 19 +++------------- python/fastdist/distributions/poisson.py | 16 +++----------- python/fastdist/distributions/uniform.py | 22 +++++-------------- python/fastdist/distributions/utils.py | 17 +++++--------- 14 files changed, 33 insertions(+), 99 deletions(-) diff --git a/python/fastdist/config.py b/python/fastdist/config.py index c8060e2..ff3b734 100644 --- a/python/fastdist/config.py +++ b/python/fastdist/config.py @@ -204,9 +204,8 @@ def _get_dist_class_map(): def _get_function_pair(fd_function): global _FUNCTION_REGISTRY - # Delay the import until the function is actually called + # Built on first use rather than at import time. if _FUNCTION_REGISTRY is None: - # Now we hard code it inside the function _FUNCTION_REGISTRY = { "normal_pdf": ("_pdf_cpu", "_pdf_cuda"), "normal_logpdf": ("_logpdf_cpu", "_logpdf_cuda"), @@ -276,12 +275,11 @@ def _benchmark(fd_function: str, display: int = 0, *args) -> int: if the benchmark does not converge within the test iterations. """ - # Creating the default space array and master array for benchmarking + # One array at the largest size; each probe takes a prefix of it. master_array = _generate_int_array(_DEFAULT_SPACE_ARRAY[-1]) if display > 1: print(f"Created array of size {master_array.shape}") - # Gets the array def get_array(size): return master_array[:size] diff --git a/python/fastdist/distributions/bernoulli.py b/python/fastdist/distributions/bernoulli.py index 098262f..0d4a8ac 100644 --- a/python/fastdist/distributions/bernoulli.py +++ b/python/fastdist/distributions/bernoulli.py @@ -1,4 +1,4 @@ -# python/distributions/bernoulli.py +# python/fastdist/distributions/bernoulli.py try: from fastdist import _fastdist as _core except ImportError as exc: # pragma: no cover - only hit in a broken install @@ -264,9 +264,8 @@ def pmf(self, k: Union[int, Sequence[int]], step_size: int = 0) -> Union[float, validated_input = self._validate_inputs(_input=k, input_name="k", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # See the note above: discriminating on ndarray is what lets a - # type checker narrow the union. Validation upstream already - # guarantees an integer scalar here. + # isinstance(..., np.ndarray) rather than numbers.Real so type + # checkers can narrow the union _validate_inputs returns. return _core.bernoulli_pmf_scalar(validated_input, self.p) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("bernoulli_pmf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -304,9 +303,6 @@ def cdf(self, k: Union[int, Sequence[int]], step_size: int = 0) -> Union[float, validated_input = self._validate_inputs(_input=k, input_name="k", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # See the note above: discriminating on ndarray is what lets a - # type checker narrow the union. Validation upstream already - # guarantees an integer scalar here. return _core.bernoulli_cdf_scalar(validated_input, self.p) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("bernoulli_cdf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -418,9 +414,6 @@ def mgf(self, t: Union[SupportsFloat, ArrayLike], validated_input = self._validate_inputs(_input=t, input_name="t", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.bernoulli_mgf_scalar(validated_input, self.p) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("bernoulli_mgf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -453,9 +446,6 @@ def cgf(self, t: Union[SupportsFloat, ArrayLike], step_size: int = 0) -> Union[f validated_input = self._validate_inputs(_input=t, input_name="t", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.bernoulli_cgf_scalar(validated_input, self.p) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("bernoulli_cgf"): config.validate_gpu_capacity(validated_input.size, 8) diff --git a/python/fastdist/distributions/beta.py b/python/fastdist/distributions/beta.py index 5fe2bf2..a0aeefb 100644 --- a/python/fastdist/distributions/beta.py +++ b/python/fastdist/distributions/beta.py @@ -1,4 +1,4 @@ -# python/distributions/bernoulli.py +# python/fastdist/distributions/beta.py try: from .. import _fastdist as _core except ImportError as exc: # pragma: no cover - only hit in a broken install diff --git a/python/fastdist/distributions/binomial.py b/python/fastdist/distributions/binomial.py index 050c1aa..95cee65 100644 --- a/python/fastdist/distributions/binomial.py +++ b/python/fastdist/distributions/binomial.py @@ -1,4 +1,4 @@ -# python/distributions/binomial.py +# python/fastdist/distributions/binomial.py try: from fastdist import _fastdist as _core except ImportError as exc: # pragma: no cover - only hit in a broken install diff --git a/python/fastdist/distributions/chi_square.py b/python/fastdist/distributions/chi_square.py index d60bc81..6e96527 100644 --- a/python/fastdist/distributions/chi_square.py +++ b/python/fastdist/distributions/chi_square.py @@ -1,4 +1,4 @@ -# python/distributions/chi_square.py +# python/fastdist/distributions/chi_square.py try: from fastdist import _fastdist as _core except ImportError as exc: # pragma: no cover - only hit in a broken install diff --git a/python/fastdist/distributions/discrete_uniform.py b/python/fastdist/distributions/discrete_uniform.py index de417e0..eb581eb 100644 --- a/python/fastdist/distributions/discrete_uniform.py +++ b/python/fastdist/distributions/discrete_uniform.py @@ -1,4 +1,4 @@ -# python/distributions/discrete_uniform.py +# python/fastdist/distributions/discrete_uniform.py try: from fastdist import _fastdist as _core except ImportError as exc: # pragma: no cover - only hit in a broken install @@ -28,10 +28,8 @@ def a(self): @a.setter def a(self, value): - # Both bounds are passed so the a < b relationship is re-checked against - # the current opposite bound, and int() matches how __init__ stores it -- - # a is an integer parameter, so assigning through the setter must not - # quietly change its type to float. + # Validate against the current opposite bound so the a < b invariant + # holds after every assignment; store as int, as __init__ does. self._validate_params(a=value, b=self._b) self._a = int(value) diff --git a/python/fastdist/distributions/exponential.py b/python/fastdist/distributions/exponential.py index f24f207..ccd09a8 100644 --- a/python/fastdist/distributions/exponential.py +++ b/python/fastdist/distributions/exponential.py @@ -1,4 +1,4 @@ -# python/distributions/exponential.py +# python/fastdist/distributions/exponential.py try: from fastdist import _fastdist as _core except ImportError as exc: # pragma: no cover - only hit in a broken install @@ -211,9 +211,8 @@ def pdf(self, x: Union[SupportsFloat, ArrayLike], validated_input = self._validate_inputs(_input=x, input_name="x", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. + # isinstance(..., np.ndarray) rather than numbers.Real so type + # checkers can narrow the union _validate_inputs returns. return _core.exponential_pdf_scalar(validated_input, self.lambda_) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("exponential_pdf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -248,9 +247,6 @@ def cdf(self, x: Union[SupportsFloat, ArrayLike], validated_input = self._validate_inputs(_input=x, input_name="x", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.exponential_cdf_scalar(validated_input, self.lambda_) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("exponential_cdf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -332,9 +328,6 @@ def mgf(self, t: Union[SupportsFloat, ArrayLike], step_size: SupportsFloat = 0) validated_input = self._validate_inputs(_input=t, input_name="t", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.exponential_mgf_scalar(validated_input, self.lambda_) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("exponential_mgf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -368,9 +361,6 @@ def cgf(self, t: Union[SupportsFloat, ArrayLike], step_size: SupportsFloat = 0) validated_input = self._validate_inputs(_input=t, input_name="t", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.exponential_cgf_scalar(validated_input, self.lambda_) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("exponential_cgf"): config.validate_gpu_capacity(validated_input.size, 8) diff --git a/python/fastdist/distributions/gamma.py b/python/fastdist/distributions/gamma.py index 9818cad..6763e91 100644 --- a/python/fastdist/distributions/gamma.py +++ b/python/fastdist/distributions/gamma.py @@ -1,4 +1,4 @@ -# python/distributions/gamma.py +# python/fastdist/distributions/gamma.py try: from .. import _fastdist as _core diff --git a/python/fastdist/distributions/geometric.py b/python/fastdist/distributions/geometric.py index 1af1d25..7742506 100644 --- a/python/fastdist/distributions/geometric.py +++ b/python/fastdist/distributions/geometric.py @@ -1,4 +1,4 @@ -# python/distributions/geometric.py +# python/fastdist/distributions/geometric.py try: from fastdist import _fastdist as _core except ImportError as exc: # pragma: no cover - only hit in a broken install diff --git a/python/fastdist/distributions/negative_binomial.py b/python/fastdist/distributions/negative_binomial.py index ed30f00..fcb10a8 100644 --- a/python/fastdist/distributions/negative_binomial.py +++ b/python/fastdist/distributions/negative_binomial.py @@ -1,4 +1,4 @@ -# python/distributions/poisson.py +# python/fastdist/distributions/negative_binomial.py try: from fastdist import _fastdist as _core except ImportError as exc: # pragma: no cover - only hit in a broken install diff --git a/python/fastdist/distributions/normal.py b/python/fastdist/distributions/normal.py index 7bc4be2..08291f3 100644 --- a/python/fastdist/distributions/normal.py +++ b/python/fastdist/distributions/normal.py @@ -1,4 +1,4 @@ -# python/distributions/normal.py +# python/fastdist/distributions/normal.py try: from .. import _fastdist as _core except ImportError as exc: # pragma: no cover - only hit in a broken install @@ -315,9 +315,8 @@ def pdf(self, x: Union[SupportsFloat, ArrayLike], validated_input = self._validate_inputs(_input=x, input_name="x", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. + # isinstance(..., np.ndarray) rather than numbers.Real so type + # checkers can narrow the union _validate_inputs returns. return _core.normal_pdf_scalar(x=validated_input, mu=self.mu, sigma=self.sigma) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("normal_pdf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -362,9 +361,6 @@ def logpdf(self, x: Union[SupportsFloat, ArrayLike], validated_input = self._validate_inputs(_input=x, input_name="x", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.normal_logpdf_scalar(x=validated_input, mu=self.mu, sigma=self.sigma) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("normal_logpdf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -410,9 +406,6 @@ def cdf(self, x: Union[SupportsFloat, ArrayLike], validated_input = self._validate_inputs(_input=x, input_name="x", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.normal_cdf_scalar(validated_input, self.mu, self.sigma) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("normal_cdf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -531,9 +524,6 @@ def mgf(self, t: Union[SupportsFloat, ArrayLike], validated_input = self._validate_inputs(_input=t, input_name="t", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.normal_mgf_scalar(validated_input, self.mu, self.sigma) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("normal_mgf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -577,9 +567,6 @@ def cgf(self, t: Union[SupportsFloat, ArrayLike], validated_input = self._validate_inputs(_input=t, input_name="t", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.normal_cgf_scalar(validated_input, self.mu, self.sigma) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("normal_cgf"): config.validate_gpu_capacity(validated_input.size, 8) diff --git a/python/fastdist/distributions/poisson.py b/python/fastdist/distributions/poisson.py index 895d8f0..4fe3963 100644 --- a/python/fastdist/distributions/poisson.py +++ b/python/fastdist/distributions/poisson.py @@ -1,4 +1,4 @@ -# python/distributions/poisson.py +# python/fastdist/distributions/poisson.py try: from .. import _fastdist as _core except ImportError as exc: # pragma: no cover - only hit in a broken install @@ -129,9 +129,8 @@ def pmf(self, x: Union[SupportsFloat, ArrayLike], validated_input = self._validate_inputs(_input=x, input_name="x", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. + # isinstance(..., np.ndarray) rather than numbers.Real so type + # checkers can narrow the union _validate_inputs returns. return _core.poisson_pmf_scalar(validated_input, self.lambda_) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("poisson_pmf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -145,9 +144,6 @@ def cdf(self, x: Union[SupportsFloat, ArrayLike], validated_input = self._validate_inputs(_input=x, input_name="x", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.poisson_cdf_scalar(validated_input, self.lambda_) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("poisson_cdf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -181,9 +177,6 @@ def mgf(self, t: Union[SupportsFloat, ArrayLike], step_size: int = 0) -> Union[float, np.ndarray]: validated_input = self._validate_inputs(_input=t, input_name="t", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.poisson_mgf_scalar(validated_input, self.lambda_) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("poisson_mgf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -197,9 +190,6 @@ def cgf(self, t: Union[SupportsFloat, ArrayLike], validated_input = self._validate_inputs(_input=t, input_name="t", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.poisson_cgf_scalar(validated_input, self.lambda_) elif _CUDA_AVAILABLE and validated_input.size > config.get_cuda_threshold("poisson_cgf"): config.validate_gpu_capacity(validated_input.size, 8) diff --git a/python/fastdist/distributions/uniform.py b/python/fastdist/distributions/uniform.py index 087fa82..d168dae 100644 --- a/python/fastdist/distributions/uniform.py +++ b/python/fastdist/distributions/uniform.py @@ -1,4 +1,4 @@ -# python/distributions/uniform.py +# python/fastdist/distributions/uniform.py try: from .. import _fastdist as _core except ImportError as exc: # pragma: no cover - only hit in a broken install @@ -102,10 +102,8 @@ def a(self, value): 0.2 """ - # The opposite bound is passed too: validating `a` alone skips the - # a < b check entirely, which let Uniform(1.0, 3.0) be driven to - # a = 10.0, b = -10.0 -- a state the constructor rejects outright, and - # from which pdf, cdf, mean, variance and sample all silently return nan. + # Validate against the current opposite bound so the a < b invariant + # holds after every assignment, not just at construction. self._validate_params(a=value, b=self._b) self._a = float(value) @@ -338,9 +336,8 @@ def pdf(self, x: Union[SupportsFloat, ArrayLike], step_size: SupportsFloat = 0) validated_input = self._validate_inputs(_input=x, input_name="x", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. + # isinstance(..., np.ndarray) rather than numbers.Real so type + # checkers can narrow the union _validate_inputs returns. return _core.uniform_pdf_scalar(validated_input, self.a, self.b) elif _CUDA_AVAILABLE and len(validated_input) > config.get_cuda_threshold("uniform_pdf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -379,9 +376,6 @@ def cdf(self, x: Union[SupportsFloat, ArrayLike], step_size: SupportsFloat = 0) validated_input = self._validate_inputs(_input=x, input_name="x", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.uniform_cdf_scalar(validated_input, self.a, self.b) elif _CUDA_AVAILABLE and len(validated_input) > config.get_cuda_threshold("uniform_cdf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -510,9 +504,6 @@ def mgf(self, t: Union[SupportsFloat, ArrayLike], step_size: SupportsFloat = 0) validated_input = self._validate_inputs(_input=t, input_name="t", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.uniform_mgf_scalar(validated_input, self.a, self.b) elif _CUDA_AVAILABLE and len(validated_input) > config.get_cuda_threshold("uniform_mgf"): config.validate_gpu_capacity(validated_input.size, 8) @@ -551,9 +542,6 @@ def cgf(self, t: Union[SupportsFloat, ArrayLike], step_size: SupportsFloat = 0) validated_input = self._validate_inputs(_input=t, input_name="t", step_size=step_size) if not isinstance(validated_input, np.ndarray): - # Discriminating on ndarray rather than numbers.Real lets a type - # checker narrow the union; the test is equivalent, since - # _validate_inputs returns either a scalar or an ndarray. return _core.uniform_cgf_scalar(validated_input, self.a, self.b) elif _CUDA_AVAILABLE and len(validated_input) > config.get_cuda_threshold("uniform_cgf"): config.validate_gpu_capacity(validated_input.size, 8) diff --git a/python/fastdist/distributions/utils.py b/python/fastdist/distributions/utils.py index b1e1247..65e240f 100644 --- a/python/fastdist/distributions/utils.py +++ b/python/fastdist/distributions/utils.py @@ -1,4 +1,4 @@ -# python/distributions/utils.py +# python/fastdist/distributions/utils.py try: from fastdist import _fastdist as _core except ImportError as exc: # pragma: no cover - only hit in a broken install @@ -127,22 +127,15 @@ def _as_list(value: object) -> list: @classmethod def sigmoid(cls, x: SupportsFloat) -> float: - """Logistic function for a single value. + """Logistic function 1 / (1 + e^-x) for a single value. - Scalar only. The signature used to advertise a sequence type as well, - but the body calls float() on the input so any sequence raised - TypeError. Use sigmoid_cpu for arrays -- the scalar/batch split is the - same one the distribution classes use, and returning an ndarray from a - function annotated -> float would be worse than not accepting one. + Scalar only; use ``sigmoid_cpu`` for arrays. """ - # _validate_input accepts a sequence even when asked for Real, and - # float() on the resulting array then fails with a numpy message about - # 0-dimensional arrays, which says nothing useful. Reject it here with - # the name of the function that does handle arrays. + # _validate_input would accept a sequence here; reject it up front so + # the error names the array entry point instead of failing in float(). if not isinstance(x, Real): raise TypeError("x must be a real number; use Utils.sigmoid_cpu for arrays") - validated_input = cls._validate_input(_input=x, input_name="x", input_type=Real) return _core.sigmoid(float(validated_input)) From 59f6e651e0a01471c7ed3a53e3f6376b3ac347d0 Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:13:57 -0400 Subject: [PATCH 03/20] Tighten comments in the CUDA utilities Comments only, kept in a separate commit so it can be reverted independently of anything else. The CUDA backend does not build on the machine this was written on (CUDA 12.4 against MSVC 19.42), so these files were not compiled here; clang-format passes on them. - Removed conversational notes ("This is the line MSVC hated, but here it's inside a .cu file, so it's safe!", "CUDA kernel remains essentially the same") and the numbered step-by-step narration of the host lifecycle. - Each kernel now states the one non-obvious thing about it: one thread per vector pair, with strides[b]..strides[b + 1] delimiting pair b in the flattened inputs. - Kept, and stated plainly, the reason the distance dispatchers manage the device lifecycle by hand instead of through executor.cuh. - executor.cuh's header comment gave its path as src/cuda/; it lives under include/fastdist/cuda/. Co-Authored-By: Claude Opus 5 --- include/fastdist/cuda/executor.cuh | 2 +- src/cuda/utils/cosine_similarity.cu | 20 +++++--------------- src/cuda/utils/euclidean_distance.cu | 14 ++++---------- src/cuda/utils/manhattan_distance.cu | 15 ++++----------- 4 files changed, 14 insertions(+), 37 deletions(-) diff --git a/include/fastdist/cuda/executor.cuh b/include/fastdist/cuda/executor.cuh index 0e67732..ed57d89 100644 --- a/include/fastdist/cuda/executor.cuh +++ b/include/fastdist/cuda/executor.cuh @@ -1,4 +1,4 @@ -// src/cuda/executor.cuh +// include/fastdist/cuda/executor.cuh #ifndef FASTDIST_EXECUTOR_CUH #define FASTDIST_EXECUTOR_CUH diff --git a/src/cuda/utils/cosine_similarity.cu b/src/cuda/utils/cosine_similarity.cu index d3b93f3..1966a86 100644 --- a/src/cuda/utils/cosine_similarity.cu +++ b/src/cuda/utils/cosine_similarity.cu @@ -9,19 +9,17 @@ namespace fastdist::cuda::utils { - // CUDA kernel remains essentially the same + // One thread per vector pair. strides[b]..strides[b + 1] delimits pair b in + // the flattened inputs; offset shifts b when a launch covers a sub-range. __global__ void cosine_similarity_kernel(const double* x_input, const double* y_input, double* output, const int* strides, const int batch_count, const int offset) { const int b = blockIdx.x * blockDim.x + threadIdx.x; - // The b index here represents the batch index relative to the current launch if (b >= batch_count) return; - // Apply offset to batch index to find correct strides const int actual_b = b + offset; const int start = strides[actual_b]; const int end = strides[actual_b + 1]; - // n is the number of elements in the specific vectors for this batch const int n = end - start; double dot = 0.0; @@ -50,7 +48,8 @@ namespace fastdist::cuda::utils { output[actual_b] = dot / (sqrt(norm_x) * sqrt(norm_y)); } - // Consolidated Dispatcher: Replaces the template call with localized logic + // Host entry point: copies the batch to the device, launches one thread per + // pair, and copies the results back. Device buffers are freed on every path. void cosine_similarity_dispatcher(const double* x_input, const double* y_input, double* output, const int* strides, const int batch_count) { if (batch_count <= 0) return; @@ -58,14 +57,13 @@ namespace fastdist::cuda::utils { double *d_x = nullptr, *d_y = nullptr, *d_output = nullptr; int* d_strides = nullptr; - // total_elements is stored at the end of the strides array + // strides has batch_count + 1 entries; the last is the total element count. const int total_elements = strides[batch_count]; const size_t inputSize = total_elements * sizeof(double); const size_t outputSize = batch_count * sizeof(double); const size_t stridesSize = (batch_count + 1) * sizeof(int); try { - // 1. Allocation if (cudaMalloc(&d_x, inputSize) != cudaSuccess) throw std::runtime_error("cudaMalloc d_x failed"); if (cudaMalloc(&d_y, inputSize) != cudaSuccess) throw std::runtime_error("cudaMalloc d_y failed"); if (cudaMalloc(&d_output, outputSize) != cudaSuccess) @@ -73,30 +71,23 @@ namespace fastdist::cuda::utils { if (cudaMalloc(&d_strides, stridesSize) != cudaSuccess) throw std::runtime_error("cudaMalloc d_strides failed"); - // 2. Host to Device Transfer cudaMemcpy(d_x, x_input, inputSize, cudaMemcpyHostToDevice); cudaMemcpy(d_y, y_input, inputSize, cudaMemcpyHostToDevice); cudaMemcpy(d_strides, strides, stridesSize, cudaMemcpyHostToDevice); - // 3. Kernel Launch Parameters - // Note: We are parallelizing over the batch_count (one thread per similarity calculation) constexpr int threadsPerBlock = 256; const int blocksPerGrid = (batch_count + threadsPerBlock - 1) / threadsPerBlock; const int offset = 0; - // This is the line MSVC hated, but here it's inside a .cu file, so it's safe! cosine_similarity_kernel<<>>(d_x, d_y, d_output, d_strides, batch_count, offset); - // 4. Error Checking & Sync if (cudaGetLastError() != cudaSuccess) throw std::runtime_error("Kernel launch failed"); if (cudaDeviceSynchronize() != cudaSuccess) throw std::runtime_error("Kernel execution failed"); - // 5. Device to Host Transfer cudaMemcpy(output, d_output, outputSize, cudaMemcpyDeviceToHost); } catch (...) { - // Cleanup on any error cudaFree(d_x); cudaFree(d_y); cudaFree(d_output); @@ -104,7 +95,6 @@ namespace fastdist::cuda::utils { throw; } - // Normal Cleanup cudaFree(d_x); cudaFree(d_y); cudaFree(d_output); diff --git a/src/cuda/utils/euclidean_distance.cu b/src/cuda/utils/euclidean_distance.cu index 6ebbfc9..39d87f4 100644 --- a/src/cuda/utils/euclidean_distance.cu +++ b/src/cuda/utils/euclidean_distance.cu @@ -10,13 +10,13 @@ namespace fastdist::cuda::utils { - // CUDA kernel: Logic for square root of summed squared differences + // One thread per vector pair: sqrt of the summed squared differences. + // strides[b]..strides[b + 1] delimits pair b in the flattened inputs. __global__ void euclidean_distance_kernel(const double* x_input, const double* y_input, double* output, const int* strides, const int batch_count, const int offset) { const int b = blockIdx.x * blockDim.x + threadIdx.x; if (b >= batch_count) return; - // Apply offset for streaming or partial batch logic const int actual_b = b + offset; const int start = strides[actual_b]; const int end = strides[actual_b + 1]; @@ -25,7 +25,6 @@ namespace fastdist::cuda::utils { double sum_sq = 0.0; for (int i = 0; i < n; ++i) { - // Indexing into flattened segments using start + i const double xv = x_input[start + i]; const double yv = y_input[start + i]; @@ -41,7 +40,8 @@ namespace fastdist::cuda::utils { output[actual_b] = sqrt(sum_sq); } - // Dispatcher: Manually inlined executor logic to bypass MSVC template issues + // Host entry point. Manages the device lifecycle directly rather than through + // the executor.cuh template, which did not build for this signature under MSVC. void euclidean_distance_dispatcher(const double* x_input, const double* y_input, double* output, const int* strides, const int batch_count) { if (batch_count <= 0) return; @@ -55,7 +55,6 @@ namespace fastdist::cuda::utils { const size_t stridesSize = (batch_count + 1) * sizeof(int); try { - // Device Allocations if (cudaMalloc(&d_x, inputSize) != cudaSuccess) throw std::runtime_error("cudaMalloc d_x failed"); if (cudaMalloc(&d_y, inputSize) != cudaSuccess) throw std::runtime_error("cudaMalloc d_y failed"); if (cudaMalloc(&d_output, outputSize) != cudaSuccess) @@ -63,25 +62,20 @@ namespace fastdist::cuda::utils { if (cudaMalloc(&d_strides, stridesSize) != cudaSuccess) throw std::runtime_error("cudaMalloc d_strides failed"); - // Transfers to Device cudaMemcpy(d_x, x_input, inputSize, cudaMemcpyHostToDevice); cudaMemcpy(d_y, y_input, inputSize, cudaMemcpyHostToDevice); cudaMemcpy(d_strides, strides, stridesSize, cudaMemcpyHostToDevice); - // Kernel Config constexpr int threadsPerBlock = 256; const int blocksPerGrid = (batch_count + threadsPerBlock - 1) / threadsPerBlock; const int offset = 0; - euclidean_distance_kernel<<>>(d_x, d_y, d_output, d_strides, batch_count, offset); - // Verify Launch and Sync if (cudaGetLastError() != cudaSuccess) throw std::runtime_error("Euclidean kernel launch failed"); if (cudaDeviceSynchronize() != cudaSuccess) throw std::runtime_error("Euclidean kernel execution failed"); - // Transfer Result back to Host cudaMemcpy(output, d_output, outputSize, cudaMemcpyDeviceToHost); } catch (...) { diff --git a/src/cuda/utils/manhattan_distance.cu b/src/cuda/utils/manhattan_distance.cu index cc1da4c..d370d05 100644 --- a/src/cuda/utils/manhattan_distance.cu +++ b/src/cuda/utils/manhattan_distance.cu @@ -10,13 +10,13 @@ namespace fastdist::cuda::utils { - // CUDA kernel: Logic for summing absolute differences + // One thread per vector pair: sum of absolute differences. + // strides[b]..strides[b + 1] delimits pair b in the flattened inputs. __global__ void manhattan_distance_kernel(const double* x_input, const double* y_input, double* output, const int* strides, const int batch_count, const int offset) { const int b = blockIdx.x * blockDim.x + threadIdx.x; if (b >= batch_count) return; - // Apply offset to batch index const int actual_b = b + offset; const int start = strides[actual_b]; const int end = strides[actual_b + 1]; @@ -25,7 +25,6 @@ namespace fastdist::cuda::utils { double sum_abs = 0.0; for (int i = 0; i < n; ++i) { - // Indexing into the flattened input arrays using start + i const double xv = x_input[start + i]; const double yv = y_input[start + i]; @@ -40,7 +39,8 @@ namespace fastdist::cuda::utils { output[actual_b] = sum_abs; } - // Dispatcher: Concrete implementation that handles the CUDA lifecycle + // Host entry point. Manages the device lifecycle directly rather than through + // the executor.cuh template, which did not build for this signature under MSVC. void manhattan_distance_dispatcher(const double* x_input, const double* y_input, double* output, const int* strides, const int batch_count) { if (batch_count <= 0) return; @@ -54,7 +54,6 @@ namespace fastdist::cuda::utils { const size_t stridesSize = (batch_count + 1) * sizeof(int); try { - // Allocate Device Memory if (cudaMalloc(&d_x, inputSize) != cudaSuccess) throw std::runtime_error("cudaMalloc d_x failed"); if (cudaMalloc(&d_y, inputSize) != cudaSuccess) throw std::runtime_error("cudaMalloc d_y failed"); if (cudaMalloc(&d_output, outputSize) != cudaSuccess) @@ -62,12 +61,10 @@ namespace fastdist::cuda::utils { if (cudaMalloc(&d_strides, stridesSize) != cudaSuccess) throw std::runtime_error("cudaMalloc d_strides failed"); - // Host to Device Transfer cudaMemcpy(d_x, x_input, inputSize, cudaMemcpyHostToDevice); cudaMemcpy(d_y, y_input, inputSize, cudaMemcpyHostToDevice); cudaMemcpy(d_strides, strides, stridesSize, cudaMemcpyHostToDevice); - // Launch Kernel (Parallelizing over the number of batches) constexpr int threadsPerBlock = 256; const int blocksPerGrid = (batch_count + threadsPerBlock - 1) / threadsPerBlock; const int offset = 0; @@ -75,15 +72,12 @@ namespace fastdist::cuda::utils { manhattan_distance_kernel<<>>(d_x, d_y, d_output, d_strides, batch_count, offset); - // Error Checking & Synchronization if (cudaGetLastError() != cudaSuccess) throw std::runtime_error("Manhattan kernel launch failed"); if (cudaDeviceSynchronize() != cudaSuccess) throw std::runtime_error("Manhattan kernel execution failed"); - // Device to Host Transfer cudaMemcpy(output, d_output, outputSize, cudaMemcpyDeviceToHost); } catch (...) { - // Cleanup on Error cudaFree(d_x); cudaFree(d_y); cudaFree(d_output); @@ -91,7 +85,6 @@ namespace fastdist::cuda::utils { throw; } - // Standard Cleanup cudaFree(d_x); cudaFree(d_y); cudaFree(d_output); From 2af0e185d70c5d39f0f08dc350b93e46274e4c5d Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:18:39 -0400 Subject: [PATCH 04/20] Make the C++ comments describe the code as it is Comments only, plus one include swap; no behaviour change. Corrected comments that were wrong: - normal.h described normal_logpdf_scalar as the PDF of the log-normal distribution. It is the natural log of the normal PDF, a different function. - binomial.cpp said the PMF goes through log space "for efficiency"; the reason is that the binomial coefficient overflows otherwise. - uniform.h called the CDF a "cumulative density function", and three headers used the non-standard "cumulative mass function (CMF)". - The gamma continued fraction was labelled only "via Lentz's method"; it computes the upper incomplete gamma and returns 1 - Q. - was included "for size_t", which lives in . - constants.h carried an IDE-generated include guard naming a Windows .pyd. Removed history. Several comments, mostly written while fixing bugs, told the story of the old defect -- "at the previous ceiling of 100", "came out as -2147483647.5", "used to run once per element", "see BENCHMARKS.md" -- rather than what the code does and why. That belongs in the commits and CHANGELOG, where it already is. Each now states the current reason in a line or two. Removed narration: three copies of the same validation note in normal.cpp, "// Basic validity check" above one-line checks, and step-by-step comments in the CPU wrapper that restated the next line. Added one thing that was missing: stepSize was documented nowhere. Each header with batch functions now says output[i] = f(x_data[i] + stepSize * i) and that invalid parameters make every output NaN. Co-Authored-By: Claude Opus 5 --- include/fastdist/config.h | 23 ++++++++--------------- include/fastdist/math/bernoulli.h | 6 ++++-- include/fastdist/math/constants.h | 8 ++++---- include/fastdist/math/discrete_uniform.h | 2 +- include/fastdist/math/exponential.h | 6 ++++-- include/fastdist/math/geometric.h | 2 +- include/fastdist/math/normal.h | 10 ++++++---- include/fastdist/math/poisson.h | 7 +++++-- include/fastdist/math/uniform.h | 11 +++++++---- include/fastdist/math/utils.h | 2 +- src/bindings/bindings.cpp | 2 +- src/math/bernoulli.cpp | 3 ++- src/math/beta.cpp | 7 ++----- src/math/binomial.cpp | 9 ++++----- src/math/exponential.cpp | 3 ++- src/math/gamma.cpp | 11 ++++------- src/math/negative_binomial.cpp | 14 ++++---------- src/math/normal.cpp | 24 ++++++------------------ src/math/poisson.cpp | 11 ++++------- src/math/uniform.cpp | 12 +++--------- src/wrappers/wrapper_utility.h | 8 ++------ 21 files changed, 75 insertions(+), 106 deletions(-) diff --git a/include/fastdist/config.h b/include/fastdist/config.h index 459abe4..cdeaf6d 100644 --- a/include/fastdist/config.h +++ b/include/fastdist/config.h @@ -4,23 +4,16 @@ #define CONFIG_H // Iteration ceiling for the Beta and Gamma series and continued fractions. -// -// Every loop using this exits as soon as its term falls below EPS, so the bound -// only matters for parameters that converge slowly, and raising it costs -// ordinary calls nothing. -// -// The gamma series is the binding constraint: near x = alpha it needs roughly -// sqrt(2 * alpha * ln(1/EPS)) terms, about 227 at alpha = 1000 and 683 at -// alpha = 10000. At the previous ceiling of 100 it simply stopped early and -// returned the truncated sum, so Gamma(1000, 0.5).cdf(500) was wrong by 9e-4 -// and Gamma(10000, ...) by 0.16, with no indication anything had gone wrong. -// -// 1000 covers alpha up to roughly 20000. Beyond that the result degrades -// silently again; a shape parameter that large needs a different algorithm -// (a normal approximation, or Temme's uniform asymptotic expansion) rather -// than a larger ceiling. +// Every loop stops as soon as its term falls below EPS, so the ceiling only +// costs anything for slowly converging parameters. The gamma series is the +// binding case: near x = alpha it needs about sqrt(2 alpha ln(1/EPS)) terms, +// so 1000 covers alpha up to roughly 20000. Beyond that the truncated sum is +// returned without warning; shapes that large need an asymptotic method such +// as Temme's expansion rather than a higher ceiling. constexpr unsigned int MAX_ITER = 1000; +// Relative convergence tolerance for those loops. constexpr double EPS = 1e-12; +// Floor that keeps Lentz's method from dividing by an exact zero. constexpr double FPMIN = 1e-30; #endif // CONFIG_H diff --git a/include/fastdist/math/bernoulli.h b/include/fastdist/math/bernoulli.h index 49305f0..20e5fd9 100644 --- a/include/fastdist/math/bernoulli.h +++ b/include/fastdist/math/bernoulli.h @@ -2,7 +2,7 @@ #ifndef BERNOULLI_H #define BERNOULLI_H -#include // For size_t +#include // size_t // Bernoulli distribution is discrete, so we use PMF instead of PDF namespace fastdist::math { @@ -23,7 +23,9 @@ namespace fastdist::math { // Computes random sample from Bernoulli distribution int bernoulli_sample(double p); - // Batch Functions + // Batch functions: output[i] = f(x_data[i] + stepSize * i) for i in [0, n). + // A stepSize of 0 evaluates x_data as given. Invalid parameters make every + // output NaN. void bernoulli_pmf_batch(const int* k_data, double* output, size_t n, double p, int stepSize); void bernoulli_cdf_batch(const int* k_data, double* output, size_t n, double p, int stepSize); void bernoulli_mgf_batch(const double* t_data, double* output, size_t n, double p, int stepSize); diff --git a/include/fastdist/math/constants.h b/include/fastdist/math/constants.h index fc166cf..408c263 100644 --- a/include/fastdist/math/constants.h +++ b/include/fastdist/math/constants.h @@ -1,10 +1,10 @@ -// /include/fastdist/math/constants.h +// Mathematical constants shared by the distribution implementations -#ifndef FASTDIST_CP314_WIN_AMD64_PYD_CONSTANTS_H -#define FASTDIST_CP314_WIN_AMD64_PYD_CONSTANTS_H +#ifndef FASTDIST_MATH_CONSTANTS_H +#define FASTDIST_MATH_CONSTANTS_H #define SQRT_2PI 2.50662827463100050241576528481104525 #define LOG_SQRT_2PI 0.91893853320467274178 #define M_PI 3.14159265358979323846 -#endif // FASTDIST_CP314_WIN_AMD64_PYD_CONSTANTS_H +#endif // FASTDIST_MATH_CONSTANTS_H diff --git a/include/fastdist/math/discrete_uniform.h b/include/fastdist/math/discrete_uniform.h index 2d08486..481258d 100644 --- a/include/fastdist/math/discrete_uniform.h +++ b/include/fastdist/math/discrete_uniform.h @@ -6,7 +6,7 @@ namespace fastdist::math { // Computes the probability mass function (PMF) of the discrete uniform distribution double discrete_uniform_pmf_scalar(int x, int a, int b); - // Computes the cumulative mass function (CMF) of the discrete uniform distribution + // Computes the cumulative distribution function (CDF) of the discrete uniform distribution double discrete_uniform_cdf_scalar(int x, int a, int b); // Computes the mean of the discrete uniform distribution double discrete_uniform_mean(int a, int b); diff --git a/include/fastdist/math/exponential.h b/include/fastdist/math/exponential.h index d2cf9f4..c893f33 100644 --- a/include/fastdist/math/exponential.h +++ b/include/fastdist/math/exponential.h @@ -2,7 +2,7 @@ #ifndef EXPONENTIAL_H #define EXPONENTIAL_H -#include // For size_t +#include // size_t namespace fastdist::math { // Computes the probability density function (PDF) of the exponential distribution @@ -22,7 +22,9 @@ namespace fastdist::math { // Computes random sample from exponential distribution double exponential_sample(double lambda); - // Batch Functions + // Batch functions: output[i] = f(x_data[i] + stepSize * i) for i in [0, n). + // A stepSize of 0 evaluates x_data as given. Invalid parameters make every + // output NaN. void exponential_pdf_batch(const double* x_data, double* output, size_t n, double lambda, double stepSize); void exponential_cdf_batch(const double* x_data, double* output, size_t n, double lambda, double stepSize); void exponential_mgf_batch(const double* t_data, double* output, size_t n, double lambda, double stepSize); diff --git a/include/fastdist/math/geometric.h b/include/fastdist/math/geometric.h index c9f91ec..303bbf3 100644 --- a/include/fastdist/math/geometric.h +++ b/include/fastdist/math/geometric.h @@ -6,7 +6,7 @@ namespace fastdist::math { // Computes the probability mass function (PMF) of the geometric distribution double geometric_pmf_scalar(int k, double p); - // Computes the cumulative mass function (CMF) of the geometric distribution + // Computes the cumulative distribution function (CDF) of the geometric distribution double geometric_cdf_scalar(int k, double p); // Computes the mean of the geometric distribution double geometric_mean(double p); diff --git a/include/fastdist/math/normal.h b/include/fastdist/math/normal.h index 863012f..bec1a53 100644 --- a/include/fastdist/math/normal.h +++ b/include/fastdist/math/normal.h @@ -2,12 +2,12 @@ #ifndef NORMAL_H #define NORMAL_H -#include // For size_t +#include // size_t namespace fastdist::math { // Computes the probability density function (PDF) of the normal distribution double normal_pdf_scalar(double x, double mu, double sigma); - // Computes the probability density function (PDF) of the log-normal distribution + // Computes the natural log of the normal PDF (not the log-normal density) double normal_logpdf_scalar(double x, double mu, double sigma); // Computes the cumulative distribution function (CDF) of the normal distribution double normal_cdf_scalar(double x, double mu, double sigma); @@ -23,12 +23,14 @@ namespace fastdist::math { double normal_cgf_scalar(double t, double mu, double sigma); // Computes random sample from normal distribution double normal_sample(double mu, double sigma); - // Creates a random sample from log normal distribution + // Draws a sample from the log-normal distribution: exp(X), X ~ N(mu, sigma^2) double normal_log_sample(double mu, double sigma); // Computes the z-score for a given x in the normal distribution double z_score(double x, double mu, double sigma); - // Batch Functions + // Batch functions: output[i] = f(x_data[i] + stepSize * i) for i in [0, n). + // A stepSize of 0 evaluates x_data as given. Invalid parameters make every + // output NaN. void normal_pdf_batch(const double* x_data, double* output, size_t n, double mu, double sigma, double stepSize = 0); void normal_logpdf_batch(const double* x_data, double* output, size_t n, double mu, double sigma, double stepSize = 0); diff --git a/include/fastdist/math/poisson.h b/include/fastdist/math/poisson.h index d94603c..6f53dcc 100644 --- a/include/fastdist/math/poisson.h +++ b/include/fastdist/math/poisson.h @@ -2,13 +2,13 @@ #ifndef POISSON_H #define POISSON_H -#include // For size_t +#include // size_t // Poisson distribution is discrete, so we use PMF instead of PDF namespace fastdist::math { // Computes the probability mass function (PMF) of the poisson distribution double poisson_pmf_scalar(double x, double lambda); - // Computes the cumulative mass function (CMF) of the poisson distribution + // Computes the cumulative distribution function (CDF) of the poisson distribution double poisson_cdf_scalar(double x, double lambda); // Computes the mean of the poisson distribution double poisson_mean(double lambda); @@ -23,6 +23,9 @@ namespace fastdist::math { // Computes a random sample from the Poisson distribution int poisson_sample(double lambda); + // Batch functions: output[i] = f(x_data[i] + stepSize * i) for i in [0, n). + // A stepSize of 0 evaluates x_data as given. Invalid parameters make every + // output NaN. void poisson_pmf_batch(const double* x_data, double* output, size_t n, double lambda, int stepSize = 0); void poisson_cdf_batch(const double* x_data, double* output, size_t n, double lambda, int stepSize = 0); void poisson_mgf_batch(const double* t_data, double* output, size_t n, double lambda, int stepSize = 0); diff --git a/include/fastdist/math/uniform.h b/include/fastdist/math/uniform.h index 590a9b6..4c2aa0c 100644 --- a/include/fastdist/math/uniform.h +++ b/include/fastdist/math/uniform.h @@ -2,14 +2,14 @@ #ifndef UNIFORM_H #define UNIFORM_H -#include // For size_t +#include // size_t -// Note: Uniform files all by default refer to continuous uniform distribution -// Continuous uniform distribution is continuous, so we use PDF instead of PMF +// Continuous uniform distribution on [a, b]. See discrete_uniform.h for the +// integer-valued case. namespace fastdist::math { // Computes the probability density function (PDF) of the continuous uniform distribution double uniform_pdf_scalar(double x, double a, double b); - // Computes the cumulative density function (CDF) of the continuous uniform distribution + // Computes the cumulative distribution function (CDF) of the continuous uniform distribution double uniform_cdf_scalar(double x, double a, double b); // Computes the mean of the continuous uniform distribution double uniform_mean(double a, double b); @@ -24,6 +24,9 @@ namespace fastdist::math { // Computes a random sample from the continuous uniform distribution double uniform_sample(double a, double b); + // Batch functions: output[i] = f(x_data[i] + stepSize * i) for i in [0, n). + // A stepSize of 0 evaluates x_data as given. Invalid parameters make every + // output NaN. void uniform_pdf_batch(const double* x_data, double* output, size_t n, double a, double b, double stepSize = 0.0); void uniform_cdf_batch(const double* x_data, double* output, size_t n, double a, double b, double stepSize = 0.0); void uniform_mgf_batch(const double* t_data, double* output, size_t n, double a, double b, double stepSize = 0.0); diff --git a/include/fastdist/math/utils.h b/include/fastdist/math/utils.h index b4a83dd..4fdcfd8 100644 --- a/include/fastdist/math/utils.h +++ b/include/fastdist/math/utils.h @@ -2,7 +2,7 @@ #ifndef UTILS_H #define UTILS_H -#include // For size_t +#include // size_t #include namespace fastdist::math { diff --git a/src/bindings/bindings.cpp b/src/bindings/bindings.cpp index 2d033a9..db965dd 100644 --- a/src/bindings/bindings.cpp +++ b/src/bindings/bindings.cpp @@ -1,4 +1,4 @@ -// CPP file to link all other bindings +// Defines the _fastdist extension module and registers every binding group #include #include #include diff --git a/src/math/bernoulli.cpp b/src/math/bernoulli.cpp index a63da08..9c29930 100644 --- a/src/math/bernoulli.cpp +++ b/src/math/bernoulli.cpp @@ -79,7 +79,8 @@ namespace fastdist::math { return dist(rng()) ? 1 : 0; } - // Batch Functions + // Batch functions evaluate at x_data[i] + stepSize * i. Invalid parameters + // make every output NaN; a non-finite input makes only its own output NaN. void bernoulli_pmf_batch(const int* k_data, double* output, const size_t n, const double p, const int stepSize) { for (size_t i = 0; i < n; i++) { output[i] = bernoulli_pmf_scalar(k_data[i] + stepSize * static_cast(i), p); diff --git a/src/math/beta.cpp b/src/math/beta.cpp index 31404b9..9f7269c 100644 --- a/src/math/beta.cpp +++ b/src/math/beta.cpp @@ -13,7 +13,6 @@ namespace fastdist::math { // f(x) = x^(α-1) * (1-x)^(β-1) / B(α,β) // ------------------------- double beta_pdf_scalar(const double x, const double alpha, const double beta) { - // Parameter validation if (!std::isfinite(x) || !std::isfinite(alpha) || !std::isfinite(beta) || alpha <= 0.0 || beta <= 0.0) { return std::numeric_limits::quiet_NaN(); } @@ -95,9 +94,7 @@ namespace fastdist::math { // RNG // ------------------------- double beta_sample(const double alpha, const double beta) { - // Every other sampler validates its parameters; this one did not, and - // std::gamma_distribution has undefined behaviour for a non-positive - // shape rather than a defined error value. + // std::gamma_distribution is undefined for a non-positive shape. if (!std::isfinite(alpha) || !std::isfinite(beta) || alpha <= 0.0 || beta <= 0.0) { return std::numeric_limits::quiet_NaN(); } @@ -110,7 +107,7 @@ namespace fastdist::math { } // ------------------------- - // Internal: incomplete beta series + // Internal: continued fraction for the incomplete beta // ------------------------- // Modified Lentz evaluation of the continued fraction for the incomplete // beta function (Numerical Recipes 6.4). Each iteration applies two diff --git a/src/math/binomial.cpp b/src/math/binomial.cpp index d2be9bb..d705615 100644 --- a/src/math/binomial.cpp +++ b/src/math/binomial.cpp @@ -8,7 +8,7 @@ namespace fastdist::math { - // Computes log PMF + // log P(X = x), through lgamma so the binomial coefficient cannot overflow double binomial_logpmf_scalar(const int x, const int n, const double p) { if (!std::isfinite(p) || p < 0.0 || p > 1.0 || n < 0) { return std::numeric_limits::quiet_NaN(); @@ -22,12 +22,12 @@ namespace fastdist::math { return log_coeff + x * std::log(p) + (n - x) * std::log1p(-p); } - // PMF uses log PMF for efficiency + // Exponentiates the log PMF, which is what keeps large n in range double binomial_pmf_scalar(const int x, const int n, const double p) { return std::exp(binomial_logpmf_scalar(x, n, p)); } - // CDF sums PMF for k = 0..x + // P(X <= x) double binomial_cdf_scalar(const int x, const int n, const double p) { if (!std::isfinite(p) || p < 0.0 || p > 1.0 || n < 0) { return std::numeric_limits::quiet_NaN(); @@ -42,8 +42,7 @@ namespace fastdist::math { // Consecutive PMF terms satisfy // P(k) = P(k-1) * ((n - k + 1) / k) * (p / (1 - p)) - // so the sum costs one exp overall instead of three lgammas, two logs - // and an exp per term. + // so the whole sum costs a single exp. // // P(0) = (1-p)^n underflows for large n, which would collapse the whole // recurrence to zero; fall back to per-term log-space evaluation there. diff --git a/src/math/exponential.cpp b/src/math/exponential.cpp index 8d99866..790779b 100644 --- a/src/math/exponential.cpp +++ b/src/math/exponential.cpp @@ -77,7 +77,8 @@ namespace fastdist::math { return dist(rng()); } - // Batch Functions + // Batch functions evaluate at x_data[i] + stepSize * i. Invalid parameters + // make every output NaN; a non-finite input makes only its own output NaN. void exponential_pdf_batch(const double* x_data, double* output, const size_t n, const double lambda, const double stepSize) { if (!std::isfinite(lambda) || lambda <= 0.0) { diff --git a/src/math/gamma.cpp b/src/math/gamma.cpp index bfdfc5a..477b29f 100644 --- a/src/math/gamma.cpp +++ b/src/math/gamma.cpp @@ -100,7 +100,7 @@ namespace fastdist::math { } // ------------------------- - // Internal: lower incomplete gamma series representation + // Internal: lower incomplete gamma P(a, x) by its power series // ------------------------- static double gamma_p_series(const double a, const double x) { double sum = 1.0 / a; @@ -116,7 +116,8 @@ namespace fastdist::math { } // ------------------------- - // Internal functions: continued fraction representation via Lentz's method + // Internal: upper incomplete gamma Q(a, x) by modified-Lentz continued + // fraction (Numerical Recipes 6.2), returned as P = 1 - Q // ------------------------- static double gamma_p_cf(const double a, const double x) { double b = x + 1.0 - a; @@ -125,11 +126,7 @@ namespace fastdist::math { double h = d; for (unsigned int i = 1; i <= MAX_ITER; ++i) { - // i is converted to double *before* the negation. Written as - // -i * (i - a), the unary minus applies to the unsigned loop - // index and wraps to 2^32 - i, so the first coefficient came out - // as -2147483647.5 instead of 0.5 and the whole fraction was - // wrong -- returning probabilities above 1.0. + // Convert before negating: -i on the unsigned index would wrap. const double di = static_cast(i); const double an = -di * (di - a); b += 2.0; diff --git a/src/math/negative_binomial.cpp b/src/math/negative_binomial.cpp index 4cccb47..4bbe11c 100644 --- a/src/math/negative_binomial.cpp +++ b/src/math/negative_binomial.cpp @@ -13,7 +13,6 @@ namespace fastdist::math { // k = number of failures, r = number of successes, p = success probability // ------------------------- double negative_binomial_pmf_scalar(const int k, const int r, const double p) { - // Parameter validation if (!std::isfinite(p) || p <= 0.0 || p >= 1.0 || r <= 0) { return std::numeric_limits::quiet_NaN(); } @@ -23,12 +22,8 @@ namespace fastdist::math { return 0.0; } - // Evaluated in log space. Forming C(k + r - 1, k) from raw factorials - // overflows a double once k + r - 1 > 170 -- inf at k = 170 and nan - // beyond -- even though the coefficient and the resulting PMF are - // comfortably inside range (for r = 3, k = 200 the true PMF is 1.6e-57). - // lgamma keeps the intermediate values small, and folding the two pows - // into the same exponent removes them from the hot path. + // Evaluated in log space: forming C(k + r - 1, k) from factorials + // overflows once k + r - 1 > 170, long before the PMF itself does. const double log_pmf = std::lgamma(static_cast(k) + r) - std::lgamma(static_cast(r)) - std::lgamma(static_cast(k) + 1.0) + r * std::log(p) + static_cast(k) * std::log1p(-p); @@ -50,8 +45,7 @@ namespace fastdist::math { // Consecutive PMF terms satisfy // P(i) = P(i-1) * ((i + r - 1) / i) * (1 - p) - // so the sum costs one exp overall instead of three tgammas and two - // pows per term. + // so the whole sum costs a single exp. // // P(0) = p^r underflows for small p with large r, which would collapse // the recurrence to zero; fall back to per-term evaluation there. @@ -131,7 +125,7 @@ namespace fastdist::math { } // ------------------------- - // Random sample using standard library + // X ~ NegativeBinomial(r, p): failures before the r-th success // ------------------------- int negative_binomial_sample(const int r, const double p) { if (!std::isfinite(p) || p <= 0.0 || p >= 1.0 || r <= 0) { diff --git a/src/math/normal.cpp b/src/math/normal.cpp index 850bfbb..85f3429 100644 --- a/src/math/normal.cpp +++ b/src/math/normal.cpp @@ -10,14 +10,10 @@ namespace fastdist::math { namespace { - // The scalar formulas with parameter validation and every loop-invariant - // term lifted into arguments, so the batch paths can compute those once - // instead of once per element. Scalar and batch both route through these, - // so there is still only one copy of each formula. - // - // The arithmetic is arranged exactly as the scalar versions had it -- - // same operations in the same order -- so hoisting does not perturb - // rounding and the results are bit-identical to before. + // Formula cores shared by the scalar and batch paths. Every term that + // depends only on the parameters is an argument, so a batch call computes + // it once per array. Both paths order the arithmetic identically, so they + // agree bit for bit. inline double normal_pdf_core(const double x, const double mu, const double sigma, const double denom) { const double z = (x - mu) / sigma; return std::exp(-0.5 * z * z) / denom; @@ -34,7 +30,6 @@ namespace fastdist::math { } } // namespace - double normal_pdf_scalar(const double x, const double mu, const double sigma) { if (!std::isfinite(x) || !std::isfinite(mu) || !std::isfinite(sigma) || sigma <= 0.0) { return std::numeric_limits::quiet_NaN(); @@ -112,11 +107,10 @@ namespace fastdist::math { double z_score(const double x, const double mu, const double sigma) { return (x - mu) / sigma; } - // Batch Functions + // Batch functions evaluate at x_data[i] + stepSize * i. Invalid parameters + // make every output NaN; a non-finite input makes only its own output NaN. void normal_pdf_batch(const double* x_data, double* output, const size_t n, const double mu, const double sigma, const double stepSize) { - // Parameter validity does not vary across the array, so it is checked - // once here rather than on every element. if (!std::isfinite(mu) || !std::isfinite(sigma) || sigma <= 0.0) { std::fill_n(output, n, std::numeric_limits::quiet_NaN()); return; @@ -136,15 +130,11 @@ namespace fastdist::math { void normal_logpdf_batch(const double* x_data, double* output, const size_t n, const double mu, const double sigma, const double stepSize) { - // Parameter validity does not vary across the array, so it is checked - // once here rather than on every element. if (!std::isfinite(mu) || !std::isfinite(sigma) || sigma <= 0.0) { std::fill_n(output, n, std::numeric_limits::quiet_NaN()); return; } - // log(sigma) in particular is a transcendental call that used to run - // once per element for a value that never changes. const double inv_sigma = 1.0 / sigma; const double log_sigma = std::log(sigma); @@ -160,8 +150,6 @@ namespace fastdist::math { void normal_cdf_batch(const double* x_data, double* output, const size_t n, const double mu, const double sigma, const double stepSize) { - // Parameter validity does not vary across the array, so it is checked - // once here rather than on every element. if (!std::isfinite(mu) || !std::isfinite(sigma) || sigma <= 0.0) { std::fill_n(output, n, std::numeric_limits::quiet_NaN()); return; diff --git a/src/math/poisson.cpp b/src/math/poisson.cpp index 670b4d7..f913326 100644 --- a/src/math/poisson.cpp +++ b/src/math/poisson.cpp @@ -38,9 +38,7 @@ namespace fastdist::math { const int ki = static_cast(std::floor(x)); // Consecutive PMF terms are related by P(i) = P(i-1) * lambda / i, so - // the sum needs one exp in total rather than a log, an lgamma and an - // exp per term. That is the difference between this being the slowest - // path in the library and it being competitive -- see BENCHMARKS.md. + // the whole sum costs a single exp. // // The recurrence has to start from P(0) = exp(-lambda), which underflows // to zero for large lambda and would collapse the whole sum to zero even @@ -115,12 +113,11 @@ namespace fastdist::math { return dist(rng()); } - // Batch Functions + // Batch functions evaluate at x_data[i] + stepSize * i. Invalid parameters + // make every output NaN; a non-finite input makes only its own output NaN. void poisson_pmf_batch(const double* x_data, double* output, const size_t n, const double lambda, const int stepSize) { - // lambda is fixed across the array, so both its validation and log() are - // hoisted; log(lambda) used to be a transcendental call per element for a - // value that never changes. + // lambda is fixed across the array: validate it and take its log once. if (!std::isfinite(lambda) || lambda <= 0.0) { std::fill_n(output, n, std::numeric_limits::quiet_NaN()); return; diff --git a/src/math/uniform.cpp b/src/math/uniform.cpp index e18af33..43e5dd6 100644 --- a/src/math/uniform.cpp +++ b/src/math/uniform.cpp @@ -10,7 +10,6 @@ namespace fastdist::math { double uniform_pdf_scalar(const double x, const double a, const double b) { - // Check parameters: a < b, finite numbers if (!std::isfinite(a) || !std::isfinite(b) || a >= b || !std::isfinite(x)) { return std::numeric_limits::quiet_NaN(); } @@ -24,7 +23,6 @@ namespace fastdist::math { } double uniform_cdf_scalar(const double x, const double a, const double b) { - // Check parameters if (!std::isfinite(a) || !std::isfinite(b) || a >= b || !std::isfinite(x)) { return std::numeric_limits::quiet_NaN(); } @@ -36,7 +34,6 @@ namespace fastdist::math { } double uniform_mean(const double a, const double b) { - // Basic validity check if (!std::isfinite(a) || !std::isfinite(b) || a >= b) { return std::numeric_limits::quiet_NaN(); } @@ -45,7 +42,6 @@ namespace fastdist::math { } double uniform_variance(const double a, const double b) { - // Basic validity check if (!std::isfinite(a) || !std::isfinite(b) || a >= b) { return std::numeric_limits::quiet_NaN(); } @@ -54,7 +50,6 @@ namespace fastdist::math { } double uniform_stddev(const double a, const double b) { - // Basic validity check if (!std::isfinite(a) || !std::isfinite(b) || a >= b) { return std::numeric_limits::quiet_NaN(); } @@ -95,12 +90,11 @@ namespace fastdist::math { return dist(rng()); } - // Batch Functions + // Batch functions evaluate at x_data[i] + stepSize * i. Invalid parameters + // make every output NaN; a non-finite input makes only its own output NaN. void uniform_pdf_batch(const double* x_data, double* output, const size_t n, const double a, const double b, const double stepSize) { - // The density is constant across the support, so the whole value -- not - // just the validation -- is loop-invariant. This used to be a division - // per element for a number that never changes. + // The density is constant on [a, b], so it is computed once. if (!std::isfinite(a) || !std::isfinite(b) || a >= b) { std::fill_n(output, n, std::numeric_limits::quiet_NaN()); return; diff --git a/src/wrappers/wrapper_utility.h b/src/wrappers/wrapper_utility.h index d9569f9..4006644 100644 --- a/src/wrappers/wrapper_utility.h +++ b/src/wrappers/wrapper_utility.h @@ -14,10 +14,10 @@ namespace py = pybind11; namespace fastdist::wrapper { - // CPU Implementation + // Runs a CPU batch function over a 1-D contiguous numpy array: validates the + // input, allocates a same-length output, and releases the GIL for the loop. template py::array_t run_cpu_wrapper(BatchFn fn, const py::array_t& input, Args&&... args) { - // Get the array's(input's) information const auto buf = input.request(); if (buf.ndim != 1) { @@ -27,16 +27,12 @@ namespace fastdist::wrapper { throw std::runtime_error("Input array must be contiguous"); } - // Make a numpy array of x's size auto result = py::array_t(buf.size); - // Get the array's('result') information const auto result_buf = result.request(); - // Creates the in and out pointers required for sending and receiving data const auto* in_ptr = static_cast(buf.ptr); auto* out_ptr = static_cast(result_buf.ptr); - // Gets the size of the array const auto n = static_cast(buf.shape[0]); py::gil_scoped_release release; From f17e195a35bb5de2da5b5603bae418542f92c30c Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:18:40 -0400 Subject: [PATCH 05/20] Bring the CHANGELOG up to date for the next release The [Unreleased] section predated most of the work on this branch, and was wrong in both directions: it listed three problems as known issues after they were fixed (the negative binomial overflow, the 2 sigma RNG tolerances and the 32-bit seeding), and it said nothing about the beta and gamma CDF defects, the density overflows, the Python setter bugs, seeding, or the benchmark suite. Rewritten against the actual commit history, keeping the existing entries. Fixed entries give the observable symptom so a user can tell whether they were affected. Known issues now lead with the PyPI name collision, which blocks the release workflow by design until the package is renamed. Also corrects the python-distro.yml comment on the C++ test step, which still described the tolerances as "(>=7 sigma)". They are seeded and sized at about 5 sigma now, and the point worth stating is that a failure reproduces. Co-Authored-By: Claude Opus 5 --- .github/workflows/python-distro.yml | 4 +- CHANGELOG.md | 73 +++++++++++++++++++++++++---- 2 files changed, 65 insertions(+), 12 deletions(-) diff --git a/.github/workflows/python-distro.yml b/.github/workflows/python-distro.yml index c623bba..74f3089 100644 --- a/.github/workflows/python-distro.yml +++ b/.github/workflows/python-distro.yml @@ -46,8 +46,8 @@ jobs: - name: Build C++ tests run: cmake --build build --target fastdist_tests --parallel - # RNG tolerances are sized from the estimator standard error (>=7 sigma), - # so a failure here is a real regression rather than a flake. + # The RNG tests are seeded, so a failure here reproduces on every run + # with the same toolchain rather than being a flake. - name: Run C++ tests run: ctest --test-dir build --output-on-failure diff --git a/CHANGELOG.md b/CHANGELOG.md index c89519e..64e7088 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,16 +12,50 @@ call in `CMakeLists.txt`. ### Fixed -- `binomial_cdf_scalar`, `poisson_cdf_scalar` and `negative_binomial_cdf_scalar` returned the raw sum of - PMF terms, which accumulates rounding error and could exceed `1.0`. For the binomial this also made the - CDF non-monotonic, because `x >= n` short-circuits to exactly `1.0` while the values just below it did - not. All three now clamp to `1.0`. +- `beta_cdf_scalar` was wrong at every point, not only at the extremes: 0.0015 against a true 0.1143 for + Beta(2, 5) at x = 0.1, and values outside [0, 1] such as -147 for Beta(0.01, 0.01) at x = 0.5. It is now + the modified-Lentz continued fraction with the standard reflection, and agrees with SciPy to ~1e-12. +- `gamma_cdf_scalar` and `chi_square_cdf_scalar` returned probabilities above 1.0 (Gamma(1.5, 1).cdf(2.5) + gave 1.000498). A unary minus applied to an unsigned loop index wrapped to 2^32 - i inside the continued + fraction. Separately, the series stopped at 100 iterations and silently truncated for large shapes (off + by 0.16 at alpha = 10000); the ceiling is now 1000. +- `beta_pdf_scalar`, `gamma_pdf_scalar` and `chi_square_pdf_scalar` returned `nan` or `inf` once a shape + parameter passed ~171 (k above ~342 for chi-square), because the normalising constants overflowed + `std::tgamma`. All three are now evaluated in log space. +- `negative_binomial_pmf_scalar` returned `inf` at k = 170 and `nan` beyond it, taking the CDF with it, for + the same reason. It is now evaluated in log space. +- `binomial_cdf_scalar`, `poisson_cdf_scalar` and `negative_binomial_cdf_scalar` could exceed `1.0` + through accumulated rounding, which also made the binomial CDF non-monotonic. All three now clamp to + `1.0`. +- `beta_sample` did not validate its parameters, and `std::gamma_distribution` has undefined behaviour for + a non-positive shape. It now returns `nan` for invalid input, like every other continuous sampler. +- Python setters: `Beta.beta` and `Binomial.p` raised `TypeError` for every value, valid or not; + `DiscreteUniform.b` recursed until `RecursionError`; `DiscreteUniform.a` changed the attribute's type to + `float`; and the `Uniform` and `DiscreteUniform` setters accepted a bound that violated `a < b`, after + which every method silently returned `nan`. +- `Utils.law_of_total_probability` rejected scalar arguments despite its signature. Scalars are now + treated as a one-element partition, and sequences of different lengths raise `ValueError`. +- An `ImportError` raised while loading a distribution module, such as a missing numpy, was reported as a + missing C++ core. The original exception is now chained, and the message says how to build the + extension. +- The C++ RNG tests failed in about 7.7% of runs because their tolerances sat near 2σ of the estimator's + own noise (#2). - `setup.py` no longer hardcodes the `Visual Studio 17 2022` CMake generator. CMake selects the newest Visual Studio present, so builds work on machines with a different version installed. Set `CMAKE_GENERATOR` to pin one. ### Added +- `fastdist.seed(value)` and `fastdist.seed_from_entropy()`, with C++ equivalents `seed_rng` and + `seed_rng_from_entropy` in `fastdist/math/rng.h`. Every sampler now draws from one shared thread-local + Mersenne Twister, seeded through `std::seed_seq`, and any signed 64-bit value is accepted. A seed + reproduces a run on one platform and toolchain, and applies to the calling thread only. +- A benchmark suite under `benchmarks/`, and `BENCHMARKS.md` as a performance log generated from its + recorded results. It compares against SciPy, checks numerical agreement before timing, flags regressions + against the run's measured noise, and times the CUDA paths when they are built. +- Type stubs for the compiled extension (`_fastdist.pyi`), with a CI check that keeps them in step with the + bindings. +- A PyPI release workflow using Trusted Publishing, with a TestPyPI dry-run option. - Cross-platform wheel building in CI via `cibuildwheel` — Linux x86_64, Windows AMD64, and macOS x86_64 + arm64, for CPython 3.10 through 3.14. - An install-from-sdist check in CI, exercising the source path an end user takes on any platform without @@ -35,6 +69,22 @@ call in `CMakeLists.txt`. ### Changed +- **Performance.** The Poisson, binomial and negative binomial CDFs sum their terms by recurrence instead + of re-deriving each one, and the batch paths hoist parameter validation and loop-invariant terms out of + their loops. On the reference machine in `BENCHMARKS.md`, `poisson_cdf` over 100k values went from + 43.5 ms to 1.3 ms (from 0.15x SciPy's speed to 4.8x), `normal_logpdf` got 80% faster, and the uniform and + normal CDF paths got 22–29% faster. +- `Utils.sigmoid` is scalar-only and raises a `TypeError` naming `Utils.sigmoid_cpu` when given a sequence. + Its annotation previously advertised sequences, which never worked. +- Annotations use `typing.SupportsFloat` rather than `numbers.Real`, which mypy cannot check, so correct + code such as `Normal(0.0, 1.0)` no longer reports errors. The package now type-checks cleanly. +- The `exponential` and `poisson` bindings name their rate keyword `lambda_`. The old name, `lambda`, is a + Python keyword and could never be passed by name. +- The sampling tests are seeded and deterministic, with tolerances at about 5× the estimator's standard + error, and CI no longer retries failed C++ tests. +- `requirements.txt` is now `requirements-dev.txt`. Runtime dependencies are declared only in the package + metadata. +- Package metadata moved from `setup.py` into `pyproject.toml`. - `NDEBUG` is undefined for the `fastdist_tests` target, so its `assert()`-based checks stay live in Release builds. They were previously compiled away, meaning the suite reported success without testing anything. - Repository layout: bindings moved from `python/bindings` to `src/bindings`; tests split into `tests/cpp` @@ -46,12 +96,15 @@ call in `CMakeLists.txt`. ### Known issues -- `negative_binomial_pmf_scalar` returns `inf` for `k` around 169 and `NaN` beyond it, because the binomial - coefficient is computed with `std::tgamma`, which overflows above ~171. Computing it in log space via - `std::lgamma` is the fix. `beta.cpp` and `gamma.cpp` use `tgamma` similarly. -- Three RNG test assertions have tolerances at roughly 2σ and fail a few percent of runs. CI absorbs this - with `ctest --repeat until-pass:3`. -- The samplers seed `std::mt19937` from a single 32-bit `random_device` word. +- The name `fastdist` belongs to an unrelated project on PyPI, so this package cannot be published under + it. The release workflow refuses to upload until the distribution is renamed. +- Sampling draws one variate per call, which makes bulk generation 30–100x slower than numpy. There is no + batch sampling entry point yet. +- The gamma CDF's iteration ceiling covers shape parameters up to roughly 20000. Beyond that the result + degrades without warning. +- The `*_cpu` bindings do not expose the `step_size` default that the C++ headers declare, and `step_size` + is a `double` for the continuous distributions but an `int` for the discrete ones. +- The CUDA backend is not built or tested in CI. --- From 1838dccaa58686d3f159124344241c58282710ce Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:27:44 -0400 Subject: [PATCH 06/20] Compile CUDA once, as C++17, so the backend builds on Windows Every .cu file failed on Windows with nvcc error : 'cudafe++' died with status 0xC0000409 That was diagnosed earlier as CUDA 12.4 being too old for MSVC 19.42. It was not. Compiling a single kernel directly shows the actual cause: nvcc -std=c++17 ... src/cuda/normal/pdf.cu -> compiles nvcc -std=c++20 ... src/cuda/normal/pdf.cu -> cudafe++ crashes nvcc 12.x's front end cannot parse MSVC's standard-library headers in C++20 mode. CMakeLists already set CUDA_STANDARD 17 -- but only on fastdist_core, and every .cu file was also listed a second time on _fastdist, which links fastdist_core anyway. That second copy inherited the project's C++20, and it was the one that crashed; all 26 failing objects were under _fastdist.dir. CMAKE_CUDA_STANDARD 17 is now set as soon as CUDA is enabled, so no target can compile CUDA as C++20, and the duplicate source list is removed. Each kernel is compiled once instead of twice, which also halves CUDA build time. Verified on Windows with CUDA 12.4 and MSVC 19.42 (Ninja): all 26 kernels compile, the extension links and exposes its *_cuda bindings, ctest passes 14/14, and the GPU and CPU paths agree numerically. Linux was never affected, since GCC's headers do not trip cudafe++; the stub CI job, which builds with CUDA there, checks the deduplicated source list. Co-Authored-By: Claude Opus 5 --- CHANGELOG.md | 6 +++++- CMakeLists.txt | 38 +++++--------------------------------- 2 files changed, 10 insertions(+), 34 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 64e7088..58cf69e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -40,6 +40,9 @@ call in `CMakeLists.txt`. extension. - The C++ RNG tests failed in about 7.7% of runs because their tolerances sat near 2σ of the estimator's own noise (#2). +- The CUDA backend did not compile on Windows. `nvcc` 12.x's front end crashes on MSVC's C++20 + standard-library headers, and every `.cu` file was compiled a second time into the Python module + target, which built as C++20. CUDA sources are now compiled once, as C++17. - `setup.py` no longer hardcodes the `Visual Studio 17 2022` CMake generator. CMake selects the newest Visual Studio present, so builds work on machines with a different version installed. Set `CMAKE_GENERATOR` to pin one. @@ -104,7 +107,8 @@ call in `CMakeLists.txt`. degrades without warning. - The `*_cpu` bindings do not expose the `step_size` default that the C++ headers declare, and `step_size` is a `double` for the continuous distributions but an `int` for the discrete ones. -- The CUDA backend is not built or tested in CI. +- CI compiles the CUDA backend (on Linux, in the type-stub job) but has no GPU runner, so the CUDA + kernels are not exercised there. --- diff --git a/CMakeLists.txt b/CMakeLists.txt index 69dcfce..64707a9 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -21,6 +21,11 @@ option(FASTDIST_ENABLE_CUDA "Enable CUDA backend" OFF) if (FASTDIST_ENABLE_CUDA) enable_language(CUDA) + # CUDA is compiled as C++17 even though the rest of the project is C++20: + # nvcc 12.x's front end (cudafe++) crashes on the MSVC standard library + # headers in C++20 mode. Set globally so no target can pick up C++20. + set(CMAKE_CUDA_STANDARD 17) + set(CMAKE_CUDA_STANDARD_REQUIRED ON) message(STATUS "CUDA backend enabled") else () message(STATUS "CUDA backend disabled") @@ -159,39 +164,6 @@ if (FASTDIST_ENABLE_CUDA) src/cuda/utils/manhattan_distance.cu src/cuda/utils/cosine_similarity.cu ) - target_sources(_fastdist PRIVATE - src/cuda/normal/pdf.cu - src/cuda/normal/logpdf.cu - src/cuda/normal/cdf.cu - src/cuda/normal/mgf.cu - src/cuda/normal/cgf.cu - - src/cuda/poisson/cdf.cu - src/cuda/poisson/cgf.cu - src/cuda/poisson/mgf.cu - src/cuda/poisson/pmf.cu - - src/cuda/bernoulli/cdf.cu - src/cuda/bernoulli/cgf.cu - src/cuda/bernoulli/mgf.cu - src/cuda/bernoulli/pmf.cu - - src/cuda/exponential/cdf.cu - src/cuda/exponential/cgf.cu - src/cuda/exponential/mgf.cu - src/cuda/exponential/pdf.cu - - src/cuda/uniform/cdf.cu - src/cuda/uniform/cgf.cu - src/cuda/uniform/mgf.cu - src/cuda/uniform/pdf.cu - - src/cuda/utils/sigmoid.cu - src/cuda/utils/logit.cu - src/cuda/utils/euclidean_distance.cu - src/cuda/utils/manhattan_distance.cu - src/cuda/utils/cosine_similarity.cu - ) set_target_properties(fastdist_core PROPERTIES CUDA_SEPARABLE_COMPILATION OFF CUDA_STANDARD 17 From 0eb604caa825a85f1ab4ecb2b4c3f319c252286a Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:27:45 -0400 Subject: [PATCH 07/20] Regenerate the extension stub for pybind11 3.1 The Type Stub CI job failed on this branch. None of the difference came from binding changes: it installs pybind11 unpinned, picked up 3.1.0, and 3.1.0 annotates arguments more accurately than the version the committed stub was generated with. Every double parameter widens from SupportsFloat to SupportsFloat | SupportsIndex and every int parameter from SupportsInt to SupportsInt | SupportsIndex -- 156 signatures -- plus three whitespace-only lines. The bindings did always accept ints, so the new hints are the correct ones. This is the regenerated-stub artifact the failing run uploaded, committed as CONTRIBUTING.md describes. mypy over python/fastdist stays clean against it. Because pybind11 floats, a contributor regenerating locally with 3.0.x will get the narrower hints and a diff. Pinning pybind11 in stub.yml would prevent that churn, at the cost of the early warning the floating version gives; that is a policy call left for the maintainers. Co-Authored-By: Claude Opus 5 --- python/fastdist/_fastdist.pyi | 318 +++++++++++++++++----------------- 1 file changed, 159 insertions(+), 159 deletions(-) diff --git a/python/fastdist/_fastdist.pyi b/python/fastdist/_fastdist.pyi index eedd465..58317f5 100644 --- a/python/fastdist/_fastdist.pyi +++ b/python/fastdist/_fastdist.pyi @@ -4,183 +4,183 @@ import numpy import numpy.typing import typing __all__: list[str] = ['bayes_rule', 'bernoulli_cdf_cpu', 'bernoulli_cdf_cuda', 'bernoulli_cdf_scalar', 'bernoulli_cgf_cpu', 'bernoulli_cgf_cuda', 'bernoulli_cgf_scalar', 'bernoulli_mean', 'bernoulli_mgf_cpu', 'bernoulli_mgf_cuda', 'bernoulli_mgf_scalar', 'bernoulli_pmf_cpu', 'bernoulli_pmf_cuda', 'bernoulli_pmf_scalar', 'bernoulli_sample', 'bernoulli_stddev', 'bernoulli_variance', 'beta_cdf_scalar', 'beta_mean', 'beta_pdf_scalar', 'beta_sample', 'beta_stddev', 'beta_variance', 'binomial', 'binomial_cdf_scalar', 'binomial_cgf_scalar', 'binomial_logpmf_scalar', 'binomial_mean', 'binomial_mgf_scalar', 'binomial_pmf_scalar', 'binomial_sample', 'binomial_stddev', 'binomial_variance', 'chebyshev_bound', 'chi_square_cdf_scalar', 'chi_square_cgf_scalar', 'chi_square_mean', 'chi_square_mgf_scalar', 'chi_square_pdf_scalar', 'chi_square_sample', 'chi_square_stddev', 'chi_square_variance', 'choose', 'coefficient_of_variation', 'cosine_similarity', 'cosine_similarity_cuda', 'covariance', 'discrete_uniform_cdf_scalar', 'discrete_uniform_cgf_scalar', 'discrete_uniform_mean', 'discrete_uniform_mgf_scalar', 'discrete_uniform_pmf_scalar', 'discrete_uniform_sample', 'discrete_uniform_stddev', 'discrete_uniform_variance', 'euclidean_distance', 'euclidean_distance_cuda', 'exponential_cdf_cpu', 'exponential_cdf_cuda', 'exponential_cdf_scalar', 'exponential_cgf_cpu', 'exponential_cgf_cuda', 'exponential_cgf_scalar', 'exponential_mean', 'exponential_mgf_cpu', 'exponential_mgf_cuda', 'exponential_mgf_scalar', 'exponential_pdf_cpu', 'exponential_pdf_cuda', 'exponential_pdf_scalar', 'exponential_sample', 'exponential_stddev', 'exponential_variance', 'factorial', 'gamma', 'gamma_cdf_scalar', 'gamma_cgf_scalar', 'gamma_mean', 'gamma_mgf_scalar', 'gamma_pdf_scalar', 'gamma_sample', 'gamma_stddev', 'gamma_variance', 'geometric_cdf_scalar', 'geometric_cgf_scalar', 'geometric_mean', 'geometric_mgf_scalar', 'geometric_pmf_scalar', 'geometric_sample', 'geometric_stddev', 'geometric_variance', 'law_of_total_probability', 'log_gamma', 'logit', 'logit_cpu', 'logit_cuda', 'manhattan_distance', 'manhattan_distance_cuda', 'negative_binomial_cdf_scalar', 'negative_binomial_cgf_scalar', 'negative_binomial_mean', 'negative_binomial_mgf_scalar', 'negative_binomial_pmf_scalar', 'negative_binomial_sample', 'negative_binomial_stddev', 'negative_binomial_variance', 'normal_cdf_cpu', 'normal_cdf_cuda', 'normal_cdf_scalar', 'normal_cgf_cpu', 'normal_cgf_cuda', 'normal_cgf_scalar', 'normal_log_sample', 'normal_logpdf_cpu', 'normal_logpdf_cuda', 'normal_logpdf_scalar', 'normal_mean', 'normal_mgf_cpu', 'normal_mgf_cuda', 'normal_mgf_scalar', 'normal_pdf_cpu', 'normal_pdf_cuda', 'normal_pdf_scalar', 'normal_sample', 'normal_stddev', 'normal_variance', 'permutation', 'poisson_cdf_cpu', 'poisson_cdf_cuda', 'poisson_cdf_scalar', 'poisson_cgf_cpu', 'poisson_cgf_cuda', 'poisson_cgf_scalar', 'poisson_mean', 'poisson_mgf_cpu', 'poisson_mgf_cuda', 'poisson_mgf_scalar', 'poisson_pmf_cpu', 'poisson_pmf_cuda', 'poisson_pmf_scalar', 'poisson_sample', 'poisson_stddev', 'poisson_variance', 'seed', 'seed_from_entropy', 'sigmoid', 'sigmoid_cpu', 'sigmoid_cuda', 'uniform_cdf_cpu', 'uniform_cdf_cuda', 'uniform_cdf_scalar', 'uniform_cgf_cpu', 'uniform_cgf_cuda', 'uniform_cgf_scalar', 'uniform_mean', 'uniform_mgf_cpu', 'uniform_mgf_cuda', 'uniform_mgf_scalar', 'uniform_pdf_cpu', 'uniform_pdf_cuda', 'uniform_pdf_scalar', 'uniform_sample', 'uniform_stddev', 'uniform_variance', 'z_score'] -def bayes_rule(p_B_given_A: typing.SupportsFloat, p_A: typing.SupportsFloat, p_B: typing.SupportsFloat) -> float: +def bayes_rule(p_B_given_A: typing.SupportsFloat | typing.SupportsIndex, p_A: typing.SupportsFloat | typing.SupportsIndex, p_B: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Apply Bayes' rule to compute posterior probability """ -def bernoulli_cdf_cpu(k: typing.Annotated[numpy.typing.ArrayLike, numpy.int32], p: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def bernoulli_cdf_cpu(k: typing.Annotated[numpy.typing.ArrayLike, numpy.int32], p: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute Bernoulli CDF on CPU """ -def bernoulli_cdf_cuda(k: typing.Annotated[numpy.typing.ArrayLike, numpy.int32], p: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def bernoulli_cdf_cuda(k: typing.Annotated[numpy.typing.ArrayLike, numpy.int32], p: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute bernoulli PDF using CUDA (GPU) """ -def bernoulli_cdf_scalar(k: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def bernoulli_cdf_scalar(k: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CDF of Bernoulli distribution """ -def bernoulli_cgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], p: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def bernoulli_cgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], p: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute Bernoulli CGF on CPU """ -def bernoulli_cgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], p: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def bernoulli_cgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], p: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute bernoulli PDF using CUDA (GPU) """ -def bernoulli_cgf_scalar(t: typing.SupportsFloat, p: typing.SupportsFloat) -> float: +def bernoulli_cgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CGF of Bernoulli distribution """ -def bernoulli_mean(p: typing.SupportsFloat) -> float: +def bernoulli_mean(p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute mean of Bernoulli distribution """ -def bernoulli_mgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], p: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def bernoulli_mgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], p: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute Bernoulli MGF on CPU """ -def bernoulli_mgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], p: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def bernoulli_mgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], p: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute bernoulli PDF using CUDA (GPU) """ -def bernoulli_mgf_scalar(t: typing.SupportsFloat, p: typing.SupportsFloat) -> float: +def bernoulli_mgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute MGF of Bernoulli distribution """ -def bernoulli_pmf_cpu(k: typing.Annotated[numpy.typing.ArrayLike, numpy.int32], p: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def bernoulli_pmf_cpu(k: typing.Annotated[numpy.typing.ArrayLike, numpy.int32], p: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute Bernoulli PDF on CPU """ -def bernoulli_pmf_cuda(k: typing.Annotated[numpy.typing.ArrayLike, numpy.int32], p: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def bernoulli_pmf_cuda(k: typing.Annotated[numpy.typing.ArrayLike, numpy.int32], p: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute bernoulli PDF using CUDA (GPU) """ -def bernoulli_pmf_scalar(k: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def bernoulli_pmf_scalar(k: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute PMF of Bernoulli distribution """ -def bernoulli_sample(p: typing.SupportsFloat) -> int: +def bernoulli_sample(p: typing.SupportsFloat | typing.SupportsIndex) -> int: """ Draw random sample from Bernoulli distribution """ -def bernoulli_stddev(p: typing.SupportsFloat) -> float: +def bernoulli_stddev(p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute standard deviation of Bernoulli distribution """ -def bernoulli_variance(p: typing.SupportsFloat) -> float: +def bernoulli_variance(p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute variance of Bernoulli distribution """ -def beta_cdf_scalar(x: typing.SupportsFloat, alpha: typing.SupportsFloat, beta: typing.SupportsFloat) -> float: +def beta_cdf_scalar(x: typing.SupportsFloat | typing.SupportsIndex, alpha: typing.SupportsFloat | typing.SupportsIndex, beta: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CDF of Beta distribution """ -def beta_mean(alpha: typing.SupportsFloat, beta: typing.SupportsFloat) -> float: +def beta_mean(alpha: typing.SupportsFloat | typing.SupportsIndex, beta: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute mean of Beta distribution """ -def beta_pdf_scalar(x: typing.SupportsFloat, alpha: typing.SupportsFloat, beta: typing.SupportsFloat) -> float: +def beta_pdf_scalar(x: typing.SupportsFloat | typing.SupportsIndex, alpha: typing.SupportsFloat | typing.SupportsIndex, beta: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute PDF of Beta distribution """ -def beta_sample(alpha: typing.SupportsFloat, beta: typing.SupportsFloat) -> float: +def beta_sample(alpha: typing.SupportsFloat | typing.SupportsIndex, beta: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Draw random sample from Beta distribution """ -def beta_stddev(alpha: typing.SupportsFloat, beta: typing.SupportsFloat) -> float: +def beta_stddev(alpha: typing.SupportsFloat | typing.SupportsIndex, beta: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute standard deviation of Beta distribution """ -def beta_variance(alpha: typing.SupportsFloat, beta: typing.SupportsFloat) -> float: +def beta_variance(alpha: typing.SupportsFloat | typing.SupportsIndex, beta: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute variance of Beta distribution """ -def binomial(n: typing.SupportsInt, a: typing.SupportsFloat, b: typing.SupportsFloat) -> float: +def binomial(n: typing.SupportsInt | typing.SupportsIndex, a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute binomial probability term """ -def binomial_cdf_scalar(x: typing.SupportsInt, n: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def binomial_cdf_scalar(x: typing.SupportsInt | typing.SupportsIndex, n: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CDF of Binomial distribution """ -def binomial_cgf_scalar(t: typing.SupportsFloat, n: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def binomial_cgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, n: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CGF of Binomial distribution """ -def binomial_logpmf_scalar(x: typing.SupportsInt, n: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def binomial_logpmf_scalar(x: typing.SupportsInt | typing.SupportsIndex, n: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute log PMF of Binomial distribution """ -def binomial_mean(n: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def binomial_mean(n: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute mean of Binomial distribution """ -def binomial_mgf_scalar(t: typing.SupportsFloat, n: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def binomial_mgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, n: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute MGF of Binomial distribution """ -def binomial_pmf_scalar(x: typing.SupportsInt, n: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def binomial_pmf_scalar(x: typing.SupportsInt | typing.SupportsIndex, n: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute PMF of Binomial distribution """ -def binomial_sample(n: typing.SupportsInt, p: typing.SupportsFloat) -> int: +def binomial_sample(n: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> int: """ Draw random sample from Binomial distribution """ -def binomial_stddev(n: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def binomial_stddev(n: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute standard deviation of Binomial distribution """ -def binomial_variance(n: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def binomial_variance(n: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute variance of Binomial distribution """ -def chebyshev_bound(variance: typing.SupportsFloat, k: typing.SupportsFloat) -> float: +def chebyshev_bound(variance: typing.SupportsFloat | typing.SupportsIndex, k: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute Chebyshev bound on tail probability """ -def chi_square_cdf_scalar(x: typing.SupportsFloat, k: typing.SupportsFloat) -> float: +def chi_square_cdf_scalar(x: typing.SupportsFloat | typing.SupportsIndex, k: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CDF of Chi-square distribution """ -def chi_square_cgf_scalar(t: typing.SupportsFloat, k: typing.SupportsFloat) -> float: +def chi_square_cgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, k: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CGF of Chi-square distribution """ -def chi_square_mean(k: typing.SupportsFloat) -> float: +def chi_square_mean(k: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute mean of Chi-square distribution """ -def chi_square_mgf_scalar(t: typing.SupportsFloat, k: typing.SupportsFloat) -> float: +def chi_square_mgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, k: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute MGF of Chi-square distribution """ -def chi_square_pdf_scalar(x: typing.SupportsFloat, k: typing.SupportsFloat) -> float: +def chi_square_pdf_scalar(x: typing.SupportsFloat | typing.SupportsIndex, k: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute PDF of Chi-square distribution """ -def chi_square_sample(k: typing.SupportsFloat) -> float: +def chi_square_sample(k: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Draw random sample from Chi-square distribution """ -def chi_square_stddev(k: typing.SupportsFloat) -> float: +def chi_square_stddev(k: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute standard deviation of Chi-square distribution """ -def chi_square_variance(k: typing.SupportsFloat) -> float: +def chi_square_variance(k: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute variance of Chi-square distribution """ -def choose(n: typing.SupportsInt, k: typing.SupportsInt) -> float: +def choose(n: typing.SupportsInt | typing.SupportsIndex, k: typing.SupportsInt | typing.SupportsIndex) -> float: """ Compute binomial coefficient n choose k """ -def coefficient_of_variation(mean: typing.SupportsFloat, stddev: typing.SupportsFloat) -> float: +def coefficient_of_variation(mean: typing.SupportsFloat | typing.SupportsIndex, stddev: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute coefficient of variation """ -def cosine_similarity(x: collections.abc.Sequence[typing.SupportsFloat], y: collections.abc.Sequence[typing.SupportsFloat]) -> float: +def cosine_similarity(x: collections.abc.Sequence[typing.SupportsFloat | typing.SupportsIndex], y: collections.abc.Sequence[typing.SupportsFloat | typing.SupportsIndex]) -> float: """ Compute cosine similarity between two vectors """ @@ -188,43 +188,43 @@ def cosine_similarity_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.flo """ Batch compute manhattan distance using CUDA (GPU) """ -def covariance(mean_x: typing.SupportsFloat, mean_y: typing.SupportsFloat, E_xy: typing.SupportsFloat) -> float: +def covariance(mean_x: typing.SupportsFloat | typing.SupportsIndex, mean_y: typing.SupportsFloat | typing.SupportsIndex, E_xy: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute covariance given means and expectation of product """ -def discrete_uniform_cdf_scalar(x: typing.SupportsInt, a: typing.SupportsInt, b: typing.SupportsInt) -> float: +def discrete_uniform_cdf_scalar(x: typing.SupportsInt | typing.SupportsIndex, a: typing.SupportsInt | typing.SupportsIndex, b: typing.SupportsInt | typing.SupportsIndex) -> float: """ Compute CDF of discrete uniform distribution """ -def discrete_uniform_cgf_scalar(t: typing.SupportsFloat, a: typing.SupportsInt, b: typing.SupportsInt) -> float: +def discrete_uniform_cgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, a: typing.SupportsInt | typing.SupportsIndex, b: typing.SupportsInt | typing.SupportsIndex) -> float: """ Compute CGF of discrete uniform distribution """ -def discrete_uniform_mean(a: typing.SupportsInt, b: typing.SupportsInt) -> float: +def discrete_uniform_mean(a: typing.SupportsInt | typing.SupportsIndex, b: typing.SupportsInt | typing.SupportsIndex) -> float: """ Compute mean of discrete uniform distribution """ -def discrete_uniform_mgf_scalar(t: typing.SupportsFloat, a: typing.SupportsInt, b: typing.SupportsInt) -> float: +def discrete_uniform_mgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, a: typing.SupportsInt | typing.SupportsIndex, b: typing.SupportsInt | typing.SupportsIndex) -> float: """ Compute MGF of discrete uniform distribution """ -def discrete_uniform_pmf_scalar(x: typing.SupportsInt, a: typing.SupportsInt, b: typing.SupportsInt) -> float: +def discrete_uniform_pmf_scalar(x: typing.SupportsInt | typing.SupportsIndex, a: typing.SupportsInt | typing.SupportsIndex, b: typing.SupportsInt | typing.SupportsIndex) -> float: """ Compute PMF of discrete uniform distribution """ -def discrete_uniform_sample(a: typing.SupportsInt, b: typing.SupportsInt) -> int: +def discrete_uniform_sample(a: typing.SupportsInt | typing.SupportsIndex, b: typing.SupportsInt | typing.SupportsIndex) -> int: """ Draw random sample from discrete uniform distribution """ -def discrete_uniform_stddev(a: typing.SupportsInt, b: typing.SupportsInt) -> float: +def discrete_uniform_stddev(a: typing.SupportsInt | typing.SupportsIndex, b: typing.SupportsInt | typing.SupportsIndex) -> float: """ Compute standard deviation of discrete uniform distribution """ -def discrete_uniform_variance(a: typing.SupportsInt, b: typing.SupportsInt) -> float: +def discrete_uniform_variance(a: typing.SupportsInt | typing.SupportsIndex, b: typing.SupportsInt | typing.SupportsIndex) -> float: """ Compute variance of discrete uniform distribution """ -def euclidean_distance(x: collections.abc.Sequence[typing.SupportsFloat], y: collections.abc.Sequence[typing.SupportsFloat]) -> float: +def euclidean_distance(x: collections.abc.Sequence[typing.SupportsFloat | typing.SupportsIndex], y: collections.abc.Sequence[typing.SupportsFloat | typing.SupportsIndex]) -> float: """ Compute Euclidean distance between two vectors """ @@ -232,151 +232,151 @@ def euclidean_distance_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.fl """ Batch compute euclidean distance using CUDA (GPU) """ -def exponential_cdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def exponential_cdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute exponential CDF on CPU """ -def exponential_cdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def exponential_cdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute exponential PDF using CUDA (GPU) """ -def exponential_cdf_scalar(x: typing.SupportsFloat, lambda_: typing.SupportsFloat) -> float: +def exponential_cdf_scalar(x: typing.SupportsFloat | typing.SupportsIndex, lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CDF of exponential distribution """ -def exponential_cgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def exponential_cgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute exponential CGF on CPU """ -def exponential_cgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def exponential_cgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute exponential PDF using CUDA (GPU) """ -def exponential_cgf_scalar(t: typing.SupportsFloat, lambda_: typing.SupportsFloat) -> float: +def exponential_cgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CGF of exponential distribution """ -def exponential_mean(lambda_: typing.SupportsFloat) -> float: +def exponential_mean(lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute mean of exponential distribution """ -def exponential_mgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def exponential_mgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute exponential MGF on CPU """ -def exponential_mgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def exponential_mgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute exponential PDF using CUDA (GPU) """ -def exponential_mgf_scalar(t: typing.SupportsFloat, lambda_: typing.SupportsFloat) -> float: +def exponential_mgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute MGF of exponential distribution """ -def exponential_pdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def exponential_pdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute exponential PDF on CPU """ -def exponential_pdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def exponential_pdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute exponential PDF using CUDA (GPU) """ -def exponential_pdf_scalar(x: typing.SupportsFloat, lambda_: typing.SupportsFloat) -> float: +def exponential_pdf_scalar(x: typing.SupportsFloat | typing.SupportsIndex, lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute PDF of exponential distribution """ -def exponential_sample(lambda_: typing.SupportsFloat) -> float: +def exponential_sample(lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Draw random sample from exponential distribution """ -def exponential_stddev(lambda_: typing.SupportsFloat) -> float: +def exponential_stddev(lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute standard deviation of exponential distribution """ -def exponential_variance(lambda_: typing.SupportsFloat) -> float: +def exponential_variance(lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute variance of exponential distribution """ -def factorial(n: typing.SupportsInt) -> float: +def factorial(n: typing.SupportsInt | typing.SupportsIndex) -> float: """ Compute factorial of integer n """ -def gamma(x: typing.SupportsFloat) -> float: +def gamma(x: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute Gamma function """ -def gamma_cdf_scalar(x: typing.SupportsFloat, alpha: typing.SupportsFloat, theta: typing.SupportsFloat) -> float: +def gamma_cdf_scalar(x: typing.SupportsFloat | typing.SupportsIndex, alpha: typing.SupportsFloat | typing.SupportsIndex, theta: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CDF of Gamma distribution """ -def gamma_cgf_scalar(t: typing.SupportsFloat, alpha: typing.SupportsFloat, theta: typing.SupportsFloat) -> float: +def gamma_cgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, alpha: typing.SupportsFloat | typing.SupportsIndex, theta: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CGF of Gamma distribution """ -def gamma_mean(alpha: typing.SupportsFloat, theta: typing.SupportsFloat) -> float: +def gamma_mean(alpha: typing.SupportsFloat | typing.SupportsIndex, theta: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute mean of Gamma distribution """ -def gamma_mgf_scalar(t: typing.SupportsFloat, alpha: typing.SupportsFloat, theta: typing.SupportsFloat) -> float: +def gamma_mgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, alpha: typing.SupportsFloat | typing.SupportsIndex, theta: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute MGF of Gamma distribution """ -def gamma_pdf_scalar(x: typing.SupportsFloat, alpha: typing.SupportsFloat, theta: typing.SupportsFloat) -> float: +def gamma_pdf_scalar(x: typing.SupportsFloat | typing.SupportsIndex, alpha: typing.SupportsFloat | typing.SupportsIndex, theta: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute PDF of Gamma distribution """ -def gamma_sample(alpha: typing.SupportsFloat, theta: typing.SupportsFloat) -> float: +def gamma_sample(alpha: typing.SupportsFloat | typing.SupportsIndex, theta: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Draw random sample from Gamma distribution """ -def gamma_stddev(alpha: typing.SupportsFloat, theta: typing.SupportsFloat) -> float: +def gamma_stddev(alpha: typing.SupportsFloat | typing.SupportsIndex, theta: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute standard deviation of Gamma distribution """ -def gamma_variance(alpha: typing.SupportsFloat, theta: typing.SupportsFloat) -> float: +def gamma_variance(alpha: typing.SupportsFloat | typing.SupportsIndex, theta: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute variance of Gamma distribution """ -def geometric_cdf_scalar(k: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def geometric_cdf_scalar(k: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CDF of geometric distribution """ -def geometric_cgf_scalar(t: typing.SupportsFloat, p: typing.SupportsFloat) -> float: +def geometric_cgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CGF of geometric distribution """ -def geometric_mean(p: typing.SupportsFloat) -> float: +def geometric_mean(p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute mean of geometric distribution """ -def geometric_mgf_scalar(t: typing.SupportsFloat, p: typing.SupportsFloat) -> float: +def geometric_mgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute MGF of geometric distribution """ -def geometric_pmf_scalar(k: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def geometric_pmf_scalar(k: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute PMF of geometric distribution """ -def geometric_sample(p: typing.SupportsFloat) -> int: +def geometric_sample(p: typing.SupportsFloat | typing.SupportsIndex) -> int: """ Draw random sample from geometric distribution """ -def geometric_stddev(p: typing.SupportsFloat) -> float: +def geometric_stddev(p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute standard deviation of geometric distribution """ -def geometric_variance(p: typing.SupportsFloat) -> float: +def geometric_variance(p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute variance of geometric distribution """ -def law_of_total_probability(probs_B_given_A: collections.abc.Sequence[typing.SupportsFloat], probs_A: collections.abc.Sequence[typing.SupportsFloat]) -> float: +def law_of_total_probability(probs_B_given_A: collections.abc.Sequence[typing.SupportsFloat | typing.SupportsIndex], probs_A: collections.abc.Sequence[typing.SupportsFloat | typing.SupportsIndex]) -> float: """ Compute probability using law of total probability """ -def log_gamma(x: typing.SupportsFloat) -> float: +def log_gamma(x: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute logarithm of Gamma function """ -def logit(p: typing.SupportsFloat) -> float: +def logit(p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute logit (inverse sigmoid) function """ @@ -389,7 +389,7 @@ def logit_cuda(p: typing.Annotated[numpy.typing.ArrayLike, numpy.float64]) -> nu """ Batch compute logit using CUDA (GPU) """ -def manhattan_distance(x: collections.abc.Sequence[typing.SupportsFloat], y: collections.abc.Sequence[typing.SupportsFloat]) -> float: +def manhattan_distance(x: collections.abc.Sequence[typing.SupportsFloat | typing.SupportsIndex], y: collections.abc.Sequence[typing.SupportsFloat | typing.SupportsIndex]) -> float: """ Compute Manhattan (L1) distance between two vectors """ @@ -397,193 +397,193 @@ def manhattan_distance_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.fl """ Batch compute manhattan distance using CUDA (GPU) """ -def negative_binomial_cdf_scalar(k: typing.SupportsInt, r: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def negative_binomial_cdf_scalar(k: typing.SupportsInt | typing.SupportsIndex, r: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CDF of negative binomial distribution """ -def negative_binomial_cgf_scalar(t: typing.SupportsFloat, r: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def negative_binomial_cgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, r: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CGF of negative binomial distribution """ -def negative_binomial_mean(r: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def negative_binomial_mean(r: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute mean of negative binomial distribution """ -def negative_binomial_mgf_scalar(t: typing.SupportsFloat, r: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def negative_binomial_mgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, r: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute MGF of negative binomial distribution """ -def negative_binomial_pmf_scalar(k: typing.SupportsInt, r: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def negative_binomial_pmf_scalar(k: typing.SupportsInt | typing.SupportsIndex, r: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute PMF of negative binomial distribution """ -def negative_binomial_sample(r: typing.SupportsInt, p: typing.SupportsFloat) -> int: +def negative_binomial_sample(r: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> int: """ Draw random sample from negative binomial distribution """ -def negative_binomial_stddev(r: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def negative_binomial_stddev(r: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute standard deviation of negative binomial distribution """ -def negative_binomial_variance(r: typing.SupportsInt, p: typing.SupportsFloat) -> float: +def negative_binomial_variance(r: typing.SupportsInt | typing.SupportsIndex, p: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute variance of negative binomial distribution """ -def normal_cdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat, sigma: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def normal_cdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute normal CDF on CPU """ -def normal_cdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat, sigma: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def normal_cdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute normal PDF using CUDA (GPU) """ -def normal_cdf_scalar(x: typing.SupportsFloat, mu: typing.SupportsFloat, sigma: typing.SupportsFloat) -> float: +def normal_cdf_scalar(x: typing.SupportsFloat | typing.SupportsIndex, mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CDF of normal distribution """ -def normal_cgf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat, sigma: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def normal_cgf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute normal CGF on CPU """ -def normal_cgf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat, sigma: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def normal_cgf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute normal PDF using CUDA (GPU) """ -def normal_cgf_scalar(t: typing.SupportsFloat, mu: typing.SupportsFloat, sigma: typing.SupportsFloat) -> float: +def normal_cgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CGF of normal distribution """ -def normal_log_sample(mu: typing.SupportsFloat, sigma: typing.SupportsFloat) -> float: +def normal_log_sample(mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Draw log-domain random sample from normal distribution """ -def normal_logpdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat, sigma: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def normal_logpdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute normal Log PDF on CPU """ -def normal_logpdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat, sigma: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def normal_logpdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute normal PDF using CUDA (GPU) """ -def normal_logpdf_scalar(x: typing.SupportsFloat, mu: typing.SupportsFloat, sigma: typing.SupportsFloat) -> float: +def normal_logpdf_scalar(x: typing.SupportsFloat | typing.SupportsIndex, mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute log-PDF of normal distribution """ -def normal_mean(mu: typing.SupportsFloat) -> float: +def normal_mean(mu: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute mean of normal distribution """ -def normal_mgf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat, sigma: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def normal_mgf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute normal MGF on CPU """ -def normal_mgf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat, sigma: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def normal_mgf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute normal PDF using CUDA (GPU) """ -def normal_mgf_scalar(t: typing.SupportsFloat, mu: typing.SupportsFloat, sigma: typing.SupportsFloat) -> float: +def normal_mgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute MGF of normal distribution """ -def normal_pdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat, sigma: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def normal_pdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute normal PDF on CPU """ -def normal_pdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat, sigma: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def normal_pdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute normal PDF using CUDA (GPU) """ -def normal_pdf_scalar(x: typing.SupportsFloat, mu: typing.SupportsFloat, sigma: typing.SupportsFloat) -> float: +def normal_pdf_scalar(x: typing.SupportsFloat | typing.SupportsIndex, mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute PDF of normal distribution """ -def normal_sample(mu: typing.SupportsFloat, sigma: typing.SupportsFloat) -> float: +def normal_sample(mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Draw random sample from normal distribution """ -def normal_stddev(sigma: typing.SupportsFloat) -> float: +def normal_stddev(sigma: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute standard deviation of normal distribution """ -def normal_variance(sigma: typing.SupportsFloat) -> float: +def normal_variance(sigma: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute variance of normal distribution """ -def permutation(n: typing.SupportsInt, k: typing.SupportsInt) -> float: +def permutation(n: typing.SupportsInt | typing.SupportsIndex, k: typing.SupportsInt | typing.SupportsIndex) -> float: """ Compute number of permutations of k items from n """ -def poisson_cdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def poisson_cdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute poisson CDF on CPU """ -def poisson_cdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def poisson_cdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute poisson CDF using CUDA (GPU) """ -def poisson_cdf_scalar(k: typing.SupportsFloat, lambda_: typing.SupportsFloat) -> float: +def poisson_cdf_scalar(k: typing.SupportsFloat | typing.SupportsIndex, lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CDF of Poisson distribution """ -def poisson_cgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def poisson_cgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute poisson CGF on CPU """ -def poisson_cgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def poisson_cgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute poisson CGF using CUDA (GPU) """ -def poisson_cgf_scalar(t: typing.SupportsFloat, lambda_: typing.SupportsFloat) -> float: +def poisson_cgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CGF of Poisson distribution """ -def poisson_mean(lambda_: typing.SupportsFloat) -> float: +def poisson_mean(lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute mean of Poisson distribution """ -def poisson_mgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def poisson_mgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute poisson MGF on CPU """ -def poisson_mgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def poisson_mgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute poisson MGF using CUDA (GPU) """ -def poisson_mgf_scalar(t: typing.SupportsFloat, lambda_: typing.SupportsFloat) -> float: +def poisson_mgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute MGF of Poisson distribution """ -def poisson_pmf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def poisson_pmf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute poisson PMF on CPU """ -def poisson_pmf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat, step_size: typing.SupportsInt) -> numpy.typing.NDArray[numpy.float64]: +def poisson_pmf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], lambda_: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsInt | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute poisson PMF using CUDA (GPU) """ -def poisson_pmf_scalar(k: typing.SupportsFloat, lambda_: typing.SupportsFloat) -> float: +def poisson_pmf_scalar(k: typing.SupportsFloat | typing.SupportsIndex, lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute PMF of Poisson distribution """ -def poisson_sample(lambda_: typing.SupportsFloat) -> int: +def poisson_sample(lambda_: typing.SupportsFloat | typing.SupportsIndex) -> int: """ Draw random sample from Poisson distribution """ -def poisson_stddev(lambda_: typing.SupportsFloat) -> float: +def poisson_stddev(lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute standard deviation of Poisson distribution """ -def poisson_variance(lambda_: typing.SupportsFloat) -> float: +def poisson_variance(lambda_: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute variance of Poisson distribution """ def seed(value: typing.SupportsInt | typing.SupportsIndex) -> None: """ Seed the sampling engine so draws are reproducible. - + Pins the calling thread's random stream to a fixed sequence: the same seed replays the same draws on every run. - + Two caveats. The seed applies to the calling thread only, so a worker thread that has not been seeded keeps its own entropy-initialised stream. And while the underlying Mersenne Twister engine is specified bit-for-bit by the C++ @@ -594,11 +594,11 @@ def seed(value: typing.SupportsInt | typing.SupportsIndex) -> None: def seed_from_entropy() -> None: """ Return the sampling engine to non-deterministic behaviour. - + Draws a fresh seed from the OS entropy source. This is the state every thread starts in, so it is only needed to undo a previous seed() call. """ -def sigmoid(x: typing.SupportsFloat) -> float: +def sigmoid(x: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute sigmoid function """ @@ -610,71 +610,71 @@ def sigmoid_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64]) -> """ Batch compute sigmoid using CUDA (GPU) """ -def uniform_cdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat, b: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def uniform_cdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute uniform CDF on CPU """ -def uniform_cdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat, b: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def uniform_cdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute uniform CDF using CUDA (GPU) """ -def uniform_cdf_scalar(x: typing.SupportsFloat, a: typing.SupportsFloat, b: typing.SupportsFloat) -> float: +def uniform_cdf_scalar(x: typing.SupportsFloat | typing.SupportsIndex, a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CDF of continuous uniform distribution """ -def uniform_cgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat, b: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def uniform_cgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute uniform CGF on CPU """ -def uniform_cgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat, b: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def uniform_cgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute uniform CGF using CUDA (GPU) """ -def uniform_cgf_scalar(t: typing.SupportsFloat, a: typing.SupportsFloat, b: typing.SupportsFloat) -> float: +def uniform_cgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute CGF of continuous uniform distribution """ -def uniform_mean(a: typing.SupportsFloat, b: typing.SupportsFloat) -> float: +def uniform_mean(a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute mean of continuous uniform distribution """ -def uniform_mgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat, b: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def uniform_mgf_cpu(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute uniform MGF on CPU """ -def uniform_mgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat, b: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def uniform_mgf_cuda(t: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute uniform MGF using CUDA (GPU) """ -def uniform_mgf_scalar(t: typing.SupportsFloat, a: typing.SupportsFloat, b: typing.SupportsFloat) -> float: +def uniform_mgf_scalar(t: typing.SupportsFloat | typing.SupportsIndex, a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute MGF of continuous uniform distribution """ -def uniform_pdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat, b: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def uniform_pdf_cpu(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute uniform PMF on CPU """ -def uniform_pdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat, b: typing.SupportsFloat, step_size: typing.SupportsFloat) -> numpy.typing.NDArray[numpy.float64]: +def uniform_pdf_cuda(x: typing.Annotated[numpy.typing.ArrayLike, numpy.float64], a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex, step_size: typing.SupportsFloat | typing.SupportsIndex) -> numpy.typing.NDArray[numpy.float64]: """ Batch compute uniform PMF using CUDA (GPU) """ -def uniform_pdf_scalar(x: typing.SupportsFloat, a: typing.SupportsFloat, b: typing.SupportsFloat) -> float: +def uniform_pdf_scalar(x: typing.SupportsFloat | typing.SupportsIndex, a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute PDF of continuous uniform distribution """ -def uniform_sample(a: typing.SupportsFloat, b: typing.SupportsFloat) -> float: +def uniform_sample(a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Draw random sample from continuous uniform distribution """ -def uniform_stddev(a: typing.SupportsFloat, b: typing.SupportsFloat) -> float: +def uniform_stddev(a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute standard deviation of continuous uniform distribution """ -def uniform_variance(a: typing.SupportsFloat, b: typing.SupportsFloat) -> float: +def uniform_variance(a: typing.SupportsFloat | typing.SupportsIndex, b: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute variance of continuous uniform distribution """ -def z_score(x: typing.SupportsFloat, mu: typing.SupportsFloat, sigma: typing.SupportsFloat) -> float: +def z_score(x: typing.SupportsFloat | typing.SupportsIndex, mu: typing.SupportsFloat | typing.SupportsIndex, sigma: typing.SupportsFloat | typing.SupportsIndex) -> float: """ Compute z-score for normal distribution """ From 18bdbec2f2391791e0ab026a0583a67929efb05d Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:33:24 -0400 Subject: [PATCH 08/20] Make the CUDA extension importable and stop it crashing on first use Correction to 1838dcc first: its message said the CUDA extension "links and exposes its *_cuda bindings" and that "the GPU and CPU paths agree numerically". Neither had been verified when it was committed, and neither was true. The build and ctest had passed, but the module could not be imported, and once it was, the first GPU call crashed. Both are fixed here, and this time the claims below were checked before committing. The module would not import --------------------------- The runtime was linked as CUDA::cudart, so _fastdist depended on cudart64_12.dll. Since Python 3.8, Windows does not search PATH for an extension's DLL dependencies, so the import failed with "DLL load failed" unless the caller added the toolkit directory with os.add_dll_directory first. It is now linked as CUDA::cudart_static, which makes the extension self-contained on every platform. The first GPU call crashed -------------------------- run_cuda_wrapper released the GIL as its first statement, then requested the input buffer and allocated the result array -- creating and touching Python objects without holding the GIL. On Python 3.14 that was an access violation on the very first call, at any size. The GIL is now held for the buffer and result handling and released only around the device work, in the same order the CPU wrapper already used. Verified on Windows, CUDA 12.4, MSVC 19.42, RTX 4070: - the extension imports on a plain interpreter with no DLL path changes, and no longer lists cudart64_12.dll among its dependencies - all 26 *_cuda bindings are exposed - every kernel matches its CPU counterpart to machine precision at n = 1, 1,000 and 1,000,000 (the last takes the four-stream path), including NaN placement for invalid parameters, and the three distance kernels agree with their scalar versions - ctest 14/14; pytest 1798 passed against the CUDA-enabled build One follow-up the checks turned up is not a GPU problem: normal_cdf loses relative precision in the lower tail on both the CPU and GPU paths. It is fixed separately. Co-Authored-By: Claude Opus 5 --- CMakeLists.txt | 10 +++++++--- src/wrappers/wrapper_utility.h | 11 +++++++---- 2 files changed, 14 insertions(+), 7 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 64707a9..e50460d 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -172,8 +172,12 @@ if (FASTDIST_ENABLE_CUDA) ) # CUDA Runtime Linking - find_package(CUDAToolkit REQUIRED) # Finds Toolkit - target_link_libraries(fastdist_core PUBLIC CUDA::cudart) # Links Runtime library + # The runtime is linked statically so the extension is self-contained. + # Linked dynamically, _fastdist depends on cudart64_*.dll, which Python 3.8+ + # on Windows will not find through PATH -- the module fails to import unless + # every caller adds the toolkit to the DLL search path first. + find_package(CUDAToolkit REQUIRED) + target_link_libraries(fastdist_core PUBLIC CUDA::cudart_static) # Links Runtime library target_include_directories(fastdist_core PRIVATE src) target_include_directories(_fastdist PRIVATE src) @@ -252,5 +256,5 @@ endforeach () # Link CUDA to Python module if enabled if (FASTDIST_ENABLE_CUDA) - target_link_libraries(_fastdist PRIVATE CUDA::cudart) + target_link_libraries(_fastdist PRIVATE CUDA::cudart_static) endif () \ No newline at end of file diff --git a/src/wrappers/wrapper_utility.h b/src/wrappers/wrapper_utility.h index 4006644..ca858b5 100644 --- a/src/wrappers/wrapper_utility.h +++ b/src/wrappers/wrapper_utility.h @@ -41,11 +41,11 @@ namespace fastdist::wrapper { } #ifdef FASTDIST_ENABLE_CUDA - // CUDA Implementation + // Runs a CUDA dispatcher over a 1-D contiguous numpy array. Requesting the + // buffer and allocating the result create and touch Python objects, so they + // happen with the GIL held; only the device work runs without it. template py::array_t run_cuda_wrapper(CudaFn fn, const pybind11::array_t& input, Args&&... args) { - py::gil_scoped_release release; - const auto buf = input.request(); if (buf.ndim != 1) { @@ -63,7 +63,10 @@ namespace fastdist::wrapper { const auto n = static_cast(buf.shape[0]); - fn(in_ptr, out_ptr, n, std::forward(args)...); + { + py::gil_scoped_release release; + fn(in_ptr, out_ptr, n, std::forward(args)...); + } return result; } From 6d814d8b17c6e47fbaa8a92f22a3d5514c1a137c Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:35:56 -0400 Subject: [PATCH 09/20] Keep the normal and exponential CDFs accurate in their tails Both CDFs subtracted two nearly equal numbers in one tail and lost all relative precision there. normal_cdf used 0.5 * (1 + erf(z)). Far below the mean erf(z) approaches -1, so the sum cancels: Phi(-8) was 1.8% off and Phi(-10) returned exactly 0 instead of 7.6e-24, and likewise at -20 and -37. That is the region p-values live in. It now uses 0.5 * erfc(-z), the same quantity without the cancellation. exponential_cdf used 1 - exp(-lambda x), which cancels for small lambda x: 2e-5 relative error at x = 1e-12, and exactly 0 at x = 1e-17. It now uses -expm1(-lambda x). Both the scalar and batch paths are changed. Tests compare both paths against math.erfc and math.expm1 directly, at 1e-12 and 1e-14 relative. The upper tails were already correct and are unchanged. Co-Authored-By: Claude Opus 5 --- src/math/exponential.cpp | 6 ++++-- src/math/normal.cpp | 4 +++- tests/python/test_exponential.py | 17 +++++++++++++++++ tests/python/test_normal.py | 22 ++++++++++++++++++++++ 4 files changed, 46 insertions(+), 3 deletions(-) diff --git a/src/math/exponential.cpp b/src/math/exponential.cpp index 790779b..1635816 100644 --- a/src/math/exponential.cpp +++ b/src/math/exponential.cpp @@ -25,7 +25,9 @@ namespace fastdist::math { if (x < 0.0) { return 0.0; } - return 1.0 - std::exp(-lambda * x); + // expm1 keeps full relative precision for small lambda * x, where + // 1 - exp(-lambda * x) cancels. + return -std::expm1(-lambda * x); } double exponential_mean(const double lambda) { @@ -109,7 +111,7 @@ namespace fastdist::math { output[i] = std::numeric_limits::quiet_NaN(); continue; } - output[i] = (x < 0.0) ? 0.0 : 1.0 - std::exp(-lambda * x); + output[i] = (x < 0.0) ? 0.0 : -std::expm1(-lambda * x); } } diff --git a/src/math/normal.cpp b/src/math/normal.cpp index 85f3429..e6854aa 100644 --- a/src/math/normal.cpp +++ b/src/math/normal.cpp @@ -26,7 +26,9 @@ namespace fastdist::math { } inline double normal_cdf_core(const double x, const double mu, const double scale) { - return 0.5 * (1.0 + std::erf((x - mu) / scale)); + // erfc rather than 1 + erf, which cancels catastrophically in the lower + // tail and loses all relative precision below about -8 sigma. + return 0.5 * std::erfc(-(x - mu) / scale); } } // namespace diff --git a/tests/python/test_exponential.py b/tests/python/test_exponential.py index cd0a883..c24e9dc 100644 --- a/tests/python/test_exponential.py +++ b/tests/python/test_exponential.py @@ -301,3 +301,20 @@ def test_sample_is_positive_and_finite(lam): def test_slots_prevent_dynamic_attributes(): with pytest.raises(AttributeError): Exponential(2.0).extra = 123 + + +# --------------------------------------------------------------------------- +# Small-argument precision +# +# 1 - exp(-lambda * x) cancels for small lambda * x: it was off by 2e-5 +# relative at x = 1e-12 and returned exactly 0 at x = 1e-17. expm1 keeps full +# precision there; checked on both the scalar and the batch path. +# --------------------------------------------------------------------------- + +@pytest.mark.parametrize("x", [1e-3, 1e-8, 1e-12, 1e-17]) +def test_cdf_keeps_relative_precision_for_small_x(x): + import fastdist._fastdist as core + expected = -math.expm1(-2.0 * x) + assert core.exponential_cdf_scalar(x, 2.0) == pytest.approx(expected, rel=1e-14) + batch = core.exponential_cdf_cpu(np.array([x]), 2.0, 0.0) + assert batch[0] == pytest.approx(expected, rel=1e-14) diff --git a/tests/python/test_normal.py b/tests/python/test_normal.py index d357fb5..91abf83 100644 --- a/tests/python/test_normal.py +++ b/tests/python/test_normal.py @@ -159,3 +159,25 @@ def test_logpdf_accepts_an_array(standard_normal): got = standard_normal.logpdf(x) expected = [math.log(1.0 / math.sqrt(2.0 * math.pi)) - v * v / 2.0 for v in x] assert np.allclose(got, expected) + + +# --------------------------------------------------------------------------- +# Lower-tail precision +# +# 0.5 * (1 + erf(z)) cancels to nothing far below the mean, so the CDF lost +# all relative precision below about -8 sigma and returned exactly 0 at -10. +# Tail probabilities are what p-values are made of, so they are checked here +# against erfc directly, on both the scalar and the batch path. +# --------------------------------------------------------------------------- + +def _phi(x): + return 0.5 * math.erfc(-x / math.sqrt(2.0)) + + +@pytest.mark.parametrize("x", [-5.0, -8.0, -10.0, -20.0, -37.0]) +def test_cdf_keeps_relative_precision_in_the_lower_tail(x): + import fastdist._fastdist as core + expected = _phi(x) + assert core.normal_cdf_scalar(x, 0.0, 1.0) == pytest.approx(expected, rel=1e-12) + batch = core.normal_cdf_cpu(np.array([x]), 0.0, 1.0, 0.0) + assert batch[0] == pytest.approx(expected, rel=1e-12) From e0b446eae0099bc491f6e54893b0ce9ed7b708b8 Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:35:57 -0400 Subject: [PATCH 10/20] Use erfc and expm1 in the CUDA normal and exponential CDF kernels The GPU kernels had the same tail cancellation as the CPU code fixed in the previous commit, and are changed the same way, kept separate so the CUDA change can be reverted on its own. Verified on an RTX 4070 against the CPU path: normal_cdf at n = 1,000,000 now agrees to 7.4e-16 relative, down from 2.3e-8 when both sides still used 1 + erf, and exponential_cdf to 4e-16. The rest of the CUDA verification passes unchanged. Co-Authored-By: Claude Opus 5 --- src/cuda/exponential/cdf.cu | 3 ++- src/cuda/normal/cdf.cu | 3 ++- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/src/cuda/exponential/cdf.cu b/src/cuda/exponential/cdf.cu index b453078..2d0d420 100644 --- a/src/cuda/exponential/cdf.cu +++ b/src/cuda/exponential/cdf.cu @@ -29,7 +29,8 @@ namespace fastdist::cuda::exponential { return; } - output[idx] = 1.0 - exp(-lambda * x_val); + // expm1 keeps relative precision for small lambda * x. + output[idx] = -expm1(-lambda * x_val); } } diff --git a/src/cuda/normal/cdf.cu b/src/cuda/normal/cdf.cu index d9e2791..bf85143 100644 --- a/src/cuda/normal/cdf.cu +++ b/src/cuda/normal/cdf.cu @@ -25,8 +25,9 @@ namespace fastdist::cuda::normal { return; } + // erfc rather than 1 + erf, which cancels in the lower tail. const double z = (x_val - mu) / (sigma * std::sqrt(2.0)); - output[idx] = 0.5 * (1.0 + std::erf(z)); + output[idx] = 0.5 * std::erfc(-z); } } From b79bc8f801fbe4e556f49d61543f1665b198aff1 Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:35:58 -0400 Subject: [PATCH 11/20] Pass step_size to normal_logpdf_cuda in the benchmark The CUDA benchmark cases were written without a CUDA build to run them on, and the commit that added them said so. Their first real run raised TypeError: normal_logpdf_cuda takes step_size like the others. The other CUDA calls were checked against the extension and are correct. Co-Authored-By: Claude Opus 5 --- benchmarks/run.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/benchmarks/run.py b/benchmarks/run.py index fb6d758..ca6b4e5 100644 --- a/benchmarks/run.py +++ b/benchmarks/run.py @@ -206,7 +206,7 @@ def cuda_cases(sizes): lambda x=x_real: core.normal_cdf_cuda(x, 0.0, 1.0, 0.0), lambda x=x_real: core.normal_cdf_cpu(x, 0.0, 1.0, 0.0)) yield ("normal_logpdf", n, - lambda x=x_real: core.normal_logpdf_cuda(x, 0.0, 1.0), + lambda x=x_real: core.normal_logpdf_cuda(x, 0.0, 1.0, 0.0), lambda x=x_real: core.normal_logpdf_cpu(x, 0.0, 1.0, 0.0)) yield ("exponential_pdf", n, lambda x=x_pos: core.exponential_pdf_cuda(x, 2.0, 0.0), From 8b241dc6c6113daaf6c98d2b2501c569cd8b44e1 Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:35:59 -0400 Subject: [PATCH 12/20] CHANGELOG: CUDA import and crash fixes, CDF tail precision Co-Authored-By: Claude Opus 5 --- CHANGELOG.md | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 58cf69e..f5c45a5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -43,6 +43,14 @@ call in `CMakeLists.txt`. - The CUDA backend did not compile on Windows. `nvcc` 12.x's front end crashes on MSVC's C++20 standard-library headers, and every `.cu` file was compiled a second time into the Python module target, which built as C++20. CUDA sources are now compiled once, as C++17. +- The CUDA extension could not be imported on Windows, because it depended on `cudart64_*.dll` and + Python does not search `PATH` for extension dependencies. Once imported, its first GPU call crashed + with an access violation, because the wrapper released the GIL before touching the input and output + arrays. The CUDA runtime is now linked statically, and the GIL is released only around the device + work. +- `normal_cdf` lost all relative precision in the lower tail: Phi(-8) was off by 1.8%, and Phi(-10) + returned exactly 0 instead of 7.6e-24. `exponential_cdf` did the same for small arguments, + returning 0 at x = 1e-17. Both now use `erfc` and `expm1` respectively, on the CPU and GPU paths. - `setup.py` no longer hardcodes the `Visual Studio 17 2022` CMake generator. CMake selects the newest Visual Studio present, so builds work on machines with a different version installed. Set `CMAKE_GENERATOR` to pin one. From f98eb9f1420e08c5fb44a98a91ea2aee106a2360 Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:41:12 -0400 Subject: [PATCH 13/20] Use the tail-safe CDF forms only where the tail needs them The previous two commits made normal_cdf and exponential_cdf accurate in their tails by switching every evaluation to erfc and expm1. That was correct but not free: the benchmark flagged batch normal_cdf 77% slower and exponential_cdf 33-36% slower. The cancellation only happens on one side, so the tail-safe form is now used only there: normal 0.5 * (1 + erf(u)) while u > -0.5, where the sum is at least 0.48 and loses nothing; 0.5 * erfc(-u) below that exponential 1 - exp(-u) from u = ln 2 up, where the result is at least 0.5; -expm1(-u) below that The accuracy guarantees are unchanged. New tests straddle both crossover points at 1e-14 relative, and the tail tests from the previous commit still pass. Measured, batch at n = 100,000 (the tail-fix commit, then this one, against the erf/exp-only code from before either): exponential_cdf 415 us -> 363 us (was 312 us) mostly recovered normal_cdf 944 us -> 892 us (was 529 us) barely recovered normal_cdf recovered much less than expected, given that ~76% of the benchmark's inputs now take the erf path. The cost of erfc itself therefore does not explain the slowdown. A likely cause is that the branch stops MSVC from auto-vectorising the loop, but that is not confirmed. normal_cdf remains 2.4x faster than SciPy, and now returns correct tail probabilities. Co-Authored-By: Claude Opus 5 --- src/math/exponential.cpp | 15 +++++++++++---- src/math/normal.cpp | 9 ++++++--- tests/python/test_exponential.py | 9 +++++++++ tests/python/test_normal.py | 9 +++++++++ 4 files changed, 35 insertions(+), 7 deletions(-) diff --git a/src/math/exponential.cpp b/src/math/exponential.cpp index 1635816..2479820 100644 --- a/src/math/exponential.cpp +++ b/src/math/exponential.cpp @@ -8,6 +8,15 @@ namespace fastdist::math { + namespace { + // P(X <= x) for x >= 0, given u = lambda * x. 1 - exp(-u) cancels for + // small u, where expm1 keeps full precision; from ln 2 up the result is at + // least 0.5, so the cheaper form loses nothing. + inline double exponential_cdf_core(const double u) { + return (u < 0.6931471805599453) ? -std::expm1(-u) : 1.0 - std::exp(-u); + } + } // namespace + double exponential_pdf_scalar(const double x, const double lambda) { if (!std::isfinite(x) || !std::isfinite(lambda) || lambda <= 0.0) { return std::numeric_limits::quiet_NaN(); @@ -25,9 +34,7 @@ namespace fastdist::math { if (x < 0.0) { return 0.0; } - // expm1 keeps full relative precision for small lambda * x, where - // 1 - exp(-lambda * x) cancels. - return -std::expm1(-lambda * x); + return exponential_cdf_core(lambda * x); } double exponential_mean(const double lambda) { @@ -111,7 +118,7 @@ namespace fastdist::math { output[i] = std::numeric_limits::quiet_NaN(); continue; } - output[i] = (x < 0.0) ? 0.0 : -std::expm1(-lambda * x); + output[i] = (x < 0.0) ? 0.0 : exponential_cdf_core(lambda * x); } } diff --git a/src/math/normal.cpp b/src/math/normal.cpp index e6854aa..186b75e 100644 --- a/src/math/normal.cpp +++ b/src/math/normal.cpp @@ -26,9 +26,12 @@ namespace fastdist::math { } inline double normal_cdf_core(const double x, const double mu, const double scale) { - // erfc rather than 1 + erf, which cancels catastrophically in the lower - // tail and loses all relative precision below about -8 sigma. - return 0.5 * std::erfc(-(x - mu) / scale); + const double u = (x - mu) / scale; + // 1 + erf(u) cancels catastrophically in the lower tail, so erfc is used + // below the crossover. Above it the sum is at least 0.48 and loses + // nothing, and erf is the cheaper of the two. + if (u > -0.5) return 0.5 * (1.0 + std::erf(u)); + return 0.5 * std::erfc(-u); } } // namespace diff --git a/tests/python/test_exponential.py b/tests/python/test_exponential.py index c24e9dc..0dd8d13 100644 --- a/tests/python/test_exponential.py +++ b/tests/python/test_exponential.py @@ -318,3 +318,12 @@ def test_cdf_keeps_relative_precision_for_small_x(x): assert core.exponential_cdf_scalar(x, 2.0) == pytest.approx(expected, rel=1e-14) batch = core.exponential_cdf_cpu(np.array([x]), 2.0, 0.0) assert batch[0] == pytest.approx(expected, rel=1e-14) + + +@pytest.mark.parametrize("x", [0.3, 0.3465, 0.3466, 0.35, 1.0, 5.0]) +def test_cdf_is_accurate_across_the_expm1_crossover(x): + """With lambda = 2 the implementation switches formulas at x = ln(2) / 2.""" + import fastdist._fastdist as core + expected = -math.expm1(-2.0 * x) + assert core.exponential_cdf_scalar(x, 2.0) == pytest.approx(expected, rel=1e-14) + assert core.exponential_cdf_cpu(np.array([x]), 2.0, 0.0)[0] == pytest.approx(expected, rel=1e-14) diff --git a/tests/python/test_normal.py b/tests/python/test_normal.py index 91abf83..34055f3 100644 --- a/tests/python/test_normal.py +++ b/tests/python/test_normal.py @@ -181,3 +181,12 @@ def test_cdf_keeps_relative_precision_in_the_lower_tail(x): assert core.normal_cdf_scalar(x, 0.0, 1.0) == pytest.approx(expected, rel=1e-12) batch = core.normal_cdf_cpu(np.array([x]), 0.0, 1.0, 0.0) assert batch[0] == pytest.approx(expected, rel=1e-12) + + +@pytest.mark.parametrize("x", [-0.8, -0.7072, -0.7071, -0.7, -0.3, 0.0, 2.0]) +def test_cdf_is_accurate_across_the_erf_erfc_crossover(x): + """The implementation switches formulas at (x - mu) / (sigma * sqrt 2) = -0.5.""" + import fastdist._fastdist as core + expected = 0.5 * math.erfc(-x / math.sqrt(2.0)) + assert core.normal_cdf_scalar(x, 0.0, 1.0) == pytest.approx(expected, rel=1e-14) + assert core.normal_cdf_cpu(np.array([x]), 0.0, 1.0, 0.0)[0] == pytest.approx(expected, rel=1e-14) From c7d7d4feb48788a2d2b7f6467f457f8222fcb2c0 Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:47:54 -0400 Subject: [PATCH 14/20] Take the minimum of 15 rounds for batch benchmark cases Two benchmark runs on this branch were disturbed by other load on the machine. The second showed only the cheapest 1M-element cases drifting, uniform_pdf and uniform_cdf by 15-25%. Each call there takes ~1.5 ms, so the default 7 rounds span ~10 ms, and one burst of background work can cover every round and leave no clean minimum to report. Case-by-case re-measurement confirmed it was noise; the code is untouched. Batch cases now take 15 rounds, as the scalar group already takes more than the default. The minimum can only move toward the true cost, so results stay comparable with earlier reports. The next run passed the drift check against the pre-branch report on every untouched case, the largest being +4.9%. Co-Authored-By: Claude Opus 5 --- benchmarks/run.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/benchmarks/run.py b/benchmarks/run.py index ca6b4e5..00c586a 100644 --- a/benchmarks/run.py +++ b/benchmarks/run.py @@ -223,7 +223,10 @@ def run(sizes, sample_sizes) -> list[Result]: # Big arrays are slow enough that one call per round is plenty; small # ones need repetition to rise above timer resolution. inner = 50 if n <= 1_000 else 1 - results.append(measure("batch", case, n, fd, sp, "scipy", inner=inner)) + # 15 rounds rather than the default 7: the cheapest cases take ~1.5 ms a + # call, so 7 rounds span ~10 ms, and one burst of background load can + # cover all of them and leave no clean minimum. + results.append(measure("batch", case, n, fd, sp, "scipy", inner=inner, repeat=15)) print(f" batch {case:<18} n={n:<9,} {_fmt(results[-1])}") for case, n, fd, sp in scalar_cases(): From b7752f22e79579ae63c86dcc8eec6c2839f2e082 Mon Sep 17 00:00:00 2001 From: ghosteau Date: Thu, 10 Sep 2026 22:47:55 -0400 Subject: [PATCH 15/20] Record the CUDA backend and CDF tail-accuracy costs in the benchmark log The entry is generated from report 024546 (commit f98eb9f), and it was only written after checking that no untouched batch case had drifted by 10% or more from the pre-branch report; the largest was +4.9%. What it records, including the unflattering parts: - the first measurements of the CUDA backend: it agrees with the CPU path to machine precision and runs at 0.06x-0.89x its speed at every size, transfer-bound at about 1.4 GB/s. Also that the Python classes auto- dispatch to it from n = 100,000, so a CUDA build currently picks the slower path. - the cost of the tail-accuracy fixes: normal_cdf +62-69%, exponential_cdf +15%, with the unconfirmed auto-vectorisation hypothesis labelled as such. - the two disturbed runs that were discarded, and why noise_pct did not catch the first. Report 023604 (commit 8b241dc) is committed as well, because the crossover commit's message cites figures from it. The two disturbed reports were deleted, not committed. Co-Authored-By: Claude Opus 5 --- BENCHMARKS.md | 138 ++++ ...0.1.0_20260911T023604+0000_8b241dc6c6.json | 666 ++++++++++++++++++ ...0.1.0_20260911T024546+0000_f98eb9f142.json | 666 ++++++++++++++++++ 3 files changed, 1470 insertions(+) create mode 100644 benchmarks/results/0.1.0_20260911T023604+0000_8b241dc6c6.json create mode 100644 benchmarks/results/0.1.0_20260911T024546+0000_f98eb9f142.json diff --git a/BENCHMARKS.md b/BENCHMARKS.md index 4eda950..1268893 100644 --- a/BENCHMARKS.md +++ b/BENCHMARKS.md @@ -369,6 +369,144 @@ at -8.7%, its original level. --- +## Unreleased — CUDA backend measured, CDF tail accuracy + +Commit `f98eb9f142`. The first run of the CUDA backend on real hardware, and the +speed cost of making two CDFs accurate in their tails. Compared against the +correctness-pass report above, taken on the same machine. + +### The GPU path is slower than the CPU path at every size + +Until this branch the CUDA backend did not build on Windows, and once built it +could not be imported, and once imported it crashed on its first call. With +those fixed it runs, agrees with the CPU path to machine precision, and loses +to it everywhere: 0.06x to 0.89x the CPU path's speed across +the cases and sizes below. + +The kernels are not the bottleneck. GPU time barely depends on which function +runs, and `normal_pdf` at n = 1,000,000 moves 16 MB through the device in +11.1 ms -- about 1.4 GB/s, a small fraction of what the bus sustains. The +executor copies from pageable host memory across four streams; pinned buffers +and fewer, larger transfers are the obvious next step, and should be measured +rather than assumed. + +This matters beyond the benchmark: the Python classes dispatch to CUDA +automatically from n = 100,000, so on a CUDA build they currently choose the +slower path. + +### Tail accuracy cost normal_cdf most of its lead + +`normal_cdf` returned exactly 0 below about -10 sigma and `exponential_cdf` +returned 0 for very small arguments. Both now use erfc / expm1 where the +naive form cancels, and the cheaper form elsewhere. Against the pre-branch +report: + +- `normal_cdf` n=100,000: 529 us -> 892 us (+69%) +- `normal_cdf` n=1,000,000: 5963 us -> 9663 us (+62%) +- `exponential_cdf` n=100,000: 312 us -> 360 us (+15%) + +`normal_cdf` is still faster than SciPy, and now correct where p-values live. +It recovered much less than expected when the tail-safe form was restricted to +the lower tail, which suggests the cost is not erfc itself; lost loop +auto-vectorisation is a plausible cause, not a confirmed one. + +### Discarded runs + +Two runs on this branch were disturbed by other load on the machine and are +not recorded. In the first, untouched functions came out 45-89% slower while +reporting within-run noise under 5% -- the blind spot `compare.py` documents, +since noise_pct cannot see a run that is uniformly slow. In the second, only +the cheapest 1M-element cases (uniform, ~1.5 ms a call) drifted, by 15-25%. +Both were re-measured case by case and confirmed as noise. Batch cases now +take the minimum of 15 rounds rather than 7, so a brief burst of background +load is less likely to cover every round of a short case. Before this entry +was written, every batch case this branch did not touch was checked against +the pre-branch report; the largest drift was +4.9% (uniform_pdf, n=1,000,000). + + +- **Version** 0.1.0 (`f98eb9f142` on `chore/release-prep`, working tree dirty) +- **Measured** 2026-09-11T02:45:46+00:00 +- **CPU** AMD Ryzen 7 7700 8-Core Processor +- **Platform** Windows-11-10.0.26200-SP0 +- **Toolchain** Python 3.14.2, numpy 2.5.2, scipy 1.18.1 +- **CUDA** available + +### batch (vs vectorised SciPy) + +| case | n | fastdist | baseline | speedup | max abs diff | +|---|---:|---:|---:|---:|---:| +| `normal_pdf` | 1,000 | 5.23 us | 31.01 us (scipy) | **5.93x** | 1.1e-16 | +| `normal_cdf` | 1,000 | 6.79 us | 29.65 us (scipy) | **4.37x** | 2.2e-16 | +| `normal_logpdf` | 1,000 | 1.89 us | 31.27 us (scipy) | **16.56x** | 8.9e-16 | +| `exponential_pdf` | 1,000 | 4.16 us | 28.87 us (scipy) | **6.94x** | 0.0e+00 | +| `exponential_cdf` | 1,000 | 4.45 us | 30.02 us (scipy) | **6.74x** | 1.1e-16 | +| `uniform_pdf` | 1,000 | 2.06 us | 32.63 us (scipy) | **15.86x** | 0.0e+00 | +| `uniform_cdf` | 1,000 | 2.10 us | 32.09 us (scipy) | **15.25x** | 0.0e+00 | +| `poisson_pmf` | 1,000 | 40.62 us | 37.69 us (scipy) | **0.93x** | 2.0e-19 | +| `poisson_cdf` | 1,000 | 9.95 us | 71.15 us (scipy) | **7.15x** | 2.2e-16 | +| `bernoulli_pmf` | 1,000 | 2.00 us | 48.84 us (scipy) | **24.42x** | 2.2e-16 | +| `normal_pdf` | 100,000 | 414.20 us | 1.44 ms (scipy) | **3.48x** | 1.1e-16 | +| `normal_cdf` | 100,000 | 892.20 us | 2.15 ms (scipy) | **2.41x** | 2.2e-16 | +| `normal_logpdf` | 100,000 | 77.90 us | 1.59 ms (scipy) | **20.38x** | 8.9e-16 | +| `exponential_pdf` | 100,000 | 312.40 us | 1.37 ms (scipy) | **4.37x** | 0.0e+00 | +| `exponential_cdf` | 100,000 | 360.10 us | 1.65 ms (scipy) | **4.58x** | 1.7e-16 | +| `uniform_pdf` | 100,000 | 93.60 us | 1.55 ms (scipy) | **16.57x** | 0.0e+00 | +| `uniform_cdf` | 100,000 | 99.30 us | 1.63 ms (scipy) | **16.43x** | 0.0e+00 | +| `poisson_pmf` | 100,000 | 4.02 ms | 3.23 ms (scipy) | **0.80x** | 2.0e-19 | +| `poisson_cdf` | 100,000 | 1.31 ms | 6.17 ms (scipy) | **4.70x** | 2.2e-16 | +| `bernoulli_pmf` | 100,000 | 292.70 us | 3.52 ms (scipy) | **12.03x** | 2.2e-16 | +| `normal_pdf` | 1,000,000 | 4.82 ms | 20.68 ms (scipy) | **4.29x** | 1.1e-16 | +| `normal_cdf` | 1,000,000 | 9.66 ms | 23.05 ms (scipy) | **2.39x** | 2.2e-16 | +| `normal_logpdf` | 1,000,000 | 1.34 ms | 22.24 ms (scipy) | **16.60x** | 8.9e-16 | +| `exponential_pdf` | 1,000,000 | 3.81 ms | 17.79 ms (scipy) | **4.67x** | 0.0e+00 | +| `exponential_cdf` | 1,000,000 | 4.31 ms | 20.35 ms (scipy) | **4.72x** | 1.7e-16 | +| `uniform_pdf` | 1,000,000 | 1.46 ms | 19.01 ms (scipy) | **13.05x** | 0.0e+00 | +| `uniform_cdf` | 1,000,000 | 1.48 ms | 20.21 ms (scipy) | **13.67x** | 0.0e+00 | +| `poisson_pmf` | 1,000,000 | 42.48 ms | 38.83 ms (scipy) | **0.91x** | 2.0e-19 | +| `poisson_cdf` | 1,000,000 | 14.32 ms | 64.10 ms (scipy) | **4.48x** | 2.2e-16 | +| `bernoulli_pmf` | 1,000,000 | 3.55 ms | 39.58 ms (scipy) | **11.16x** | 2.2e-16 | + +### scalar (per-call cost, not throughput) + +| case | n | fastdist | baseline | speedup | max abs diff | +|---|---:|---:|---:|---:|---:| +| `normal_pdf` | 20,000 | 7.40 ms | 466.93 ms (scipy) | **63.09x** | 1.1e-16 | +| `normal_cdf` | 20,000 | 7.67 ms | 443.36 ms (scipy) | **57.81x** | 2.2e-16 | +| `gamma_cdf` | 20,000 | 12.23 ms | 444.12 ms (scipy) | **36.33x** | 1.2e-13 | +| `chi_square_cdf` | 20,000 | 11.84 ms | 447.19 ms (scipy) | **37.76x** | 1.2e-13 | +| `beta_cdf` | 20,000 | 9.90 ms | 484.99 ms (scipy) | **48.98x** | 8.9e-16 | + +### cuda (GPU path vs the CPU path; below 1x means the GPU is slower) + +| case | n | fastdist | baseline | speedup | max abs diff | +|---|---:|---:|---:|---:|---:| +| `normal_pdf` | 1,000 | 42.20 us | 5.30 us (fastdist-cpu) | **0.13x** | 5.6e-17 | +| `normal_cdf` | 1,000 | 39.80 us | 7.20 us (fastdist-cpu) | **0.18x** | 2.2e-16 | +| `normal_logpdf` | 1,000 | 34.30 us | 2.00 us (fastdist-cpu) | **0.06x** | 0.0e+00 | +| `exponential_pdf` | 1,000 | 33.30 us | 4.30 us (fastdist-cpu) | **0.13x** | 1.1e-16 | +| `uniform_pdf` | 1,000 | 39.30 us | 3.60 us (fastdist-cpu) | **0.09x** | 0.0e+00 | +| `normal_pdf` | 100,000 | 1.11 ms | 413.90 us (fastdist-cpu) | **0.37x** | 5.6e-17 | +| `normal_cdf` | 100,000 | 1.13 ms | 888.00 us (fastdist-cpu) | **0.78x** | 2.2e-16 | +| `normal_logpdf` | 100,000 | 1.12 ms | 77.80 us (fastdist-cpu) | **0.07x** | 0.0e+00 | +| `exponential_pdf` | 100,000 | 1.11 ms | 311.80 us (fastdist-cpu) | **0.28x** | 2.2e-16 | +| `uniform_pdf` | 100,000 | 1.10 ms | 94.10 us (fastdist-cpu) | **0.09x** | 0.0e+00 | +| `normal_pdf` | 1,000,000 | 11.08 ms | 5.00 ms (fastdist-cpu) | **0.45x** | 5.6e-17 | +| `normal_cdf` | 1,000,000 | 11.23 ms | 9.98 ms (fastdist-cpu) | **0.89x** | 2.2e-16 | +| `normal_logpdf` | 1,000,000 | 11.06 ms | 1.39 ms (fastdist-cpu) | **0.13x** | 0.0e+00 | +| `exponential_pdf` | 1,000,000 | 10.85 ms | 3.74 ms (fastdist-cpu) | **0.35x** | 2.2e-16 | +| `uniform_pdf` | 1,000,000 | 10.84 ms | 1.55 ms (fastdist-cpu) | **0.14x** | 0.0e+00 | + +### sample (vs numpy) + +| case | n | fastdist | baseline | speedup | max abs diff | +|---|---:|---:|---:|---:|---:| +| `normal_sample` | 100,000 | 27.79 ms | 871.10 us (numpy) | **0.03x** | - | +| `uniform_sample` | 100,000 | 25.48 ms | 226.80 us (numpy) | **0.01x** | - | +| `normal_sample` | 1,000,000 | 294.54 ms | 10.22 ms (numpy) | **0.03x** | - | +| `uniform_sample` | 1,000,000 | 273.14 ms | 3.17 ms (numpy) | **0.01x** | - | + +--- + ## Changes to record here Add an entry when a release ships, or when a change is made specifically to diff --git a/benchmarks/results/0.1.0_20260911T023604+0000_8b241dc6c6.json b/benchmarks/results/0.1.0_20260911T023604+0000_8b241dc6c6.json new file mode 100644 index 0000000..5da3cc0 --- /dev/null +++ b/benchmarks/results/0.1.0_20260911T023604+0000_8b241dc6c6.json @@ -0,0 +1,666 @@ +{ + "environment": { + "timestamp_utc": "2026-09-11T02:36:04+00:00", + "fastdist_version": "0.1.0", + "git_commit": "8b241dc6c6113daaf6c98d2b2501c569cd8b44e1", + "git_branch": "chore/release-prep", + "git_dirty": true, + "cuda_available": true, + "python": "3.14.2", + "numpy": "2.5.2", + "scipy": "1.18.1", + "platform": "Windows-11-10.0.26200-SP0", + "processor": "AMD Ryzen 7 7700 8-Core Processor", + "machine": "AMD64" + }, + "results": [ + { + "group": "batch", + "case": "normal_pdf", + "n": 1000, + "fastdist_s": 5.2380000124685466e-06, + "baseline_s": 3.110799996647984e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.11454840917674304, + "baseline_noise_pct": 1.170117502645845, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 5.9389079596086845 + }, + { + "group": "batch", + "case": "normal_cdf", + "n": 1000, + "fastdist_s": 7.327999919652939e-06, + "baseline_s": 2.954600000521168e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.3275134144512533, + "baseline_noise_pct": 3.2288634268624676, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.0319323593293666 + }, + { + "group": "batch", + "case": "normal_logpdf", + "n": 1000, + "fastdist_s": 1.889999839477241e-06, + "baseline_s": 3.168399998685345e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.3174627617227563, + "baseline_noise_pct": 0.5870478860615128, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 16.764022580878628 + }, + { + "group": "batch", + "case": "exponential_pdf", + "n": 1000, + "fastdist_s": 4.148000152781606e-06, + "baseline_s": 2.9149999900255352e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.4107905885456584, + "baseline_noise_pct": 4.542024668326912, + "max_abs_diff": 0.0, + "speedup": 7.027482841510424 + }, + { + "group": "batch", + "case": "exponential_cdf", + "n": 1000, + "fastdist_s": 4.99800022225827e-06, + "baseline_s": 2.9898000066168605e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.2801083421776435, + "baseline_noise_pct": 0.5418424946022439, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 5.981992544341995 + }, + { + "group": "batch", + "case": "uniform_pdf", + "n": 1000, + "fastdist_s": 2.062000276055187e-06, + "baseline_s": 3.264600003603846e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.1939737577203935, + "baseline_noise_pct": 14.151810201018264, + "max_abs_diff": 0.0, + "speedup": 15.832199643781586 + }, + { + "group": "batch", + "case": "uniform_cdf", + "n": 1000, + "fastdist_s": 2.1219998598098755e-06, + "baseline_s": 3.2224000024143605e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.47125715389207634, + "baseline_noise_pct": 5.225918619227006, + "max_abs_diff": 0.0, + "speedup": 15.185674907174958 + }, + { + "group": "batch", + "case": "poisson_pmf", + "n": 1000, + "fastdist_s": 4.0616000187583265e-05, + "baseline_s": 3.878999996231869e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 3.0628320425948927, + "baseline_noise_pct": 16.24129919079122, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.9550423425046466 + }, + { + "group": "batch", + "case": "poisson_cdf", + "n": 1000, + "fastdist_s": 9.94000001810491e-06, + "baseline_s": 7.433599996147677e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.0060350920752759, + "baseline_noise_pct": 2.507533272579516, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 7.478470807452688 + }, + { + "group": "batch", + "case": "bernoulli_pmf", + "n": 1000, + "fastdist_s": 2.503999858163297e-06, + "baseline_s": 4.9005999753717335e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.4792369031012167, + "baseline_noise_pct": 11.451659529746918, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 19.571087272210793 + }, + { + "group": "batch", + "case": "normal_pdf", + "n": 100000, + "fastdist_s": 0.000414699999964796, + "baseline_s": 0.0015069000073708594, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.09645600333525546, + "baseline_noise_pct": 10.086932942217821, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 3.6337111345521595 + }, + { + "group": "batch", + "case": "normal_cdf", + "n": 100000, + "fastdist_s": 0.0009442000009585172, + "baseline_s": 0.002169400002458133, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.3071375741639854, + "baseline_noise_pct": 5.5084351906468925, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 2.2976064395846616 + }, + { + "group": "batch", + "case": "normal_logpdf", + "n": 100000, + "fastdist_s": 7.779999577905983e-05, + "baseline_s": 0.0017853000026661903, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.7712141404222488, + "baseline_noise_pct": 4.542653429784446, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 22.94730205045475 + }, + { + "group": "batch", + "case": "exponential_pdf", + "n": 100000, + "fastdist_s": 0.0003091000107815489, + "baseline_s": 0.0015696000045863912, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.09996186183722, + "baseline_noise_pct": 4.109326046720159, + "max_abs_diff": 0.0, + "speedup": 5.07796813276619 + }, + { + "group": "batch", + "case": "exponential_cdf", + "n": 100000, + "fastdist_s": 0.00041459999920334667, + "baseline_s": 0.0016423999913968146, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.7477073099507057, + "baseline_noise_pct": 7.12981031440182, + "max_abs_diff": 1.6653345369377348e-16, + "speedup": 3.961408573450757 + }, + { + "group": "batch", + "case": "uniform_pdf", + "n": 100000, + "fastdist_s": 9.379998664371669e-05, + "baseline_s": 0.0015208000113489106, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.4264580540208127, + "baseline_noise_pct": 9.514727790629939, + "max_abs_diff": 0.0, + "speedup": 16.213222045813406 + }, + { + "group": "batch", + "case": "uniform_cdf", + "n": 100000, + "fastdist_s": 9.940000018104911e-05, + "baseline_s": 0.0016667000018060207, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.3017985230245263, + "baseline_noise_pct": 10.013800003687964, + "max_abs_diff": 0.0, + "speedup": 16.767605621431194 + }, + { + "group": "batch", + "case": "poisson_pmf", + "n": 100000, + "fastdist_s": 0.004074299999047071, + "baseline_s": 0.003268100001150742, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.5683677286372937, + "baseline_noise_pct": 6.12588343073898, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.8021255189640211 + }, + { + "group": "batch", + "case": "poisson_cdf", + "n": 100000, + "fastdist_s": 0.0013078999909339473, + "baseline_s": 0.006226600002264604, + "baseline_name": "scipy", + "fastdist_noise_pct": 10.474807779590343, + "baseline_noise_pct": 1.0439083954856727, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.760761560842511 + }, + { + "group": "batch", + "case": "bernoulli_pmf", + "n": 100000, + "fastdist_s": 0.000298400002066046, + "baseline_s": 0.003552900001523085, + "baseline_name": "scipy", + "fastdist_noise_pct": 7.506701000863557, + "baseline_noise_pct": 4.387964585989902, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 11.906501263149147 + }, + { + "group": "batch", + "case": "normal_pdf", + "n": 1000000, + "fastdist_s": 0.004858900007093325, + "baseline_s": 0.02154460000747349, + "baseline_name": "scipy", + "fastdist_noise_pct": 3.6119282709927014, + "baseline_noise_pct": 3.3641840136534715, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 4.434048853860203 + }, + { + "group": "batch", + "case": "normal_cdf", + "n": 1000000, + "fastdist_s": 0.01054679999651853, + "baseline_s": 0.023725500010186806, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.5404482638096939, + "baseline_noise_pct": 4.799055791784744, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 2.249544887360956 + }, + { + "group": "batch", + "case": "normal_logpdf", + "n": 1000000, + "fastdist_s": 0.0013540000072680414, + "baseline_s": 0.02257359999930486, + "baseline_name": "scipy", + "fastdist_noise_pct": 8.020679238690228, + "baseline_noise_pct": 2.309334848364925, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 16.67178720689337 + }, + { + "group": "batch", + "case": "exponential_pdf", + "n": 1000000, + "fastdist_s": 0.003838100004941225, + "baseline_s": 0.017944499995792285, + "baseline_name": "scipy", + "fastdist_noise_pct": 4.444907732796975, + "baseline_noise_pct": 2.9435202497630737, + "max_abs_diff": 0.0, + "speedup": 4.675360197152309 + }, + { + "group": "batch", + "case": "exponential_cdf", + "n": 1000000, + "fastdist_s": 0.005187000002479181, + "baseline_s": 0.02060689999780152, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.369384806015115, + "baseline_noise_pct": 2.0221382191708788, + "max_abs_diff": 1.6653345369377348e-16, + "speedup": 3.972797375737852 + }, + { + "group": "batch", + "case": "uniform_pdf", + "n": 1000000, + "fastdist_s": 0.0014645999908680096, + "baseline_s": 0.019122099998639897, + "baseline_name": "scipy", + "fastdist_noise_pct": 6.493241273210123, + "baseline_noise_pct": 4.461330114531253, + "max_abs_diff": 0.0, + "speedup": 13.056192897630018 + }, + { + "group": "batch", + "case": "uniform_cdf", + "n": 1000000, + "fastdist_s": 0.0015101999888429418, + "baseline_s": 0.02031880000140518, + "baseline_name": "scipy", + "fastdist_noise_pct": 4.012714261463846, + "baseline_noise_pct": 1.7820934645305309, + "max_abs_diff": 0.0, + "speedup": 13.454377004050091 + }, + { + "group": "batch", + "case": "poisson_pmf", + "n": 1000000, + "fastdist_s": 0.04319499999110121, + "baseline_s": 0.03937739999673795, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.266118769837569, + "baseline_noise_pct": 1.123995004035343, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.9116194005058516 + }, + { + "group": "batch", + "case": "poisson_cdf", + "n": 1000000, + "fastdist_s": 0.01443969999672845, + "baseline_s": 0.06604430000879802, + "baseline_name": "scipy", + "fastdist_noise_pct": 3.3179359382923415, + "baseline_noise_pct": 1.5295793573579533, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.5738000113410555 + }, + { + "group": "batch", + "case": "bernoulli_pmf", + "n": 1000000, + "fastdist_s": 0.003531700000166893, + "baseline_s": 0.042256800006725825, + "baseline_name": "scipy", + "fastdist_noise_pct": 4.411473032071124, + "baseline_noise_pct": 5.978919369151085, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 11.9650026912617 + }, + { + "group": "scalar", + "case": "normal_pdf", + "n": 20000, + "fastdist_s": 0.007597999996505678, + "baseline_s": 0.456388100006734, + "baseline_name": "scipy", + "fastdist_noise_pct": 3.046854462752559, + "baseline_noise_pct": 1.84597275675356, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 60.066872889790346 + }, + { + "group": "scalar", + "case": "normal_cdf", + "n": 20000, + "fastdist_s": 0.0077825000043958426, + "baseline_s": 0.44200750000891276, + "baseline_name": "scipy", + "fastdist_noise_pct": 4.528108006373442, + "baseline_noise_pct": 1.7704676918777948, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 56.79505297259886 + }, + { + "group": "scalar", + "case": "gamma_cdf", + "n": 20000, + "fastdist_s": 0.00900299999921117, + "baseline_s": 0.45333340000070166, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.8479396025982995, + "baseline_noise_pct": 7.633212112671277, + "max_abs_diff": 1.177946629127291e-13, + "speedup": 50.353593251185394 + }, + { + "group": "scalar", + "case": "chi_square_cdf", + "n": 20000, + "fastdist_s": 0.008751700006541796, + "baseline_s": 0.45349369999894407, + "baseline_name": "scipy", + "fastdist_noise_pct": 13.720762702462103, + "baseline_noise_pct": 4.373974766247948, + "max_abs_diff": 1.177946629127291e-13, + "speedup": 51.817783934545595 + }, + { + "group": "scalar", + "case": "beta_cdf", + "n": 20000, + "fastdist_s": 0.01005129999248311, + "baseline_s": 0.4951255000050878, + "baseline_name": "scipy", + "fastdist_noise_pct": 6.564325053728422, + "baseline_noise_pct": 8.991174965983225, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 49.25984702231241 + }, + { + "group": "cuda", + "case": "normal_pdf", + "n": 1000, + "fastdist_s": 3.819999983534217e-05, + "baseline_s": 5.599998985417187e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 4.712039690920367, + "baseline_noise_pct": 1.7857282065540798, + "max_abs_diff": 5.551115123125783e-17, + "speedup": 0.14659683271087706 + }, + { + "group": "cuda", + "case": "normal_cdf", + "n": 1000, + "fastdist_s": 4.369999805931002e-05, + "baseline_s": 1.0000003385357559e-05, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 48.05492166107622, + "baseline_noise_pct": 26.999905412583036, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.22883303957555026 + }, + { + "group": "cuda", + "case": "normal_logpdf", + "n": 1000, + "fastdist_s": 3.8700003642588854e-05, + "baseline_s": 2.5999906938523054e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 6.20155942369909, + "baseline_noise_pct": 3.84619689931158, + "max_abs_diff": 0.0, + "speedup": 0.06718321573983134 + }, + { + "group": "cuda", + "case": "exponential_pdf", + "n": 1000, + "fastdist_s": 3.5600009141489863e-05, + "baseline_s": 4.3000036384910345e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 26.404443401648294, + "baseline_noise_pct": 4.65119427128808, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 0.12078658804274338 + }, + { + "group": "cuda", + "case": "uniform_pdf", + "n": 1000, + "fastdist_s": 3.239999932702631e-05, + "baseline_s": 2.2999884095042944e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 28.703698297470797, + "baseline_noise_pct": 4.348513799081326, + "max_abs_diff": 0.0, + "speedup": 0.07098729806410119 + }, + { + "group": "cuda", + "case": "normal_pdf", + "n": 100000, + "fastdist_s": 0.001108700002077967, + "baseline_s": 0.0004167000006418675, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 2.6517534933127864, + "baseline_noise_pct": 8.711303063068147, + "max_abs_diff": 5.551115123125783e-17, + "speedup": 0.37584558479378805 + }, + { + "group": "cuda", + "case": "normal_cdf", + "n": 100000, + "fastdist_s": 0.001159499995992519, + "baseline_s": 0.0009452000085730106, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 2.8891770434456436, + "baseline_noise_pct": 0.3173909395391945, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.8151789666578911 + }, + { + "group": "cuda", + "case": "normal_logpdf", + "n": 100000, + "fastdist_s": 0.0011075999937020242, + "baseline_s": 7.850000110920519e-05, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 3.178043183084589, + "baseline_noise_pct": 1.910824010993462, + "max_abs_diff": 0.0, + "speedup": 0.07087396312348113 + }, + { + "group": "cuda", + "case": "exponential_pdf", + "n": 100000, + "fastdist_s": 0.0011264000058872625, + "baseline_s": 0.00030929999775253236, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 0.9055402037415398, + "baseline_noise_pct": 0.22631921604649535, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.27459161588773034 + }, + { + "group": "cuda", + "case": "uniform_pdf", + "n": 100000, + "fastdist_s": 0.0011177999986102805, + "baseline_s": 9.410000347997993e-05, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 1.63714384047487, + "baseline_noise_pct": 0.31879672830894845, + "max_abs_diff": 0.0, + "speedup": 0.08418322025136071 + }, + { + "group": "cuda", + "case": "normal_pdf", + "n": 1000000, + "fastdist_s": 0.015967400002409704, + "baseline_s": 0.005260800011456013, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 1.027092684391029, + "baseline_noise_pct": 7.382907315099063, + "max_abs_diff": 5.551115123125783e-17, + "speedup": 0.32947129843694556 + }, + { + "group": "cuda", + "case": "normal_cdf", + "n": 1000000, + "fastdist_s": 0.015940700002829544, + "baseline_s": 0.010924700007308275, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 2.208184099109717, + "baseline_noise_pct": 4.7781631275143965, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.6853337686154994 + }, + { + "group": "cuda", + "case": "normal_logpdf", + "n": 1000000, + "fastdist_s": 0.015992900007404387, + "baseline_s": 0.0013840000028721988, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 1.0129494748302612, + "baseline_noise_pct": 15.130058372946998, + "max_abs_diff": 0.0, + "speedup": 0.08653840155515478 + }, + { + "group": "cuda", + "case": "exponential_pdf", + "n": 1000000, + "fastdist_s": 0.010882299990043975, + "baseline_s": 0.0038100000092526898, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 1.2019518536078209, + "baseline_noise_pct": 10.086613801302963, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.35010981251558876 + }, + { + "group": "cuda", + "case": "uniform_pdf", + "n": 1000000, + "fastdist_s": 0.010918500003754161, + "baseline_s": 0.0015541000029770657, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 45.49526035718022, + "baseline_noise_pct": 12.290071347456879, + "max_abs_diff": 0.0, + "speedup": 0.14233640174407766 + }, + { + "group": "sample", + "case": "normal_sample", + "n": 100000, + "fastdist_s": 0.028246899993973784, + "baseline_s": 0.0008859999943524599, + "baseline_name": "numpy", + "fastdist_noise_pct": 4.2928604826692585, + "baseline_noise_pct": 1.7607226264347926, + "max_abs_diff": null, + "speedup": 0.03136627362795492 + }, + { + "group": "sample", + "case": "uniform_sample", + "n": 100000, + "fastdist_s": 0.025612199999159202, + "baseline_s": 0.00023960000544320792, + "baseline_name": "numpy", + "fastdist_noise_pct": 2.5862675166502083, + "baseline_noise_pct": 2.3372282379787124, + "max_abs_diff": null, + "speedup": 0.00935491701029484 + }, + { + "group": "sample", + "case": "normal_sample", + "n": 1000000, + "fastdist_s": 0.29395609999482986, + "baseline_s": 0.010242099990136921, + "baseline_name": "numpy", + "fastdist_noise_pct": 2.708601728019544, + "baseline_noise_pct": 4.1934759146813825, + "max_abs_diff": null, + "speedup": 0.03484227743638271 + }, + { + "group": "sample", + "case": "uniform_sample", + "n": 1000000, + "fastdist_s": 0.26212269999086857, + "baseline_s": 0.0032296999997925013, + "baseline_name": "numpy", + "fastdist_noise_pct": 1.8441745111688737, + "baseline_noise_pct": 9.84611546767437, + "max_abs_diff": null, + "speedup": 0.012321328904001876 + } + ] +} diff --git a/benchmarks/results/0.1.0_20260911T024546+0000_f98eb9f142.json b/benchmarks/results/0.1.0_20260911T024546+0000_f98eb9f142.json new file mode 100644 index 0000000..659ae2d --- /dev/null +++ b/benchmarks/results/0.1.0_20260911T024546+0000_f98eb9f142.json @@ -0,0 +1,666 @@ +{ + "environment": { + "timestamp_utc": "2026-09-11T02:45:46+00:00", + "fastdist_version": "0.1.0", + "git_commit": "f98eb9f1420e08c5fb44a98a91ea2aee106a2360", + "git_branch": "chore/release-prep", + "git_dirty": true, + "cuda_available": true, + "python": "3.14.2", + "numpy": "2.5.2", + "scipy": "1.18.1", + "platform": "Windows-11-10.0.26200-SP0", + "processor": "AMD Ryzen 7 7700 8-Core Processor", + "machine": "AMD64" + }, + "results": [ + { + "group": "batch", + "case": "normal_pdf", + "n": 1000, + "fastdist_s": 5.227999936323613e-06, + "baseline_s": 3.1005999771878124e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 3.481259064677035, + "baseline_noise_pct": 2.7285043064019914, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 5.930757488432926 + }, + { + "group": "batch", + "case": "normal_cdf", + "n": 1000, + "fastdist_s": 6.791999912820756e-06, + "baseline_s": 2.9647999908775092e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.7361615183766742, + "baseline_noise_pct": 1.0321110823947002, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.365135496072483 + }, + { + "group": "batch", + "case": "normal_logpdf", + "n": 1000, + "fastdist_s": 1.888000115286559e-06, + "baseline_s": 3.127200005110353e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.6355980212479526, + "baseline_noise_pct": 4.4896390382356755, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 16.563558337684263 + }, + { + "group": "batch", + "case": "exponential_pdf", + "n": 1000, + "fastdist_s": 4.157999937888235e-06, + "baseline_s": 2.886800008127466e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.28860249045637076, + "baseline_noise_pct": 4.406263259631216, + "max_abs_diff": 0.0, + "speedup": 6.9427610660177494 + }, + { + "group": "batch", + "case": "exponential_cdf", + "n": 1000, + "fastdist_s": 4.451999848242849e-06, + "baseline_s": 3.0018000106792897e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.7637010110835853, + "baseline_noise_pct": 2.9782124700421186, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 6.7425878549031495 + }, + { + "group": "batch", + "case": "uniform_pdf", + "n": 1000, + "fastdist_s": 2.057999954558909e-06, + "baseline_s": 3.2633999944664537e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.3887298878809621, + "baseline_noise_pct": 7.194950448842367, + "max_abs_diff": 0.0, + "speedup": 15.857143180384075 + }, + { + "group": "batch", + "case": "uniform_cdf", + "n": 1000, + "fastdist_s": 2.1040000137872994e-06, + "baseline_s": 3.208599984645844e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.7604482266513346, + "baseline_noise_pct": 1.9198414507919286, + "max_abs_diff": 0.0, + "speedup": 15.249999827092264 + }, + { + "group": "batch", + "case": "poisson_pmf", + "n": 1000, + "fastdist_s": 4.062399995746091e-05, + "baseline_s": 3.76920000417158e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 4.480110009957443, + "baseline_noise_pct": 3.8416639200048888, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.9278259177132895 + }, + { + "group": "batch", + "case": "poisson_cdf", + "n": 1000, + "fastdist_s": 9.954000124707819e-06, + "baseline_s": 7.115000014891848e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.381753132389614, + "baseline_noise_pct": 5.591004530016305, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 7.147880174555148 + }, + { + "group": "batch", + "case": "bernoulli_pmf", + "n": 1000, + "fastdist_s": 2.0000000949949025e-06, + "baseline_s": 4.8836000205483286e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.8999922583814679, + "baseline_noise_pct": 0.9378321361555324, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 24.41799894294893 + }, + { + "group": "batch", + "case": "normal_pdf", + "n": 100000, + "fastdist_s": 0.0004141999961575493, + "baseline_s": 0.00144160000490956, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.43457247003164423, + "baseline_noise_pct": 10.391232317671701, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 3.4804442739811576 + }, + { + "group": "batch", + "case": "normal_cdf", + "n": 100000, + "fastdist_s": 0.0008922000124584883, + "baseline_s": 0.0021500000002561137, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.8518252638925038, + "baseline_noise_pct": 2.102325345290721, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 2.409773560002217 + }, + { + "group": "batch", + "case": "normal_logpdf", + "n": 100000, + "fastdist_s": 7.789999654050916e-05, + "baseline_s": 0.0015875999961281195, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.4390243902439024, + "baseline_noise_pct": 15.192743994880761, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 20.379975181417933 + }, + { + "group": "batch", + "case": "exponential_pdf", + "n": 100000, + "fastdist_s": 0.0003124000068055466, + "baseline_s": 0.001365300006000325, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.976951937456578, + "baseline_noise_pct": 8.877169195148527, + "max_abs_diff": 0.0, + "speedup": 4.370358438724863 + }, + { + "group": "batch", + "case": "exponential_cdf", + "n": 100000, + "fastdist_s": 0.0003600999916670844, + "baseline_s": 0.0016500999918207526, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.638712306204632, + "baseline_noise_pct": 6.369311859935316, + "max_abs_diff": 1.6653345369377348e-16, + "speedup": 4.582338322702003 + }, + { + "group": "batch", + "case": "uniform_pdf", + "n": 100000, + "fastdist_s": 9.359999967273325e-05, + "baseline_s": 0.001550800006953068, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.21367684145081975, + "baseline_noise_pct": 12.657981715614309, + "max_abs_diff": 0.0, + "speedup": 16.568376200591313 + }, + { + "group": "batch", + "case": "uniform_cdf", + "n": 100000, + "fastdist_s": 9.929999941959977e-05, + "baseline_s": 0.0016317000117851421, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.6042342116847923, + "baseline_noise_pct": 7.7832929548976875, + "max_abs_diff": 0.0, + "speedup": 16.4320243839103 + }, + { + "group": "batch", + "case": "poisson_pmf", + "n": 100000, + "fastdist_s": 0.004024299996672198, + "baseline_s": 0.00322839998989366, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.5977936399347108, + "baseline_noise_pct": 1.4000746382897526, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.8022264723214747 + }, + { + "group": "batch", + "case": "poisson_cdf", + "n": 100000, + "fastdist_s": 0.0013113000022713095, + "baseline_s": 0.0061671000003116205, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.0752692348453616, + "baseline_noise_pct": 3.393815456084893, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.703042774063567 + }, + { + "group": "batch", + "case": "bernoulli_pmf", + "n": 100000, + "fastdist_s": 0.0002927000023191795, + "baseline_s": 0.00352030000067316, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.0157111961400784, + "baseline_noise_pct": 1.8492742358876249, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 12.026989999249784 + }, + { + "group": "batch", + "case": "normal_pdf", + "n": 1000000, + "fastdist_s": 0.004818599991267547, + "baseline_s": 0.02067799998621922, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.8615365301663838, + "baseline_noise_pct": 2.65257769159205, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 4.291287930870521 + }, + { + "group": "batch", + "case": "normal_cdf", + "n": 1000000, + "fastdist_s": 0.009662900003604591, + "baseline_s": 0.023054500008584, + "baseline_name": "scipy", + "fastdist_noise_pct": 7.382876818026343, + "baseline_noise_pct": 6.264286777269887, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 2.3858779455426307 + }, + { + "group": "batch", + "case": "normal_logpdf", + "n": 1000000, + "fastdist_s": 0.0013399000017670915, + "baseline_s": 0.022242899998673238, + "baseline_name": "scipy", + "fastdist_noise_pct": 7.39607416717708, + "baseline_noise_pct": 9.033444406343039, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 16.60041791875422 + }, + { + "group": "batch", + "case": "exponential_pdf", + "n": 1000000, + "fastdist_s": 0.003807000000961125, + "baseline_s": 0.017786800002795644, + "baseline_name": "scipy", + "fastdist_noise_pct": 8.799579831508774, + "baseline_noise_pct": 5.508579370061459, + "max_abs_diff": 0.0, + "speedup": 4.672130285869489 + }, + { + "group": "batch", + "case": "exponential_cdf", + "n": 1000000, + "fastdist_s": 0.004310299991630018, + "baseline_s": 0.020354399995994754, + "baseline_name": "scipy", + "fastdist_noise_pct": 8.173445474088657, + "baseline_noise_pct": 4.574440923118727, + "max_abs_diff": 1.6653345369377348e-16, + "speedup": 4.722269919847823 + }, + { + "group": "batch", + "case": "uniform_pdf", + "n": 1000000, + "fastdist_s": 0.0014565000019501895, + "baseline_s": 0.019010800009709783, + "baseline_name": "scipy", + "fastdist_noise_pct": 13.353930412851373, + "baseline_noise_pct": 3.8425525944766705, + "max_abs_diff": 0.0, + "speedup": 13.052385845695268 + }, + { + "group": "batch", + "case": "uniform_cdf", + "n": 1000000, + "fastdist_s": 0.0014785999956075102, + "baseline_s": 0.020212700008414686, + "baseline_name": "scipy", + "fastdist_noise_pct": 10.455837515672146, + "baseline_noise_pct": 3.348389810749926, + "max_abs_diff": 0.0, + "speedup": 13.670161009374224 + }, + { + "group": "batch", + "case": "poisson_pmf", + "n": 1000000, + "fastdist_s": 0.042482799995923415, + "baseline_s": 0.038826999996672384, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.0008097661890645, + "baseline_noise_pct": 2.1209467605800185, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.9139463500616288 + }, + { + "group": "batch", + "case": "poisson_cdf", + "n": 1000000, + "fastdist_s": 0.014319800000521354, + "baseline_s": 0.06409549999807496, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.348496484621267, + "baseline_noise_pct": 2.279099165514824, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.476005251172598 + }, + { + "group": "batch", + "case": "bernoulli_pmf", + "n": 1000000, + "fastdist_s": 0.0035454000026220456, + "baseline_s": 0.03958229999989271, + "baseline_name": "scipy", + "fastdist_noise_pct": 4.191346430711005, + "baseline_noise_pct": 1.9463244980355954, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 11.164410213408676 + }, + { + "group": "scalar", + "case": "normal_pdf", + "n": 20000, + "fastdist_s": 0.007400599992251955, + "baseline_s": 0.46692790000815876, + "baseline_name": "scipy", + "fastdist_noise_pct": 4.403697106707576, + "baseline_noise_pct": 2.9855572975224227, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 63.09324926316894 + }, + { + "group": "scalar", + "case": "normal_cdf", + "n": 20000, + "fastdist_s": 0.0076689999987138435, + "baseline_s": 0.4433584000071278, + "baseline_name": "scipy", + "fastdist_noise_pct": 8.102751367556984, + "baseline_noise_pct": 44.329756689952205, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 57.811761648387375 + }, + { + "group": "scalar", + "case": "gamma_cdf", + "n": 20000, + "fastdist_s": 0.012225700003909878, + "baseline_s": 0.44411899999249727, + "baseline_name": "scipy", + "fastdist_noise_pct": 13.579590497367136, + "baseline_noise_pct": 48.97664815326942, + "max_abs_diff": 1.177946629127291e-13, + "speedup": 36.326672489138815 + }, + { + "group": "scalar", + "case": "chi_square_cdf", + "n": 20000, + "fastdist_s": 0.011841999992611818, + "baseline_s": 0.4471897000039462, + "baseline_name": "scipy", + "fastdist_noise_pct": 12.589089821608326, + "baseline_noise_pct": 1.791297964644493, + "max_abs_diff": 1.177946629127291e-13, + "speedup": 37.763021472973 + }, + { + "group": "scalar", + "case": "beta_cdf", + "n": 20000, + "fastdist_s": 0.009902699996018782, + "baseline_s": 0.48498940000717994, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.77702039464432, + "baseline_noise_pct": 1.304523355333471, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 48.97547135651509 + }, + { + "group": "cuda", + "case": "normal_pdf", + "n": 1000, + "fastdist_s": 4.220000118948519e-05, + "baseline_s": 5.299996701069176e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 9.715644549825136, + "baseline_noise_pct": 3.7736159884463207, + "max_abs_diff": 5.551115123125783e-17, + "speedup": 0.12559233534784248 + }, + { + "group": "cuda", + "case": "normal_cdf", + "n": 1000, + "fastdist_s": 3.979999746661633e-05, + "baseline_s": 7.200011168606579e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 18.59296133321877, + "baseline_noise_pct": 4.166691930369193, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.1809048147464292 + }, + { + "group": "cuda", + "case": "normal_logpdf", + "n": 1000, + "fastdist_s": 3.429999924264848e-05, + "baseline_s": 2.0000006770715117e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 18.367359926145845, + "baseline_noise_pct": 5.000036379775755, + "max_abs_diff": 0.0, + "speedup": 0.05830905892804566 + }, + { + "group": "cuda", + "case": "exponential_pdf", + "n": 1000, + "fastdist_s": 3.330000618007034e-05, + "baseline_s": 4.3000036384910345e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 21.321304917628748, + "baseline_noise_pct": 2.3252587192971768, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 0.12912921442833053 + }, + { + "group": "cuda", + "case": "uniform_pdf", + "n": 1000, + "fastdist_s": 3.929999365936965e-05, + "baseline_s": 3.5999983083456755e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 5.089061068066317, + "baseline_noise_pct": 5.555600468895267, + "max_abs_diff": 0.0, + "speedup": 0.09160302516963352 + }, + { + "group": "cuda", + "case": "normal_pdf", + "n": 100000, + "fastdist_s": 0.0011142000003019348, + "baseline_s": 0.0004138999938732013, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 1.9834866454179798, + "baseline_noise_pct": 0.6040117229583907, + "max_abs_diff": 5.551115123125783e-17, + "speedup": 0.37147728752561426 + }, + { + "group": "cuda", + "case": "normal_cdf", + "n": 100000, + "fastdist_s": 0.0011347000108798966, + "baseline_s": 0.0008879999950295314, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 0.8989151545377602, + "baseline_noise_pct": 1.024776028734789, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.7825856935886842 + }, + { + "group": "cuda", + "case": "normal_logpdf", + "n": 100000, + "fastdist_s": 0.0011178999993717298, + "baseline_s": 7.779999577905983e-05, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 2.066374567027151, + "baseline_noise_pct": 2.9563208716186202, + "max_abs_diff": 0.0, + "speedup": 0.06959477218247084 + }, + { + "group": "cuda", + "case": "exponential_pdf", + "n": 100000, + "fastdist_s": 0.0011070999898947775, + "baseline_s": 0.00031180000223685056, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 1.3548922853318428, + "baseline_noise_pct": 0.12828833961761693, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.28163671310889005 + }, + { + "group": "cuda", + "case": "uniform_pdf", + "n": 100000, + "fastdist_s": 0.001104700000723824, + "baseline_s": 9.40999889280647e-05, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 1.511722294184848, + "baseline_noise_pct": 0.21255695892462415, + "max_abs_diff": 0.0, + "speedup": 0.08518148716068463 + }, + { + "group": "cuda", + "case": "normal_pdf", + "n": 1000000, + "fastdist_s": 0.011082500001066364, + "baseline_s": 0.00500260000990238, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 0.6839611558285116, + "baseline_noise_pct": 2.3887577044887554, + "max_abs_diff": 5.551115123125783e-17, + "speedup": 0.4513963464399754 + }, + { + "group": "cuda", + "case": "normal_cdf", + "n": 1000000, + "fastdist_s": 0.011227099996176548, + "baseline_s": 0.009981000010157004, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 0.7223593482073891, + "baseline_noise_pct": 2.405570528546293, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.8890096296956551 + }, + { + "group": "cuda", + "case": "normal_logpdf", + "n": 1000000, + "fastdist_s": 0.011060100005124696, + "baseline_s": 0.0013895000010961667, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 0.6844422229949404, + "baseline_noise_pct": 5.656711177304609, + "max_abs_diff": 0.0, + "speedup": 0.1256317755221329 + }, + { + "group": "cuda", + "case": "exponential_pdf", + "n": 1000000, + "fastdist_s": 0.010848000005353242, + "baseline_s": 0.0037426000053528696, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 0.4166666197163502, + "baseline_noise_pct": 7.452038266225737, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.3450036876388257 + }, + { + "group": "cuda", + "case": "uniform_pdf", + "n": 1000000, + "fastdist_s": 0.010841900002560578, + "baseline_s": 0.0015518999862251803, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 0.08670065961170549, + "baseline_noise_pct": 15.96752538345291, + "max_abs_diff": 0.0, + "speedup": 0.14313911637800214 + }, + { + "group": "sample", + "case": "normal_sample", + "n": 100000, + "fastdist_s": 0.02779049999662675, + "baseline_s": 0.0008710999973118305, + "baseline_name": "numpy", + "fastdist_noise_pct": 1.496554558574892, + "baseline_noise_pct": 1.021697806083693, + "max_abs_diff": null, + "speedup": 0.03134524378537867 + }, + { + "group": "sample", + "case": "uniform_sample", + "n": 100000, + "fastdist_s": 0.025481100004981272, + "baseline_s": 0.00022679999528918415, + "baseline_name": "numpy", + "fastdist_noise_pct": 1.4783505697430512, + "baseline_noise_pct": 0.13227614223073036, + "max_abs_diff": null, + "speedup": 0.00890071446071195 + }, + { + "group": "sample", + "case": "normal_sample", + "n": 1000000, + "fastdist_s": 0.29453850000572857, + "baseline_s": 0.010220500000286847, + "baseline_name": "numpy", + "fastdist_noise_pct": 1.4221570347204793, + "baseline_noise_pct": 0.34049203558537494, + "max_abs_diff": null, + "speedup": 0.034700047702042575 + }, + { + "group": "sample", + "case": "uniform_sample", + "n": 1000000, + "fastdist_s": 0.27313639999192674, + "baseline_s": 0.003173000004608184, + "baseline_name": "numpy", + "fastdist_noise_pct": 1.0111431566147289, + "baseline_noise_pct": 8.011345724245563, + "max_abs_diff": null, + "speedup": 0.011616906441990047 + } + ] +} From 9427b2245547c8c79354b6f0efe177cedd91c4f8 Mon Sep 17 00:00:00 2001 From: ghosteau Date: Fri, 11 Sep 2026 21:23:45 -0400 Subject: [PATCH 16/20] Warm the GPU before timing it, and withdraw the cold CUDA figures The previous entry in BENCHMARKS.md concluded that the CUDA path loses to the CPU path at every size. That was wrong, and the error was mine: an idle NVIDIA card drops its clocks and downtrains its PCIe link, and the benchmark runs its CPU groups first, so by the time the cuda group ran the card sat at 210 MHz on a Gen1 link instead of 2865 MHz on Gen4. Each case is far too short to train it back up, and the harness's single warmup call does not come close. Same call, same machine, normal_pdf at n = 1,000,000, minimum of 7 rounds: 10.16 ms cold against 2.47 ms warm, a 4.1x difference. The cold figure is what got recorded. A standalone probe confirms the hardware was never the constraint: 8 MB copies sustain 22.7 GB/s pageable and 25.8 GB/s pinned, the kernel alone is 0.19 ms, and the round trip the executor performs is 1.44 ms. run.py now runs sustained GPU work before the cuda group and reports the device state it reached, and times those cases over 15 rounds rather than 7. Re-measured warm, the picture is per function rather than uniform. At n = 1,000,000 the GPU wins normal_cdf 4.59x, normal_pdf 2.58x and exponential_pdf 2.05x, and still loses normal_logpdf 0.68x and uniform_pdf 0.77x, where the CPU path is fast enough that the transfer dominates. At n = 1,000 it loses everything to launch overhead. The superseded section is marked in place rather than rewritten, and the cold report stays committed, so the mistake and its evidence remain visible. This also revises the recommendation that came out of the bad data: turning auto-dispatch off wholesale is not the answer. The thresholds want to be per function. Co-Authored-By: Claude Opus 5 --- BENCHMARKS.md | 92 +++ ...0.1.0_20260912T012110+0000_b7752f22e7.json | 666 ++++++++++++++++++ benchmarks/run.py | 34 +- 3 files changed, 790 insertions(+), 2 deletions(-) create mode 100644 benchmarks/results/0.1.0_20260912T012110+0000_b7752f22e7.json diff --git a/BENCHMARKS.md b/BENCHMARKS.md index 1268893..e0b753f 100644 --- a/BENCHMARKS.md +++ b/BENCHMARKS.md @@ -377,6 +377,9 @@ correctness-pass report above, taken on the same machine. ### The GPU path is slower than the CPU path at every size +**Superseded.** These GPU figures were measured on an idle, downclocked card and +are wrong. See the correction entry below. + Until this branch the CUDA backend did not build on Windows, and once built it could not be imported, and once imported it crashed on its first call. With those fixed it runs, agrees with the CPU path to machine precision, and loses @@ -507,6 +510,95 @@ the pre-branch report; the largest drift was +4.9% (uniform_pdf, n=1,000,000). --- +## Unreleased — correction: the CUDA figures above were measured on a cold GPU + +The previous entry concluded that the GPU path loses to the CPU path at every +size. That conclusion was an artifact of how it was measured, not a property of +the backend, and it is withdrawn. + +### What went wrong + +An idle NVIDIA card drops its clocks and downtrains its PCIe link. On this +machine that is 210 MHz and Gen1, against 2865 MHz and Gen4 under load. The +benchmark runs its CPU groups first, which takes minutes, so the GPU was cold +by the time the `cuda` group ran, and each case is far too short to train it +back up. The single warmup call the harness makes is nowhere near enough. + +Same call, same machine, `normal_pdf` at n = 1,000,000, minimum of 7 rounds: + +| GPU state before timing | clocks / link | time | +|---|---|---:| +| idle during the CPU groups | 210 MHz, Gen1 | 10.16 ms | +| after 3 s of GPU work | 2865 MHz, Gen4 | 2.47 ms | + +A standalone CUDA probe confirms the hardware was never the problem: 8 MB +copies sustain 22.7 GB/s pageable and 25.8 GB/s pinned, the kernel alone takes +0.19 ms, and the full round trip the executor performs takes 1.44 ms. + +`benchmarks/run.py` now warms the GPU before timing the `cuda` group and +records the device state it reached, which is printed in the run and shown +below. + +### The corrected picture + +The GPU wins where there is real arithmetic per element, and loses where the +CPU path is already very fast and the transfer dominates. At n = 1,000,000 it +wins for `exponential_pdf`, `normal_cdf`, `normal_pdf` and loses for `normal_logpdf`, `uniform_pdf`. + +| case | n | cold (withdrawn) | warm (this run) | +|---|---:|---:|---:| +| `normal_cdf` | 1,000,000 | 0.89x | 4.59x | +| `normal_pdf` | 1,000,000 | 0.45x | 2.58x | +| `exponential_pdf` | 1,000,000 | 0.35x | 2.05x | +| `normal_logpdf` | 1,000,000 | 0.13x | 0.68x | +| `uniform_pdf` | 1,000,000 | 0.14x | 0.77x | + +At n = 1,000 the GPU loses every case: launch and transfer overhead swamps a +few microseconds of work. + +This bears directly on the auto-dispatch thresholds, which default to 100,000 +for every function. That is about right for `normal_pdf`, `normal_cdf` and +`exponential_pdf`, which are 2.2x to 4.2x faster on the GPU there, and wrong for +`normal_logpdf` and `uniform_pdf`, which are still slower on the GPU at +n = 1,000,000. The thresholds want to be per function, which is what +`config.auto_tune` is for -- though its search starts at 500,000 and cannot +return "never", so it cannot express the `uniform_pdf` case today. + +One more thing this run shows: the first call to each kernel costs about 12.8 ms +against 2.7 ms afterwards, because `CMAKE_CUDA_ARCHITECTURES` is never set. The +binary carries PTX for compute_52 and the driver JIT-compiles it for the actual +card on first use. + + +- **Version** 0.1.0 (`b7752f22e7` on `chore/release-prep`, working tree dirty) +- **Measured** 2026-09-12T01:21:10+00:00 +- **CPU** AMD Ryzen 7 7700 8-Core Processor +- **Platform** Windows-11-10.0.26200-SP0 +- **Toolchain** Python 3.14.2, numpy 2.5.2, scipy 1.18.1 +- **CUDA** available + +### cuda (GPU path vs the CPU path; below 1x means the GPU is slower) + +| case | n | fastdist | baseline | speedup | max abs diff | +|---|---:|---:|---:|---:|---:| +| `normal_pdf` | 1,000 | 32.40 us | 6.70 us (fastdist-cpu) | **0.21x** | 5.6e-17 | +| `normal_cdf` | 1,000 | 25.30 us | 7.00 us (fastdist-cpu) | **0.28x** | 2.2e-16 | +| `normal_logpdf` | 1,000 | 27.30 us | 4.40 us (fastdist-cpu) | **0.16x** | 0.0e+00 | +| `exponential_pdf` | 1,000 | 27.80 us | 5.30 us (fastdist-cpu) | **0.19x** | 1.1e-16 | +| `uniform_pdf` | 1,000 | 29.90 us | 3.50 us (fastdist-cpu) | **0.12x** | 0.0e+00 | +| `normal_pdf` | 100,000 | 193.40 us | 419.10 us (fastdist-cpu) | **2.17x** | 5.6e-17 | +| `normal_cdf` | 100,000 | 211.90 us | 896.70 us (fastdist-cpu) | **4.23x** | 2.2e-16 | +| `normal_logpdf` | 100,000 | 200.20 us | 78.30 us (fastdist-cpu) | **0.39x** | 0.0e+00 | +| `exponential_pdf` | 100,000 | 188.20 us | 311.90 us (fastdist-cpu) | **1.66x** | 2.2e-16 | +| `uniform_pdf` | 100,000 | 184.70 us | 95.50 us (fastdist-cpu) | **0.52x** | 0.0e+00 | +| `normal_pdf` | 1,000,000 | 1.96 ms | 5.05 ms (fastdist-cpu) | **2.58x** | 5.6e-17 | +| `normal_cdf` | 1,000,000 | 2.17 ms | 9.96 ms (fastdist-cpu) | **4.59x** | 2.2e-16 | +| `normal_logpdf` | 1,000,000 | 2.02 ms | 1.37 ms (fastdist-cpu) | **0.68x** | 0.0e+00 | +| `exponential_pdf` | 1,000,000 | 1.92 ms | 3.94 ms (fastdist-cpu) | **2.05x** | 2.2e-16 | +| `uniform_pdf` | 1,000,000 | 1.86 ms | 1.43 ms (fastdist-cpu) | **0.77x** | 0.0e+00 | + +--- + ## Changes to record here Add an entry when a release ships, or when a change is made specifically to diff --git a/benchmarks/results/0.1.0_20260912T012110+0000_b7752f22e7.json b/benchmarks/results/0.1.0_20260912T012110+0000_b7752f22e7.json new file mode 100644 index 0000000..88c162c --- /dev/null +++ b/benchmarks/results/0.1.0_20260912T012110+0000_b7752f22e7.json @@ -0,0 +1,666 @@ +{ + "environment": { + "timestamp_utc": "2026-09-12T01:21:10+00:00", + "fastdist_version": "0.1.0", + "git_commit": "b7752f22e79579ae63c86dcc8eec6c2839f2e082", + "git_branch": "chore/release-prep", + "git_dirty": true, + "cuda_available": true, + "python": "3.14.2", + "numpy": "2.5.2", + "scipy": "1.18.1", + "platform": "Windows-11-10.0.26200-SP0", + "processor": "AMD Ryzen 7 7700 8-Core Processor", + "machine": "AMD64" + }, + "results": [ + { + "group": "batch", + "case": "normal_pdf", + "n": 1000, + "fastdist_s": 5.280000041238964e-06, + "baseline_s": 3.17219999851659e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.1742403321540706, + "baseline_noise_pct": 1.015068796768806, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 6.00795449572047 + }, + { + "group": "batch", + "case": "normal_cdf", + "n": 1000, + "fastdist_s": 6.7819998366758225e-06, + "baseline_s": 2.9927999712526798e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.3538806152434959, + "baseline_noise_pct": 0.6749544771276303, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.41285762802317 + }, + { + "group": "batch", + "case": "normal_logpdf", + "n": 1000, + "fastdist_s": 1.873999717645347e-06, + "baseline_s": 3.193399985320866e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.8537953171091134, + "baseline_noise_pct": 0.7077095750126259, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 17.040557451808613 + }, + { + "group": "batch", + "case": "exponential_pdf", + "n": 1000, + "fastdist_s": 4.140000091865659e-06, + "baseline_s": 2.970999979879707e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.3864763641752243, + "baseline_noise_pct": 1.2453735831768373, + "max_abs_diff": 0.0, + "speedup": 7.176328294574623 + }, + { + "group": "batch", + "case": "exponential_cdf", + "n": 1000, + "fastdist_s": 4.456000169739127e-06, + "baseline_s": 3.0140000162646175e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.4039527911852564, + "baseline_noise_pct": 0.8493683464578627, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 6.763913602905158 + }, + { + "group": "batch", + "case": "uniform_pdf", + "n": 1000, + "fastdist_s": 2.043999847956002e-06, + "baseline_s": 3.2561999978497626e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.6849367732051003, + "baseline_noise_pct": 2.112892330127997, + "max_abs_diff": 0.0, + "speedup": 15.930529550214791 + }, + { + "group": "batch", + "case": "uniform_cdf", + "n": 1000, + "fastdist_s": 2.109999768435955e-06, + "baseline_s": 3.246000036597252e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.5687523586519367, + "baseline_noise_pct": 1.3739966620435202, + "max_abs_diff": 0.0, + "speedup": 15.38388811769094 + }, + { + "group": "batch", + "case": "poisson_pmf", + "n": 1000, + "fastdist_s": 4.060800012666732e-05, + "baseline_s": 3.827399981673807e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.3053580074299978, + "baseline_noise_pct": 1.2227623801120773, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.9425236332090013 + }, + { + "group": "batch", + "case": "poisson_cdf", + "n": 1000, + "fastdist_s": 9.926000493578614e-06, + "baseline_s": 6.975400028750301e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.4029733874423299, + "baseline_noise_pct": 3.9481606849868447, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 7.027402460097466 + }, + { + "group": "batch", + "case": "bernoulli_pmf", + "n": 1000, + "fastdist_s": 1.981999957934022e-06, + "baseline_s": 4.902400018181652e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.6145431050939105, + "baseline_noise_pct": 0.7955288385749166, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 24.734612120233184 + }, + { + "group": "batch", + "case": "normal_pdf", + "n": 100000, + "fastdist_s": 0.00041390000842511654, + "baseline_s": 0.0014322999923024327, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.1449636522064908, + "baseline_noise_pct": 8.070935400529606, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 3.460497615722003 + }, + { + "group": "batch", + "case": "normal_cdf", + "n": 100000, + "fastdist_s": 0.0008914000063668936, + "baseline_s": 0.0021191999840084463, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.33654854366411996, + "baseline_noise_pct": 1.0145346524488028, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 2.3773838555888447 + }, + { + "group": "batch", + "case": "normal_logpdf", + "n": 100000, + "fastdist_s": 7.910002022981644e-05, + "baseline_s": 0.0017220000154338777, + "baseline_name": "scipy", + "fastdist_noise_pct": 3.919081805658578, + "baseline_noise_pct": 2.938442356923492, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 21.7699061318922 + }, + { + "group": "batch", + "case": "exponential_pdf", + "n": 100000, + "fastdist_s": 0.00030899999546818435, + "baseline_s": 0.001293000008445233, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.5566443904626466, + "baseline_noise_pct": 3.6890939327134746, + "max_abs_diff": 0.0, + "speedup": 4.184466108118 + }, + { + "group": "batch", + "case": "exponential_cdf", + "n": 100000, + "fastdist_s": 0.0003539999888744205, + "baseline_s": 0.0016522999794688076, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.9774080275766202, + "baseline_noise_pct": 6.875266138530395, + "max_abs_diff": 1.6653345369377348e-16, + "speedup": 4.667514212987593 + }, + { + "group": "batch", + "case": "uniform_pdf", + "n": 100000, + "fastdist_s": 9.370001498609781e-05, + "baseline_s": 0.0014755000011064112, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.21344876297869114, + "baseline_noise_pct": 8.885122299404191, + "max_abs_diff": 0.0, + "speedup": 15.74706259465732 + }, + { + "group": "batch", + "case": "uniform_cdf", + "n": 100000, + "fastdist_s": 9.929999941959977e-05, + "baseline_s": 0.0016921000205911696, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.10070570194746539, + "baseline_noise_pct": 1.7138459687230747, + "max_abs_diff": 0.0, + "speedup": 17.040282280778985 + }, + { + "group": "batch", + "case": "poisson_pmf", + "n": 100000, + "fastdist_s": 0.0040083000203594565, + "baseline_s": 0.0030960000003688037, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.14719423915081262, + "baseline_noise_pct": 0.5846260143686062, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.7723972718217735 + }, + { + "group": "batch", + "case": "poisson_cdf", + "n": 100000, + "fastdist_s": 0.001305099984165281, + "baseline_s": 0.006005600007483736, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.5593439174554613, + "baseline_noise_pct": 2.0630741783490154, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.60163978265988 + }, + { + "group": "batch", + "case": "bernoulli_pmf", + "n": 100000, + "fastdist_s": 0.0002928999892901629, + "baseline_s": 0.0034214000042993575, + "baseline_name": "scipy", + "fastdist_noise_pct": 4.848078297554037, + "baseline_noise_pct": 1.554918967982971, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 11.681120277918241 + }, + { + "group": "batch", + "case": "normal_pdf", + "n": 1000000, + "fastdist_s": 0.004730400018161163, + "baseline_s": 0.01965510001173243, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.1055296004086053, + "baseline_noise_pct": 1.4082859514871244, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 4.155060869328533 + }, + { + "group": "batch", + "case": "normal_cdf", + "n": 1000000, + "fastdist_s": 0.009499399980995804, + "baseline_s": 0.02197219998924993, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.5958273537268727, + "baseline_noise_pct": 2.542758601011949, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 2.3130092461846865 + }, + { + "group": "batch", + "case": "normal_logpdf", + "n": 1000000, + "fastdist_s": 0.0013212999911047518, + "baseline_s": 0.02158719999715686, + "baseline_name": "scipy", + "fastdist_noise_pct": 11.261636190795137, + "baseline_noise_pct": 1.4939409047714771, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 16.33784919585717 + }, + { + "group": "batch", + "case": "exponential_pdf", + "n": 1000000, + "fastdist_s": 0.0037620000075548887, + "baseline_s": 0.017204100004164502, + "baseline_name": "scipy", + "fastdist_noise_pct": 6.7623600919612965, + "baseline_noise_pct": 3.1329741192004192, + "max_abs_diff": 0.0, + "speedup": 4.573125988733398 + }, + { + "group": "batch", + "case": "exponential_cdf", + "n": 1000000, + "fastdist_s": 0.004369000002043322, + "baseline_s": 0.019657199998619035, + "baseline_name": "scipy", + "fastdist_noise_pct": 5.584801903802188, + "baseline_noise_pct": 1.9539914859229277, + "max_abs_diff": 1.6653345369377348e-16, + "speedup": 4.499244675995795 + }, + { + "group": "batch", + "case": "uniform_pdf", + "n": 1000000, + "fastdist_s": 0.0014437000208999962, + "baseline_s": 0.01832169998669997, + "baseline_name": "scipy", + "fastdist_noise_pct": 3.449470120595329, + "baseline_noise_pct": 3.201122239967099, + "max_abs_diff": 0.0, + "speedup": 12.690794293455992 + }, + { + "group": "batch", + "case": "uniform_cdf", + "n": 1000000, + "fastdist_s": 0.0014933999918866903, + "baseline_s": 0.01940150000154972, + "baseline_name": "scipy", + "fastdist_noise_pct": 7.298782353334496, + "baseline_noise_pct": 2.5745431579916516, + "max_abs_diff": 0.0, + "speedup": 12.99149598697854 + }, + { + "group": "batch", + "case": "poisson_pmf", + "n": 1000000, + "fastdist_s": 0.04207510000560433, + "baseline_s": 0.038483699987409636, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.1319022193195423, + "baseline_noise_pct": 1.0258889808943503, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.9146431020314552 + }, + { + "group": "batch", + "case": "poisson_cdf", + "n": 1000000, + "fastdist_s": 0.01433009997708723, + "baseline_s": 0.06366220000199974, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.2023644753641816, + "baseline_noise_pct": 1.6898567507217563, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.4425510013043095 + }, + { + "group": "batch", + "case": "bernoulli_pmf", + "n": 1000000, + "fastdist_s": 0.003533200011588633, + "baseline_s": 0.03915130000677891, + "baseline_name": "scipy", + "fastdist_noise_pct": 4.200157515873911, + "baseline_noise_pct": 1.8203737406158362, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 11.080974719338153 + }, + { + "group": "scalar", + "case": "normal_pdf", + "n": 20000, + "fastdist_s": 0.00756949998321943, + "baseline_s": 0.4613825000124052, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.6896758537662224, + "baseline_noise_pct": 2.0413214576598815, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 60.952837180160984 + }, + { + "group": "scalar", + "case": "normal_cdf", + "n": 20000, + "fastdist_s": 0.007806199981132522, + "baseline_s": 0.45165249999263324, + "baseline_name": "scipy", + "fastdist_noise_pct": 4.156952511665757, + "baseline_noise_pct": 1.010998497804553, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 57.858176972697485 + }, + { + "group": "scalar", + "case": "gamma_cdf", + "n": 20000, + "fastdist_s": 0.008554099971661344, + "baseline_s": 0.41531800001394004, + "baseline_name": "scipy", + "fastdist_noise_pct": 3.6473740001318813, + "baseline_noise_pct": 1.0038813619174314, + "max_abs_diff": 1.177946629127291e-13, + "speedup": 48.551922632402736 + }, + { + "group": "scalar", + "case": "chi_square_cdf", + "n": 20000, + "fastdist_s": 0.008138000004692003, + "baseline_s": 0.46164749999297783, + "baseline_name": "scipy", + "fastdist_noise_pct": 7.726714002197936, + "baseline_noise_pct": 21.813093325199663, + "max_abs_diff": 1.177946629127291e-13, + "speedup": 56.72738998854912 + }, + { + "group": "scalar", + "case": "beta_cdf", + "n": 20000, + "fastdist_s": 0.01007650000974536, + "baseline_s": 0.4931773999996949, + "baseline_name": "scipy", + "fastdist_noise_pct": 65.17838509275724, + "baseline_noise_pct": 30.161884953627293, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 48.943323527288705 + }, + { + "group": "cuda", + "case": "normal_pdf", + "n": 1000, + "fastdist_s": 3.2400013878941536e-05, + "baseline_s": 6.699992809444666e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 30.555505651889593, + "baseline_noise_pct": 1.4929846661743624, + "max_abs_diff": 5.551115123125783e-17, + "speedup": 0.2067898129450908 + }, + { + "group": "cuda", + "case": "normal_cdf", + "n": 1000, + "fastdist_s": 2.5300018023699522e-05, + "baseline_s": 6.999995093792677e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 54.94063052886109, + "baseline_noise_pct": 1.4285833076942267, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.2766794508697783 + }, + { + "group": "cuda", + "case": "normal_logpdf", + "n": 1000, + "fastdist_s": 2.7300004148855805e-05, + "baseline_s": 4.399975296109915e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 73.26008692769138, + "baseline_noise_pct": 4.545514677673268, + "max_abs_diff": 0.0, + "speedup": 0.16117123177412873 + }, + { + "group": "cuda", + "case": "exponential_pdf", + "n": 1000, + "fastdist_s": 2.780000795610249e-05, + "baseline_s": 5.299982149153948e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 23.38125692916988, + "baseline_noise_pct": 3.7736263494887594, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 0.19064678533628002 + }, + { + "group": "cuda", + "case": "uniform_pdf", + "n": 1000, + "fastdist_s": 2.989999484270811e-05, + "baseline_s": 3.4999975468963385e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 22.74245733708666, + "baseline_noise_pct": 2.8571666153884534, + "max_abs_diff": 0.0, + "speedup": 0.11705679433419379 + }, + { + "group": "cuda", + "case": "normal_pdf", + "n": 100000, + "fastdist_s": 0.00019339998834766448, + "baseline_s": 0.00041909998981282115, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 10.289559052864387, + "baseline_noise_pct": 3.221192033480725, + "max_abs_diff": 5.551115123125783e-17, + "speedup": 2.167011453275934 + }, + { + "group": "cuda", + "case": "normal_cdf", + "n": 100000, + "fastdist_s": 0.00021190001280047, + "baseline_s": 0.0008966999885160476, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 10.051904803697491, + "baseline_noise_pct": 1.5166708281433883, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.231712762379119 + }, + { + "group": "cuda", + "case": "normal_logpdf", + "n": 100000, + "fastdist_s": 0.00020019998191855848, + "baseline_s": 7.830001413822174e-05, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 3.8461689426629775, + "baseline_noise_pct": 2.298867664200585, + "max_abs_diff": 0.0, + "speedup": 0.39110899705312785 + }, + { + "group": "cuda", + "case": "exponential_pdf", + "n": 100000, + "fastdist_s": 0.00018820000695995986, + "baseline_s": 0.00031189998844638467, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 4.729016595213827, + "baseline_noise_pct": 6.893231768208463, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 1.657279367225222 + }, + { + "group": "cuda", + "case": "uniform_pdf", + "n": 100000, + "fastdist_s": 0.00018470000941306353, + "baseline_s": 9.549999958835542e-05, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 3.410930938047765, + "baseline_noise_pct": 0.6282770379919271, + "max_abs_diff": 0.0, + "speedup": 0.5170546546902388 + }, + { + "group": "cuda", + "case": "normal_pdf", + "n": 1000000, + "fastdist_s": 0.0019588000141084194, + "baseline_s": 0.005046999984188005, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 7.177862142103715, + "baseline_noise_pct": 1.491975839795586, + "max_abs_diff": 5.551115123125783e-17, + "speedup": 2.576577469796084 + }, + { + "group": "cuda", + "case": "normal_cdf", + "n": 1000000, + "fastdist_s": 0.0021715000038966537, + "baseline_s": 0.009963300020899624, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 3.826847447893623, + "baseline_noise_pct": 2.774181039279664, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.588210915505851 + }, + { + "group": "cuda", + "case": "normal_logpdf", + "n": 1000000, + "fastdist_s": 0.002023000008193776, + "baseline_s": 0.001368099998217076, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 5.017300378474957, + "baseline_noise_pct": 7.229003472165051, + "max_abs_diff": 0.0, + "speedup": 0.6762728584655698 + }, + { + "group": "cuda", + "case": "exponential_pdf", + "n": 1000000, + "fastdist_s": 0.0019223999988753349, + "baseline_s": 0.0039360999944619834, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 3.6776953036358093, + "baseline_noise_pct": 5.058306519818503, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 2.0474927157536036 + }, + { + "group": "cuda", + "case": "uniform_pdf", + "n": 1000000, + "fastdist_s": 0.0018558000156190246, + "baseline_s": 0.0014310000115074217, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 4.80116436649228, + "baseline_noise_pct": 6.02375865045763, + "max_abs_diff": 0.0, + "speedup": 0.7710960229893599 + }, + { + "group": "sample", + "case": "normal_sample", + "n": 100000, + "fastdist_s": 0.028305300016654655, + "baseline_s": 0.0009171999990940094, + "baseline_name": "numpy", + "fastdist_noise_pct": 1.6650591216430803, + "baseline_noise_pct": 2.627563117107823, + "max_abs_diff": null, + "speedup": 0.03240382538091223 + }, + { + "group": "sample", + "case": "uniform_sample", + "n": 100000, + "fastdist_s": 0.025234400003682822, + "baseline_s": 0.00024029999622143805, + "baseline_name": "numpy", + "fastdist_noise_pct": 1.48408521461864, + "baseline_noise_pct": 0.9571321038993441, + "max_abs_diff": null, + "speedup": 0.009522714872807262 + }, + { + "group": "sample", + "case": "normal_sample", + "n": 1000000, + "fastdist_s": 0.2949687000073027, + "baseline_s": 0.010196000017458573, + "baseline_name": "numpy", + "fastdist_noise_pct": 0.7327896197154811, + "baseline_noise_pct": 0.9523340771032798, + "max_abs_diff": null, + "speedup": 0.034566379474181994 + }, + { + "group": "sample", + "case": "uniform_sample", + "n": 1000000, + "fastdist_s": 0.2677903999865521, + "baseline_s": 0.0033625999931246042, + "baseline_name": "numpy", + "fastdist_noise_pct": 6.222590511590453, + "baseline_noise_pct": 5.329210316642336, + "max_abs_diff": null, + "speedup": 0.012556835470179169 + } + ] +} diff --git a/benchmarks/run.py b/benchmarks/run.py index 00c586a..44f744e 100644 --- a/benchmarks/run.py +++ b/benchmarks/run.py @@ -35,6 +35,13 @@ GPU timings include the host-to-device copy and the copy back, because a caller cannot avoid those. + The GPU is warmed before these are timed. An idle NVIDIA card drops + its clocks and downtrains the PCIe link -- on the reference machine, + 210 MHz and Gen1 against 2865 MHz and Gen4 -- and does not recover + within a short burst. Timed cold, straight after the CPU groups, the + same call measured 4.1x slower and made the GPU look like a loss at + every size. + sample Drawing variates. fastdist samples one value per call, while numpy fills an array in one call, so numpy is expected to win by a wide margin. It is measured anyway: this is a real gap in the library and @@ -216,6 +223,26 @@ def cuda_cases(sizes): lambda x=x_real: core.uniform_pdf_cpu(x, -3.0, 3.0, 0.0)) +def warm_up_gpu(seconds: float = 3.0) -> str: + """Run sustained GPU work so clocks and the PCIe link train up before timing. + + Returns the device state afterwards, for the report. + """ + import subprocess + import time + + x = np.random.default_rng(4242).normal(0.0, 1.0, 1_000_000) + deadline = time.time() + seconds + while time.time() < deadline: + core.normal_pdf_cuda(x, 0.0, 1.0, 0.0) + + probe = subprocess.run( + ["nvidia-smi", "--query-gpu=name,clocks.current.sm,pcie.link.gen.current,pcie.link.width.current", + "--format=csv,noheader"], + capture_output=True, text=True) + return probe.stdout.strip() if probe.returncode == 0 else "unknown" + + def run(sizes, sample_sizes) -> list[Result]: results: list[Result] = [] @@ -238,12 +265,15 @@ def run(sizes, sample_sizes) -> list[Result]: results.append(measure("scalar", case, n, fd, sp, "scipy", repeat=21)) print(f" scalar {case:<18} n={n:<9,} {_fmt(results[-1])}") - for case, n, gpu, cpu in cuda_cases(sizes): + cuda_list = list(cuda_cases(sizes)) + if cuda_list: + print(f" warming the GPU; device now at {warm_up_gpu()}") + for case, n, gpu, cpu in cuda_list: # The "fastdist" column is the GPU path and the baseline is the CPU # path, so `speedup` reads as "how much the GPU buys over the CPU". # measure() also checks the two agree numerically, which is the part # worth having: a kernel that is fast and wrong is the failure mode. - results.append(measure("cuda", case, n, gpu, cpu, "fastdist-cpu")) + results.append(measure("cuda", case, n, gpu, cpu, "fastdist-cpu", repeat=15)) print(f" cuda {case:<18} n={n:<9,} {_fmt(results[-1])}") for case, n, fd, np_fn in sample_cases(sample_sizes): From da5f274059956eb4a82cd491daa3a6383b89ac79 Mon Sep 17 00:00:00 2001 From: ghosteau Date: Fri, 11 Sep 2026 21:34:04 -0400 Subject: [PATCH 17/20] Apply one validation contract across every distribution Adds tests/python/test_validation.py, which states the contract and holds all twelve classes to it, then fixes the three places that did not meet it. The contract: a parameter that cannot describe a distribution is refused at construction (ValueError for a bad value, TypeError for a bad type), and a non-finite input x is answered with nan rather than an exception, matching the C++ core and numpy's elementwise behaviour. Fixed: - Seven classes accepted nan as a parameter. Every comparison against nan is false, so a range check like `p < 0 or p > 1` waves it through, and the result was a distribution that silently returned nan from every method. Bernoulli, Beta, ChiSquare, Exponential, Gamma, Poisson and Uniform now check finiteness the way Normal already did; Binomial, Geometric and NegativeBinomial happened to reject nan through the shape of their range checks and now do so explicitly. - The array-capable classes accepted a string as data: numpy reads "0.5" as a number, so Normal(0, 1).pdf("0.5") returned array([0.352]) instead of raising. The scalar-only classes already refused it. str and bytes are now rejected before the array branch. - gamma_pdf_scalar tested x < 0 before finiteness, so f(-inf) was 0.0 while every other density returns nan for a non-finite input. chi_square_pdf_scalar delegates to it and had its own copy of the same ordering. Both now check finiteness first. 1879 Python tests pass, ctest 14/14 on both the CPU and CUDA builds, mypy clean. Co-Authored-By: Claude Opus 5 --- python/fastdist/distributions/bernoulli.py | 8 ++ python/fastdist/distributions/beta.py | 5 + python/fastdist/distributions/binomial.py | 3 + python/fastdist/distributions/chi_square.py | 3 + python/fastdist/distributions/exponential.py | 8 ++ python/fastdist/distributions/gamma.py | 5 + python/fastdist/distributions/geometric.py | 3 + .../distributions/negative_binomial.py | 3 + python/fastdist/distributions/normal.py | 5 + python/fastdist/distributions/poisson.py | 8 ++ python/fastdist/distributions/uniform.py | 10 ++ python/fastdist/distributions/utils.py | 5 + src/math/chi_square.cpp | 4 +- src/math/gamma.cpp | 4 +- tests/python/test_validation.py | 117 ++++++++++++++++++ 15 files changed, 188 insertions(+), 3 deletions(-) create mode 100644 tests/python/test_validation.py diff --git a/python/fastdist/distributions/bernoulli.py b/python/fastdist/distributions/bernoulli.py index 0d4a8ac..66df527 100644 --- a/python/fastdist/distributions/bernoulli.py +++ b/python/fastdist/distributions/bernoulli.py @@ -11,6 +11,7 @@ from fastdist import config +import math import numpy as np from numbers import Real from typing import Sequence, SupportsFloat, Union, cast @@ -105,6 +106,8 @@ def _validate_params(p: SupportsFloat) -> None: if not isinstance(p, Real): raise TypeError("p must be a real number") + if not math.isfinite(p): + raise ValueError("p must be finite") if float(p) < 0 or float(p) > 1: raise ValueError("p must be in the interval [0, 1]") @@ -143,6 +146,11 @@ def _validate_inputs(_input: Union[int, SupportsFloat, Sequence[int], ArrayLike] if _input is None: raise TypeError(f"{input_name} must not be None") + # numpy reads "0.5" as data, so a string would reach the array branch + # and come back as a one-element result instead of a TypeError. + if isinstance(_input, (str, bytes)): + raise TypeError(f"{input_name} must be a real number or a sequence of them") + # Scalar input validated: Union[int, float, np.ndarray] if isinstance(_input, Real): diff --git a/python/fastdist/distributions/beta.py b/python/fastdist/distributions/beta.py index a0aeefb..602db1b 100644 --- a/python/fastdist/distributions/beta.py +++ b/python/fastdist/distributions/beta.py @@ -9,6 +9,7 @@ "extension has been built." ) from exc +import math import numpy as np from typing import Sequence, Union from numpy.typing import NDArray @@ -49,12 +50,16 @@ def _validate_params(alpha: Union[int, float, None] = None, beta: Union[int, flo if alpha is not None: if not isinstance(alpha, (int, float)): raise TypeError("alpha must be a real number") + if not math.isfinite(alpha): + raise ValueError("alpha must be finite") if alpha <= 0: raise ValueError("alpha must be positive") if beta is not None: if not isinstance(beta, (int, float)): raise TypeError("beta must be a real number") + if not math.isfinite(beta): + raise ValueError("beta must be finite") if beta <= 0: raise ValueError("beta must be positive") diff --git a/python/fastdist/distributions/binomial.py b/python/fastdist/distributions/binomial.py index 95cee65..7d54490 100644 --- a/python/fastdist/distributions/binomial.py +++ b/python/fastdist/distributions/binomial.py @@ -9,6 +9,7 @@ "extension has been built." ) from exc +import math import numpy as np from typing import Sequence, SupportsFloat, Union from numpy.typing import NDArray @@ -55,6 +56,8 @@ def _validate_params(n: Union[int, None] = None, p: Union[SupportsFloat, None] = if p is not None: if not isinstance(p, (int, float)): raise TypeError("p must be a real number") + if not math.isfinite(p): + raise ValueError("p must be finite") if not 0 <= p <= 1: raise ValueError("p must be in the interval [0, 1]") diff --git a/python/fastdist/distributions/chi_square.py b/python/fastdist/distributions/chi_square.py index 6e96527..13fb2a7 100644 --- a/python/fastdist/distributions/chi_square.py +++ b/python/fastdist/distributions/chi_square.py @@ -9,6 +9,7 @@ "extension has been built." ) from exc +import math import numpy as np from typing import Sequence, Union from numpy.typing import NDArray @@ -38,6 +39,8 @@ def _validate_params(k: Union[int, float]) -> None: """Internal validation shared by all methods.""" if not isinstance(k, (int, float)): raise TypeError("k must be a real number") + if not math.isfinite(k): + raise ValueError("k must be finite") if k <= 0: raise ValueError("k must be positive") diff --git a/python/fastdist/distributions/exponential.py b/python/fastdist/distributions/exponential.py index ccd09a8..181a2b2 100644 --- a/python/fastdist/distributions/exponential.py +++ b/python/fastdist/distributions/exponential.py @@ -11,6 +11,7 @@ from fastdist import config +import math import numpy as np from numbers import Real from typing import SupportsFloat, Union, cast @@ -80,6 +81,8 @@ def _validate_params(lambda_: SupportsFloat) -> None: """ if not isinstance(lambda_, Real): raise TypeError("lambda_ must be a real number") + if not math.isfinite(lambda_): + raise ValueError("lambda_ must be finite") if lambda_ <= 0: raise ValueError("lambda_ must be positive") @@ -118,6 +121,11 @@ def _validate_inputs(_input: Union[SupportsFloat, ArrayLike], input_name: str, if _input is None: raise TypeError(f"{input_name} must not be None") + # numpy reads "0.5" as data, so a string would reach the array branch + # and come back as a one-element result instead of a TypeError. + if isinstance(_input, (str, bytes)): + raise TypeError(f"{input_name} must be a real number or a sequence of them") + # Declared up front: without it the type is inferred from the scalar # branch alone and the array branch looks like a bad assignment. validated: Union[float, np.ndarray] diff --git a/python/fastdist/distributions/gamma.py b/python/fastdist/distributions/gamma.py index 6763e91..020e74b 100644 --- a/python/fastdist/distributions/gamma.py +++ b/python/fastdist/distributions/gamma.py @@ -10,6 +10,7 @@ "extension has been built." ) from exc +import math import numpy as np from typing import Sequence, Union from numpy.typing import NDArray @@ -50,11 +51,15 @@ def _validate_params(alpha: Union[int, float, None] = None, theta: Union[int, fl if alpha is not None: if not isinstance(alpha, (int, float)): raise TypeError("alpha must be a real number") + if not math.isfinite(alpha): + raise ValueError("alpha must be finite") if alpha <= 0: raise ValueError("alpha must be positive") if theta is not None: if not isinstance(theta, (int, float)): raise TypeError("theta must be a real number") + if not math.isfinite(theta): + raise ValueError("theta must be finite") if theta <= 0: raise ValueError("theta must be positive") diff --git a/python/fastdist/distributions/geometric.py b/python/fastdist/distributions/geometric.py index 7742506..15571a7 100644 --- a/python/fastdist/distributions/geometric.py +++ b/python/fastdist/distributions/geometric.py @@ -9,6 +9,7 @@ "extension has been built." ) from exc +import math import numpy as np from typing import Sequence, Union from numpy.typing import NDArray @@ -38,6 +39,8 @@ def _validate_params(p: Union[int, float]) -> None: """Internal validation shared by all methods.""" if not isinstance(p, (int, float)): raise TypeError("p must be a real number") + if not math.isfinite(p): + raise ValueError("p must be finite") if not (0 < p <= 1): raise ValueError("p must be in the interval (0, 1]") diff --git a/python/fastdist/distributions/negative_binomial.py b/python/fastdist/distributions/negative_binomial.py index fcb10a8..cb398ce 100644 --- a/python/fastdist/distributions/negative_binomial.py +++ b/python/fastdist/distributions/negative_binomial.py @@ -9,6 +9,7 @@ "extension has been built." ) from exc +import math import numpy as np from typing import Sequence, Union from numpy.typing import NDArray @@ -54,6 +55,8 @@ def _validate_params(r: Union[int, None] = None, p: Union[int, float, None] = No if p is not None: if not isinstance(p, (int, float)): raise TypeError("p must be a real number") + if not math.isfinite(p): + raise ValueError("p must be finite") if not 0 <= p <= 1: raise ValueError("p must be in [0, 1]") diff --git a/python/fastdist/distributions/normal.py b/python/fastdist/distributions/normal.py index 08291f3..9f66ef8 100644 --- a/python/fastdist/distributions/normal.py +++ b/python/fastdist/distributions/normal.py @@ -213,6 +213,11 @@ def _validate_inputs(_input: Union[SupportsFloat, ArrayLike], input_name: str, # Declared up front: without it the type is inferred from the scalar # branch alone and the array branch looks like a bad assignment. validated: Union[float, np.ndarray] + # numpy reads "0.5" as data, so a string would reach the array branch + # and come back as a one-element result instead of a TypeError. + if isinstance(_input, (str, bytes)): + raise TypeError(f"{input_name} must be a real number or a sequence of them") + if isinstance(_input, Real): validated = cast(float, _input) else: diff --git a/python/fastdist/distributions/poisson.py b/python/fastdist/distributions/poisson.py index 4fe3963..0d222cc 100644 --- a/python/fastdist/distributions/poisson.py +++ b/python/fastdist/distributions/poisson.py @@ -13,6 +13,7 @@ from numbers import Real from typing import SupportsFloat, Union, cast +import math import numpy as np from numpy.typing import ArrayLike, NDArray @@ -45,6 +46,8 @@ def _validate_params(lambda_: SupportsFloat) -> None: """Internal validation shared by all methods.""" if not isinstance(lambda_, Real): raise TypeError("lambda_ must be a real number") + if not math.isfinite(lambda_): + raise ValueError("lambda_ must be finite") if lambda_ <= 0: raise ValueError("lambda_ must be positive") @@ -54,6 +57,11 @@ def _validate_inputs(_input: Union[SupportsFloat, ArrayLike], input_name: str, s if _input is None: raise TypeError(f"{input_name} cannot be None") + # numpy reads "0.5" as data, so a string would reach the array branch + # and come back as a one-element result instead of a TypeError. + if isinstance(_input, (str, bytes)): + raise TypeError(f"{input_name} must be a real number or a sequence of them") + # Declared up front: without it the type is inferred from the scalar # branch alone and the array branch looks like a bad assignment. validated: Union[float, np.ndarray] diff --git a/python/fastdist/distributions/uniform.py b/python/fastdist/distributions/uniform.py index d168dae..c953bfd 100644 --- a/python/fastdist/distributions/uniform.py +++ b/python/fastdist/distributions/uniform.py @@ -13,6 +13,7 @@ from numbers import Real from typing import Sequence, SupportsFloat, Union, cast +import math import numpy as np from numpy.typing import ArrayLike, NDArray @@ -190,8 +191,12 @@ def _validate_params(a: Union[SupportsFloat, None] = None, b: Union[SupportsFloa if a is not None and not isinstance(a, Real): raise TypeError("a must be a real number") + if a is not None and not math.isfinite(a): + raise ValueError("a must be finite") if b is not None and not isinstance(b, Real): raise TypeError("b must be a real number") + if b is not None and not math.isfinite(b): + raise ValueError("b must be finite") if a is not None and b is not None and a >= b: raise ValueError("a must be less than b") @@ -233,6 +238,11 @@ def _validate_inputs(_input: Union[SupportsFloat, ArrayLike], input_name: str, s # Declared up front: without it the type is inferred from the scalar # branch alone and the array branch looks like a bad assignment. validated: Union[float, np.ndarray] + # numpy reads "0.5" as data, so a string would reach the array branch + # and come back as a one-element result instead of a TypeError. + if isinstance(_input, (str, bytes)): + raise TypeError(f"{input_name} must be a real number or a sequence of them") + if isinstance(_input, Real): validated = cast(float, _input) else: diff --git a/python/fastdist/distributions/utils.py b/python/fastdist/distributions/utils.py index 65e240f..29cb311 100644 --- a/python/fastdist/distributions/utils.py +++ b/python/fastdist/distributions/utils.py @@ -26,6 +26,11 @@ def _validate_input(_input: Union[SupportsFloat, ArrayLike], input_name: str, in Union[float, np.ndarray]: if _input is None: raise TypeError(f"{input_name} must not be None") + + # numpy reads "0.5" as data, so a string would reach the array branch + # and come back as a one-element result instead of a TypeError. + if isinstance(_input, (str, bytes)): + raise TypeError(f"{input_name} must be a real number or a sequence of them") if isinstance(_input, Sequence) and not isinstance(_input, (str, bytes)): dims = 1 if dims is None else dims if input_name in (None, ""): diff --git a/src/math/chi_square.cpp b/src/math/chi_square.cpp index 4a4d836..c9265bf 100644 --- a/src/math/chi_square.cpp +++ b/src/math/chi_square.cpp @@ -12,7 +12,9 @@ namespace fastdist::math { // f(x) = x^{k/2-1} * exp(-x/2) / (2^{k/2} Γ(k/2)) // ------------------------- double chi_square_pdf_scalar(const double x, const double k) { - if (!std::isfinite(k) || k <= 0.0) return std::numeric_limits::quiet_NaN(); // invalid params + if (!std::isfinite(x) || !std::isfinite(k) || k <= 0.0) { + return std::numeric_limits::quiet_NaN(); // invalid params or non-finite x + } if (x < 0.0) return 0.0; return gamma_pdf_scalar(x, k / 2.0, 2.0); diff --git a/src/math/gamma.cpp b/src/math/gamma.cpp index 477b29f..3a75300 100644 --- a/src/math/gamma.cpp +++ b/src/math/gamma.cpp @@ -13,8 +13,8 @@ namespace fastdist::math { // f(x) = x^{α-1} e^{-x/θ} / (Γ(α) θ^α) // ------------------------- double gamma_pdf_scalar(const double x, const double alpha, const double theta) { - if (!std::isfinite(alpha) || !std::isfinite(theta) || alpha <= 0.0 || theta <= 0.0) { - return std::numeric_limits::quiet_NaN(); // invalid params + if (!std::isfinite(x) || !std::isfinite(alpha) || !std::isfinite(theta) || alpha <= 0.0 || theta <= 0.0) { + return std::numeric_limits::quiet_NaN(); // invalid params or non-finite x } if (x < 0.0) return 0.0; diff --git a/tests/python/test_validation.py b/tests/python/test_validation.py new file mode 100644 index 0000000..243390e --- /dev/null +++ b/tests/python/test_validation.py @@ -0,0 +1,117 @@ +"""The validation contract, enforced uniformly across every distribution class. + +The library has one rule at each layer, and these tests hold every class to it: + + * A *parameter* that cannot describe a distribution is refused at + construction, with ValueError for a bad value and TypeError for a bad type. + Nothing is constructed, so no later call can return nonsense. + * An *input* x that is not finite is not an error. It yields nan, matching + what the C++ core returns and what numpy does elementwise, so a single bad + value in an array does not abort the whole call. + +NaN is the case worth testing explicitly: every comparison against it is False, +so a range check like `p < 0 or p > 1` passes it through. Each class needs an +explicit finiteness check, and these tests fail if one loses it. +""" + +import math + +import pytest + +from fastdist import (Bernoulli, Beta, Binomial, ChiSquare, DiscreteUniform, Exponential, + Gamma, Geometric, NegativeBinomial, Normal, Poisson, Uniform) + +NAN = float("nan") +INF = float("inf") + +# (class, valid args, index of a real-valued parameter, out-of-range value) +REAL_PARAM_CASES = [ + (Bernoulli, (0.3,), 0, 1.5), + (Beta, (2.0, 5.0), 0, 0.0), + (Beta, (2.0, 5.0), 1, -1.0), + (Binomial, (10, 0.3), 1, 1.5), + (ChiSquare, (6.0,), 0, 0.0), + (Exponential, (2.0,), 0, 0.0), + (Gamma, (3.0, 2.0), 0, 0.0), + (Gamma, (3.0, 2.0), 1, -1.0), + (Geometric, (0.25,), 0, 0.0), + (NegativeBinomial, (4, 0.3), 1, 1.5), + (Normal, (0.0, 1.0), 1, 0.0), + (Poisson, (4.0,), 0, 0.0), + (Uniform, (0.0, 1.0), 0, 2.0), +] + + +def _with(args, index, value): + out = list(args) + out[index] = value + return tuple(out) + + +@pytest.mark.parametrize("cls, args, index, bad", REAL_PARAM_CASES) +def test_non_finite_parameters_are_refused(cls, args, index, bad): + """nan and inf must raise, not build a distribution that returns nan forever.""" + for value in (NAN, INF, -INF): + with pytest.raises(ValueError): + cls(*_with(args, index, value)) + + +@pytest.mark.parametrize("cls, args, index, bad", REAL_PARAM_CASES) +def test_out_of_range_parameters_are_refused(cls, args, index, bad): + with pytest.raises(ValueError): + cls(*_with(args, index, bad)) + + +@pytest.mark.parametrize("cls, args, index, bad", REAL_PARAM_CASES) +def test_non_numeric_parameters_are_refused(cls, args, index, bad): + for value in ("0.5", None, [0.5]): + with pytest.raises(TypeError): + cls(*_with(args, index, value)) + + +@pytest.mark.parametrize("cls, args", [ + (Binomial, (10, 0.3)), + (NegativeBinomial, (4, 0.3)), + (DiscreteUniform, (1, 6)), +]) +def test_integer_parameters_reject_floats(cls, args): + """These take counts; a float is a mistake, not something to round.""" + with pytest.raises(TypeError): + cls(*(( 2.5,) + args[1:])) + + +def test_bounds_must_be_ordered(): + with pytest.raises(ValueError): + Uniform(5.0, 1.0) + with pytest.raises(ValueError): + DiscreteUniform(6, 1) + + +# (a valid instance, a method taking x, a point inside the support) +INPUT_CASES = [ + (Normal(0.0, 1.0), "pdf", 0.5), + (Normal(0.0, 1.0), "cdf", 0.5), + (Exponential(2.0), "pdf", 0.5), + (Exponential(2.0), "cdf", 0.5), + (Poisson(4.0), "pmf", 3.0), + (Beta(2.0, 5.0), "pdf_scalar", 0.5), + (Gamma(3.0, 2.0), "pmf_scalar", 1.0), + (ChiSquare(6.0), "pdf", 4.0), +] + + +@pytest.mark.parametrize("dist, method, good", INPUT_CASES) +def test_non_finite_input_yields_nan(dist, method, good): + """A non-finite x is answered with nan rather than an exception.""" + fn = getattr(dist, method) + assert math.isfinite(fn(good)) + for value in (NAN, INF, -INF): + assert math.isnan(fn(value)), f"{type(dist).__name__}.{method}({value}) should be nan" + + +@pytest.mark.parametrize("dist, method, good", INPUT_CASES) +def test_non_numeric_input_raises(dist, method, good): + fn = getattr(dist, method) + for value in ("0.5", None): + with pytest.raises(TypeError): + fn(value) From 488f3063d701a1ee55cc57a98844e2e7b712203e Mon Sep 17 00:00:00 2001 From: ghosteau Date: Fri, 11 Sep 2026 21:38:29 -0400 Subject: [PATCH 18/20] Measure against numpy and scipy.special too, so the SciPy claim is honest Every benchmark so far compared against scipy.stats. That is what a user would replace, but it is a soft target: scipy.stats.norm.pdf carries argument validation, broadcasting and masking that a direct expression does not, so most of the recorded margin is that machinery rather than faster arithmetic. Quoting "3x to 16x faster than SciPy" without saying so would be a misleading claim. The suite now also runs a primitives group against the numpy or scipy.special expression a user would write by hand, and BENCHMARKS.md reports both for every case. Against primitives the library is roughly at parity: behind on most cases at n = 100,000 and ahead on six of ten at n = 1,000,000, where numpy's intermediate arrays stop fitting in cache and the single-pass loop pulls ahead. Two informative extremes. poisson_cdf is 3.6x faster than scipy.special.pdtr, because summing by recurrence beats a general implementation. bernoulli_pmf is 0.16x at 100,000: the primitive is one np.where, and no C++ loop beats a single vectorised select. Adds examples/release_benchmarks.ipynb, executed so it ships with real output. It reads the recorded JSON rather than re-timing, shows both baselines side by side with a chart, and states which of the two to quote. scipy, matplotlib, jupyter and nbformat are added to requirements-dev.txt; the benchmarks needed scipy already and it was never declared. Co-Authored-By: Claude Opus 5 --- BENCHMARKS.md | 104 +- ...0.1.0_20260912T013406+0000_da5f274059.json | 1026 +++++++++++++++++ benchmarks/run.py | 65 +- examples/release_benchmarks.ipynb | 328 ++++++ requirements-dev.txt | 4 + 5 files changed, 1524 insertions(+), 3 deletions(-) create mode 100644 benchmarks/results/0.1.0_20260912T013406+0000_da5f274059.json create mode 100644 examples/release_benchmarks.ipynb diff --git a/BENCHMARKS.md b/BENCHMARKS.md index e0b753f..0bec272 100644 --- a/BENCHMARKS.md +++ b/BENCHMARKS.md @@ -42,8 +42,12 @@ python benchmarks/table.py --latest ## Method, and what the numbers do not say -The baseline is SciPy, because that is the realistic alternative for someone -who would otherwise use this library. +There are two baselines. `scipy.stats` is what a user would otherwise call, +and it is the comparison most people mean. The `primitives` group is the same +quantity written directly with numpy or `scipy.special`, which is a much harder +target: most of the margin over `scipy.stats` is that library's generic +distribution machinery rather than faster arithmetic. Both are reported for +every case, because quoting only the first would oversell the library. Each case is timed as several independent rounds, and the **minimum** round is reported. Noise on a shared machine can only ever add time, so the minimum is @@ -599,6 +603,102 @@ card on first use. --- +## Unreleased -- a second baseline, so the SciPy comparison is not oversold + +Every entry above compares against `scipy.stats`. That is what a user would +replace, but it is a soft target: `scipy.stats.norm.pdf` carries argument +validation, broadcasting and masking that a direct expression does not, and +most of the margin recorded above is that machinery rather than faster +arithmetic. + +The suite now also measures a `primitives` group: the same quantity written +directly with numpy or `scipy.special`, which is what a competent user would +write if they cared about speed. It is the harder baseline and the honest +ceiling. + +| case | vs `scipy.stats` (100k) | vs primitives (100k) | vs `scipy.stats` (1M) | vs primitives (1M) | +|---|---:|---:|---:|---:| +| `bernoulli_pmf` | 12.02x | 0.16x | 10.88x | 0.31x | +| `exponential_cdf` | 4.67x | 1.00x | 4.50x | 1.46x | +| `exponential_pdf` | 4.34x | 0.82x | 4.50x | 1.31x | +| `normal_cdf` | 2.42x | 0.94x | 2.29x | 0.95x | +| `normal_logpdf` | 18.28x | 0.56x | 15.72x | 2.25x | +| `normal_pdf` | 3.63x | 0.68x | 4.15x | 1.25x | +| `poisson_cdf` | 4.63x | 3.72x | 4.44x | 3.58x | +| `poisson_pmf` | 0.78x | 0.58x | 0.90x | 0.65x | +| `uniform_cdf` | 17.07x | 0.51x | 12.86x | 1.98x | +| `uniform_pdf` | 16.60x | 0.64x | 12.84x | 0.99x | + +### Reading + +Against `scipy.stats` the library is 2.3x to 16x faster. Against the +primitives it is roughly at parity: behind on most cases at n = 100,000, ahead +on six of ten at n = 1,000,000. + +The size dependence is the interesting part. A numpy expression materialises a +temporary array per operation; fastdist makes one pass and writes one output. +At 100,000 elements those temporaries still sit in cache and numpy wins. At +1,000,000 they do not, and the single pass pulls ahead. + +Two cases stand out at each end. `poisson_cdf` is 3.58x faster than +`scipy.special.pdtr`, because summing the terms by recurrence beats a general +implementation. `bernoulli_pmf` is 0.16x at 100,000: the primitive is a single +`np.where`, and nothing in a C++ loop can beat one vectorised select. + +What this means for how the numbers get quoted: + +- "3x to 16x faster than SciPy" is true only of `scipy.stats`, and needs the + reason attached, or it is a misleading claim. +- "Faster than hand-written numpy" is true only at a million elements and only + for some functions. It is not a general claim. +- The scalar group remains the library's strongest honest result: calling into + fastdist from a Python loop costs far less than calling `scipy.stats`. + + +- **Version** 0.1.0 (`da5f274059` on `chore/release-prep`, working tree dirty) +- **Measured** 2026-09-12T01:34:06+00:00 +- **CPU** AMD Ryzen 7 7700 8-Core Processor +- **Platform** Windows-11-10.0.26200-SP0 +- **Toolchain** Python 3.14.2, numpy 2.5.2, scipy 1.18.1 +- **CUDA** available + +### primitives (vs the numpy / scipy.special expression) + +| case | n | fastdist | baseline | speedup | max abs diff | +|---|---:|---:|---:|---:|---:| +| `normal_pdf` | 1,000 | 5.24 us | 4.59 us (numpy/scipy.special) | **0.88x** | 1.1e-16 | +| `normal_cdf` | 1,000 | 6.80 us | 4.32 us (numpy/scipy.special) | **0.63x** | 2.2e-16 | +| `normal_logpdf` | 1,000 | 1.88 us | 1.83 us (numpy/scipy.special) | **0.97x** | 8.9e-16 | +| `exponential_pdf` | 1,000 | 4.16 us | 4.03 us (numpy/scipy.special) | **0.97x** | 0.0e+00 | +| `exponential_cdf` | 1,000 | 4.49 us | 4.67 us (numpy/scipy.special) | **1.04x** | 0.0e+00 | +| `uniform_pdf` | 1,000 | 2.07 us | 2.91 us (numpy/scipy.special) | **1.41x** | 0.0e+00 | +| `uniform_cdf` | 1,000 | 2.12 us | 3.35 us (numpy/scipy.special) | **1.58x** | 0.0e+00 | +| `poisson_pmf` | 1,000 | 40.82 us | 15.11 us (numpy/scipy.special) | **0.37x** | 2.0e-19 | +| `poisson_cdf` | 1,000 | 10.05 us | 36.52 us (numpy/scipy.special) | **3.63x** | 2.2e-16 | +| `bernoulli_pmf` | 1,000 | 1.96 us | 2.04 us (numpy/scipy.special) | **1.04x** | 0.0e+00 | +| `normal_pdf` | 100,000 | 415.90 us | 281.70 us (numpy/scipy.special) | **0.68x** | 1.1e-16 | +| `normal_cdf` | 100,000 | 894.80 us | 836.80 us (numpy/scipy.special) | **0.94x** | 2.2e-16 | +| `normal_logpdf` | 100,000 | 77.60 us | 43.60 us (numpy/scipy.special) | **0.56x** | 8.9e-16 | +| `exponential_pdf` | 100,000 | 311.50 us | 255.00 us (numpy/scipy.special) | **0.82x** | 0.0e+00 | +| `exponential_cdf` | 100,000 | 366.40 us | 367.00 us (numpy/scipy.special) | **1.00x** | 0.0e+00 | +| `uniform_pdf` | 100,000 | 94.70 us | 60.70 us (numpy/scipy.special) | **0.64x** | 0.0e+00 | +| `uniform_cdf` | 100,000 | 100.50 us | 50.90 us (numpy/scipy.special) | **0.51x** | 0.0e+00 | +| `poisson_pmf` | 100,000 | 4.02 ms | 2.34 ms (numpy/scipy.special) | **0.58x** | 2.0e-19 | +| `poisson_cdf` | 100,000 | 1.30 ms | 4.84 ms (numpy/scipy.special) | **3.72x** | 2.2e-16 | +| `bernoulli_pmf` | 100,000 | 293.00 us | 45.90 us (numpy/scipy.special) | **0.16x** | 0.0e+00 | +| `normal_pdf` | 1,000,000 | 4.77 ms | 5.96 ms (numpy/scipy.special) | **1.25x** | 1.1e-16 | +| `normal_cdf` | 1,000,000 | 9.61 ms | 9.15 ms (numpy/scipy.special) | **0.95x** | 2.2e-16 | +| `normal_logpdf` | 1,000,000 | 1.28 ms | 2.89 ms (numpy/scipy.special) | **2.25x** | 8.9e-16 | +| `exponential_pdf` | 1,000,000 | 3.73 ms | 4.88 ms (numpy/scipy.special) | **1.31x** | 0.0e+00 | +| `exponential_cdf` | 1,000,000 | 4.26 ms | 6.22 ms (numpy/scipy.special) | **1.46x** | 0.0e+00 | +| `uniform_pdf` | 1,000,000 | 1.41 ms | 1.40 ms (numpy/scipy.special) | **0.99x** | 0.0e+00 | +| `uniform_cdf` | 1,000,000 | 1.47 ms | 2.91 ms (numpy/scipy.special) | **1.98x** | 0.0e+00 | +| `poisson_pmf` | 1,000,000 | 41.82 ms | 27.22 ms (numpy/scipy.special) | **0.65x** | 2.0e-19 | +| `poisson_cdf` | 1,000,000 | 14.09 ms | 50.43 ms (numpy/scipy.special) | **3.58x** | 2.2e-16 | +| `bernoulli_pmf` | 1,000,000 | 3.51 ms | 1.07 ms (numpy/scipy.special) | **0.31x** | 0.0e+00 | + +--- + ## Changes to record here Add an entry when a release ships, or when a change is made specifically to diff --git a/benchmarks/results/0.1.0_20260912T013406+0000_da5f274059.json b/benchmarks/results/0.1.0_20260912T013406+0000_da5f274059.json new file mode 100644 index 0000000..fdc0896 --- /dev/null +++ b/benchmarks/results/0.1.0_20260912T013406+0000_da5f274059.json @@ -0,0 +1,1026 @@ +{ + "environment": { + "timestamp_utc": "2026-09-12T01:34:06+00:00", + "fastdist_version": "0.1.0", + "git_commit": "da5f274059956eb4a82cd491daa3a6383b89ac79", + "git_branch": "chore/release-prep", + "git_dirty": true, + "cuda_available": true, + "python": "3.14.2", + "numpy": "2.5.2", + "scipy": "1.18.1", + "platform": "Windows-11-10.0.26200-SP0", + "processor": "AMD Ryzen 7 7700 8-Core Processor", + "machine": "AMD64" + }, + "results": [ + { + "group": "batch", + "case": "normal_pdf", + "n": 1000, + "fastdist_s": 5.231999675743282e-06, + "baseline_s": 3.18539998261258e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.146797794116407, + "baseline_noise_pct": 1.161549284231593, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 6.0883030963874205 + }, + { + "group": "batch", + "case": "normal_cdf", + "n": 1000, + "fastdist_s": 6.787999882362783e-06, + "baseline_s": 3.052000014577061e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.6482076576346734, + "baseline_noise_pct": 0.360418325411763, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.4961698106493095 + }, + { + "group": "batch", + "case": "normal_logpdf", + "n": 1000, + "fastdist_s": 1.8699996871873736e-06, + "baseline_s": 3.152000019326806e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.6417473506949962, + "baseline_noise_pct": 7.442892781858588, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 16.855617896212923 + }, + { + "group": "batch", + "case": "exponential_pdf", + "n": 1000, + "fastdist_s": 4.1440001223236325e-06, + "baseline_s": 2.991599962115288e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.2895774859965757, + "baseline_noise_pct": 8.002409403381963, + "max_abs_diff": 0.0, + "speedup": 7.219111664595781 + }, + { + "group": "batch", + "case": "exponential_cdf", + "n": 1000, + "fastdist_s": 4.472000291571021e-06, + "baseline_s": 3.0565999913960696e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.0125115224347994, + "baseline_noise_pct": 0.9291383140931491, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 6.834972701493901 + }, + { + "group": "batch", + "case": "uniform_pdf", + "n": 1000, + "fastdist_s": 2.041999832727015e-06, + "baseline_s": 3.214399970602244e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.4897197337954095, + "baseline_noise_pct": 3.845196956172472, + "max_abs_diff": 0.0, + "speedup": 15.741431116130563 + }, + { + "group": "batch", + "case": "uniform_cdf", + "n": 1000, + "fastdist_s": 2.115999814122915e-06, + "baseline_s": 3.2145999721251425e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.6616308049493134, + "baseline_noise_pct": 0.6159408487320197, + "max_abs_diff": 0.0, + "speedup": 15.191872658351809 + }, + { + "group": "batch", + "case": "poisson_pmf", + "n": 1000, + "fastdist_s": 4.068400012329221e-05, + "baseline_s": 3.88459995156154e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.24087990799858805, + "baseline_noise_pct": 0.6899044889980731, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.9548225198577628 + }, + { + "group": "batch", + "case": "poisson_cdf", + "n": 1000, + "fastdist_s": 9.938000002875925e-06, + "baseline_s": 7.077799993567169e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.20124926830428622, + "baseline_noise_pct": 1.9101978279195033, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 7.121956119459598 + }, + { + "group": "batch", + "case": "bernoulli_pmf", + "n": 1000, + "fastdist_s": 1.9880000036209823e-06, + "baseline_s": 4.923600004985929e-05, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.0060145964200429, + "baseline_noise_pct": 1.653261423614009, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 24.766599577555265 + }, + { + "group": "batch", + "case": "normal_pdf", + "n": 100000, + "fastdist_s": 0.00041379997855983675, + "baseline_s": 0.0015007999900262803, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.26583797408633625, + "baseline_noise_pct": 8.588754127026158, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 3.6268730492678363 + }, + { + "group": "batch", + "case": "normal_cdf", + "n": 100000, + "fastdist_s": 0.0008910000033210963, + "baseline_s": 0.002153899986296892, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.6958461255991314, + "baseline_noise_pct": 2.0753062124564776, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 2.4173961596728244 + }, + { + "group": "batch", + "case": "normal_logpdf", + "n": 100000, + "fastdist_s": 9.68000094871968e-05, + "baseline_s": 0.0017694999987725168, + "baseline_name": "scipy", + "fastdist_noise_pct": 3.2024432783537575, + "baseline_noise_pct": 4.922295733174503, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 18.279956873419096 + }, + { + "group": "batch", + "case": "exponential_pdf", + "n": 100000, + "fastdist_s": 0.0003090999962296337, + "baseline_s": 0.0013429999817162752, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.9382054395257516, + "baseline_noise_pct": 11.794493234475372, + "max_abs_diff": 0.0, + "speedup": 4.3448722034876575 + }, + { + "group": "batch", + "case": "exponential_cdf", + "n": 100000, + "fastdist_s": 0.00036050000926479697, + "baseline_s": 0.0016819000011309981, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.3855775467304166, + "baseline_noise_pct": 7.438017787408416, + "max_abs_diff": 1.6653345369377348e-16, + "speedup": 4.665464515690476 + }, + { + "group": "batch", + "case": "uniform_pdf", + "n": 100000, + "fastdist_s": 9.360001422464848e-05, + "baseline_s": 0.0015536000137217343, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.3205152123459923, + "baseline_noise_pct": 11.637485718310666, + "max_abs_diff": 0.0, + "speedup": 16.59828822240298 + }, + { + "group": "batch", + "case": "uniform_cdf", + "n": 100000, + "fastdist_s": 9.940000018104911e-05, + "baseline_s": 0.0016969999996945262, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.20120877518550032, + "baseline_noise_pct": 8.326458403749054, + "max_abs_diff": 0.0, + "speedup": 17.07243457347663 + }, + { + "group": "batch", + "case": "poisson_pmf", + "n": 100000, + "fastdist_s": 0.004024199995910749, + "baseline_s": 0.00312619999749586, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.860195900557472, + "baseline_noise_pct": 1.458639818676303, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.7768500573213545 + }, + { + "group": "batch", + "case": "poisson_cdf", + "n": 100000, + "fastdist_s": 0.0013086000108160079, + "baseline_s": 0.006060499988961965, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.3362353094713952, + "baseline_noise_pct": 1.240822040658116, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.6312852964006925 + }, + { + "group": "batch", + "case": "bernoulli_pmf", + "n": 100000, + "fastdist_s": 0.00029090000316500664, + "baseline_s": 0.0034954000147990882, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.4094097515139192, + "baseline_noise_pct": 1.0613947274224043, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 12.015812914296873 + }, + { + "group": "batch", + "case": "normal_pdf", + "n": 1000000, + "fastdist_s": 0.0048633999831508845, + "baseline_s": 0.020179100014502183, + "baseline_name": "scipy", + "fastdist_noise_pct": 11.6297246875412, + "baseline_noise_pct": 3.602241802762313, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 4.149175491304873 + }, + { + "group": "batch", + "case": "normal_cdf", + "n": 1000000, + "fastdist_s": 0.009670100000221282, + "baseline_s": 0.022181900014402345, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.8097125586749878, + "baseline_noise_pct": 4.262484192695984, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 2.2938645943573235 + }, + { + "group": "batch", + "case": "normal_logpdf", + "n": 1000000, + "fastdist_s": 0.0013467000098899007, + "baseline_s": 0.021172500011743978, + "baseline_name": "scipy", + "fastdist_noise_pct": 4.232568281730782, + "baseline_noise_pct": 5.404652311180709, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 15.721764206027542 + }, + { + "group": "batch", + "case": "exponential_pdf", + "n": 1000000, + "fastdist_s": 0.003785800014156848, + "baseline_s": 0.01702499997918494, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.755031404046175, + "baseline_noise_pct": 1.8414098685268432, + "max_abs_diff": 0.0, + "speedup": 4.497067968598614 + }, + { + "group": "batch", + "case": "exponential_cdf", + "n": 1000000, + "fastdist_s": 0.004264599992893636, + "baseline_s": 0.01920589999645017, + "baseline_name": "scipy", + "fastdist_noise_pct": 3.1585613483714727, + "baseline_noise_pct": 2.1191403946920926, + "max_abs_diff": 1.6653345369377348e-16, + "speedup": 4.503564233094344 + }, + { + "group": "batch", + "case": "uniform_pdf", + "n": 1000000, + "fastdist_s": 0.0014275999856181443, + "baseline_s": 0.018332999985432252, + "baseline_name": "scipy", + "fastdist_noise_pct": 3.551415537670032, + "baseline_noise_pct": 1.93694429461917, + "max_abs_diff": 0.0, + "speedup": 12.8418325652295 + }, + { + "group": "batch", + "case": "uniform_cdf", + "n": 1000000, + "fastdist_s": 0.001480600010836497, + "baseline_s": 0.01904469999135472, + "baseline_name": "scipy", + "fastdist_noise_pct": 3.208157705726545, + "baseline_noise_pct": 3.9270769293932655, + "max_abs_diff": 0.0, + "speedup": 12.862825781417497 + }, + { + "group": "batch", + "case": "poisson_pmf", + "n": 1000000, + "fastdist_s": 0.0419201000186149, + "baseline_s": 0.03769800000009127, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.0655985936285095, + "baseline_noise_pct": 2.480502917587205, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.8992822055136133 + }, + { + "group": "batch", + "case": "poisson_cdf", + "n": 1000000, + "fastdist_s": 0.01404270000057295, + "baseline_s": 0.062395800021477044, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.5618578955877331, + "baseline_noise_pct": 1.236781957864403, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.443290821489547 + }, + { + "group": "batch", + "case": "bernoulli_pmf", + "n": 1000000, + "fastdist_s": 0.0035071000165771693, + "baseline_s": 0.03816879997611977, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.1841399369189354, + "baseline_noise_pct": 3.6370020428244936, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 10.883293831286695 + }, + { + "group": "primitives", + "case": "normal_pdf", + "n": 1000, + "fastdist_s": 5.236000288277865e-06, + "baseline_s": 4.594000056385994e-06, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 1.0695156847181468, + "baseline_noise_pct": 0.6530306501215267, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 0.8773872810264826 + }, + { + "group": "primitives", + "case": "normal_cdf", + "n": 1000, + "fastdist_s": 6.804000004194677e-06, + "baseline_s": 4.31800028309226e-06, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 0.6760694909130981, + "baseline_noise_pct": 0.13895426803130023, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.6346267313977375 + }, + { + "group": "primitives", + "case": "normal_logpdf", + "n": 1000, + "fastdist_s": 1.8820003606379032e-06, + "baseline_s": 1.8320005619898438e-06, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 21.36022803047829, + "baseline_noise_pct": 37.1178574503718, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 0.973432630676483 + }, + { + "group": "primitives", + "case": "exponential_pdf", + "n": 1000, + "fastdist_s": 4.1639996925368906e-06, + "baseline_s": 4.03400044888258e-06, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 2.641697013446325, + "baseline_noise_pct": 1.8343924413738797, + "max_abs_diff": 0.0, + "speedup": 0.9687801985462902 + }, + { + "group": "primitives", + "case": "exponential_cdf", + "n": 1000, + "fastdist_s": 4.486000398173928e-06, + "baseline_s": 4.674000083468855e-06, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 0.49040533859963187, + "baseline_noise_pct": 0.684631172779737, + "max_abs_diff": 0.0, + "speedup": 1.0419080848435622 + }, + { + "group": "primitives", + "case": "uniform_pdf", + "n": 1000, + "fastdist_s": 2.0700000459328295e-06, + "baseline_s": 2.9120000544935464e-06, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 0.4830954552190355, + "baseline_noise_pct": 2.1291170625600446, + "max_abs_diff": 0.0, + "speedup": 1.406763280133782 + }, + { + "group": "primitives", + "case": "uniform_cdf", + "n": 1000, + "fastdist_s": 2.1219998598098755e-06, + "baseline_s": 3.34800046402961e-06, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 0.3770057231136611, + "baseline_noise_pct": 0.35840823280140577, + "max_abs_diff": 0.0, + "speedup": 1.5777571560865138 + }, + { + "group": "primitives", + "case": "poisson_pmf", + "n": 1000, + "fastdist_s": 4.08179999794811e-05, + "baseline_s": 1.5109999803826213e-05, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 0.8378648977059339, + "baseline_noise_pct": 0.7412327743266206, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.37017981800729816 + }, + { + "group": "primitives", + "case": "poisson_cdf", + "n": 1000, + "fastdist_s": 1.0052000288851558e-05, + "baseline_s": 3.65219998639077e-05, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 0.49740564200842513, + "baseline_noise_pct": 14.210612664560301, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 3.6333066866714483 + }, + { + "group": "primitives", + "case": "bernoulli_pmf", + "n": 1000, + "fastdist_s": 1.9640004029497502e-06, + "baseline_s": 2.036000369116664e-06, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 0.4073349936146516, + "baseline_noise_pct": 0.49113419722330315, + "max_abs_diff": 0.0, + "speedup": 1.03665985305236 + }, + { + "group": "primitives", + "case": "normal_pdf", + "n": 100000, + "fastdist_s": 0.0004158999945502728, + "baseline_s": 0.000281700020423159, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 1.154125392795, + "baseline_noise_pct": 0.9584633509240853, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 0.6773263383371069 + }, + { + "group": "primitives", + "case": "normal_cdf", + "n": 100000, + "fastdist_s": 0.0008947999740485102, + "baseline_s": 0.000836799998069182, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 0.6034881728931508, + "baseline_noise_pct": 1.3145346052996447, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.9351810710086321 + }, + { + "group": "primitives", + "case": "normal_logpdf", + "n": 100000, + "fastdist_s": 7.760000880807638e-05, + "baseline_s": 4.3599982745945454e-05, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 1.1597650090986966, + "baseline_noise_pct": 0.45871927074850277, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 0.5618553839829937 + }, + { + "group": "primitives", + "case": "exponential_pdf", + "n": 100000, + "fastdist_s": 0.0003115000145044178, + "baseline_s": 0.00025499999173916876, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 1.7656447563262054, + "baseline_noise_pct": 0.8627402820880299, + "max_abs_diff": 0.0, + "speedup": 0.8186195180275097 + }, + { + "group": "primitives", + "case": "exponential_cdf", + "n": 100000, + "fastdist_s": 0.0003664000250864774, + "baseline_s": 0.00036700000055134296, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 0.6277260513182152, + "baseline_noise_pct": 0.5449632768343661, + "max_abs_diff": 0.0, + "speedup": 1.001637487510335 + }, + { + "group": "primitives", + "case": "uniform_pdf", + "n": 100000, + "fastdist_s": 9.469999349676073e-05, + "baseline_s": 6.0699996538460255e-05, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 0.42238972889795434, + "baseline_noise_pct": 0.658983638563968, + "max_abs_diff": 0.0, + "speedup": 0.6409714963764653 + }, + { + "group": "primitives", + "case": "uniform_cdf", + "n": 100000, + "fastdist_s": 0.00010050000855699182, + "baseline_s": 5.089998012408614e-05, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 0.2985097102532931, + "baseline_noise_pct": 3.9293617167264183, + "max_abs_diff": 0.0, + "speedup": 0.5064674207984932 + }, + { + "group": "primitives", + "case": "poisson_pmf", + "n": 100000, + "fastdist_s": 0.00401959998998791, + "baseline_s": 0.0023441999801434577, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 1.0299531097873005, + "baseline_noise_pct": 2.7557374660160447, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.5831923539611982 + }, + { + "group": "primitives", + "case": "poisson_cdf", + "n": 100000, + "fastdist_s": 0.001303300028666854, + "baseline_s": 0.0048448999878019094, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 0.9207377192244361, + "baseline_noise_pct": 0.7017688093439053, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 3.7174095612947693 + }, + { + "group": "primitives", + "case": "bernoulli_pmf", + "n": 100000, + "fastdist_s": 0.0002930000191554427, + "baseline_s": 4.5900000259280205e-05, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 1.6382177570305116, + "baseline_noise_pct": 0.43573316289521613, + "max_abs_diff": 0.0, + "speedup": 0.15665528074566193 + }, + { + "group": "primitives", + "case": "normal_pdf", + "n": 1000000, + "fastdist_s": 0.004770500003360212, + "baseline_s": 0.005964299984043464, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 3.8193059479822518, + "baseline_noise_pct": 4.568851488129559, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 1.2502463011932443 + }, + { + "group": "primitives", + "case": "normal_cdf", + "n": 1000000, + "fastdist_s": 0.009609100001398474, + "baseline_s": 0.009152700018603355, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 0.6535469858355746, + "baseline_noise_pct": 1.6137310324521112, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.9525033579909985 + }, + { + "group": "primitives", + "case": "normal_logpdf", + "n": 1000000, + "fastdist_s": 0.001284400001168251, + "baseline_s": 0.002887800015741959, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 7.707878935316442, + "baseline_noise_pct": 8.07188857725858, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 2.248365005539789 + }, + { + "group": "primitives", + "case": "exponential_pdf", + "n": 1000000, + "fastdist_s": 0.003734799975063652, + "baseline_s": 0.004879899992374703, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 5.676341838315083, + "baseline_noise_pct": 4.493944779252375, + "max_abs_diff": 0.0, + "speedup": 1.306602769882351 + }, + { + "group": "primitives", + "case": "exponential_cdf", + "n": 1000000, + "fastdist_s": 0.0042591999808792025, + "baseline_s": 0.0062188999727368355, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 4.510236861735697, + "baseline_noise_pct": 3.391275082579645, + "max_abs_diff": 0.0, + "speedup": 1.4601098799434873 + }, + { + "group": "primitives", + "case": "uniform_pdf", + "n": 1000000, + "fastdist_s": 0.0014079000102356076, + "baseline_s": 0.0013952000008430332, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 7.003337834549344, + "baseline_noise_pct": 7.031251385232397, + "max_abs_diff": 0.0, + "speedup": 0.9909794663681769 + }, + { + "group": "primitives", + "case": "uniform_cdf", + "n": 1000000, + "fastdist_s": 0.0014720000035595149, + "baseline_s": 0.0029085999995004386, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 8.002717016556856, + "baseline_noise_pct": 6.900227424331516, + "max_abs_diff": 0.0, + "speedup": 1.9759510818390023 + }, + { + "group": "primitives", + "case": "poisson_pmf", + "n": 1000000, + "fastdist_s": 0.041819000005489215, + "baseline_s": 0.027223700017202646, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 1.4208852754973678, + "baseline_noise_pct": 2.7038939876804045, + "max_abs_diff": 1.9651164376299768e-19, + "speedup": 0.6509887853279425 + }, + { + "group": "primitives", + "case": "poisson_cdf", + "n": 1000000, + "fastdist_s": 0.014085200004046783, + "baseline_s": 0.05043210001895204, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 1.001050725309743, + "baseline_noise_pct": 0.929368376422481, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 3.580502939572212 + }, + { + "group": "primitives", + "case": "bernoulli_pmf", + "n": 1000000, + "fastdist_s": 0.0035120000247843564, + "baseline_s": 0.001073099992936477, + "baseline_name": "numpy/scipy.special", + "fastdist_noise_pct": 2.0330293336610112, + "baseline_noise_pct": 7.967571644553568, + "max_abs_diff": 0.0, + "speedup": 0.30555238763199255 + }, + { + "group": "scalar", + "case": "normal_pdf", + "n": 20000, + "fastdist_s": 0.007401600014418364, + "baseline_s": 0.4237614999874495, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.1173258386461244, + "baseline_noise_pct": 2.078032106882392, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 57.252688494644325 + }, + { + "group": "scalar", + "case": "normal_cdf", + "n": 20000, + "fastdist_s": 0.00761550001334399, + "baseline_s": 0.4195663999998942, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.4129075585301705, + "baseline_noise_pct": 1.2277913632842907, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 55.09374292754564 + }, + { + "group": "scalar", + "case": "gamma_cdf", + "n": 20000, + "fastdist_s": 0.008807000005617738, + "baseline_s": 0.41784440001356415, + "baseline_name": "scipy", + "fastdist_noise_pct": 0.8436469130561416, + "baseline_noise_pct": 1.0247594578436494, + "max_abs_diff": 1.177946629127291e-13, + "speedup": 47.44457814772714 + }, + { + "group": "scalar", + "case": "chi_square_cdf", + "n": 20000, + "fastdist_s": 0.007931300002383068, + "baseline_s": 0.42480959999375045, + "baseline_name": "scipy", + "fastdist_noise_pct": 1.9126753498628986, + "baseline_noise_pct": 2.8283259170638972, + "max_abs_diff": 1.177946629127291e-13, + "speedup": 53.561156413968774 + }, + { + "group": "scalar", + "case": "beta_cdf", + "n": 20000, + "fastdist_s": 0.0098852000082843, + "baseline_s": 0.4836280000163242, + "baseline_name": "scipy", + "fastdist_noise_pct": 2.7141583010132293, + "baseline_noise_pct": 1.8432555558949844, + "max_abs_diff": 8.881784197001252e-16, + "speedup": 48.92445267784358 + }, + { + "group": "cuda", + "case": "normal_pdf", + "n": 1000, + "fastdist_s": 2.3900007363408804e-05, + "baseline_s": 6.700021913275123e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 42.67777076904717, + "baseline_noise_pct": 2.9850875935554777, + "max_abs_diff": 5.551115123125783e-17, + "speedup": 0.28033555853764863 + }, + { + "group": "cuda", + "case": "normal_cdf", + "n": 1000, + "fastdist_s": 2.2799998987466097e-05, + "baseline_s": 7.099995855242014e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 43.8596020944547, + "baseline_noise_pct": 12.676160259721096, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 0.31140334081352866 + }, + { + "group": "cuda", + "case": "normal_logpdf", + "n": 1000, + "fastdist_s": 2.850001328624785e-05, + "baseline_s": 2.3999891709536314e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 24.561272725230353, + "baseline_noise_pct": 50.00060633253701, + "max_abs_diff": 0.0, + "speedup": 0.08421010709183428 + }, + { + "group": "cuda", + "case": "exponential_pdf", + "n": 1000, + "fastdist_s": 2.5300018023699522e-05, + "baseline_s": 4.299974534660578e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 36.7588018893319, + "baseline_noise_pct": 2.326289713427098, + "max_abs_diff": 1.1102230246251565e-16, + "speedup": 0.16995934669424434 + }, + { + "group": "cuda", + "case": "uniform_pdf", + "n": 1000, + "fastdist_s": 2.4699984351173043e-05, + "baseline_s": 2.2999884095042944e-06, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 31.983951642835677, + "baseline_noise_pct": 4.347881103926507, + "max_abs_diff": 0.0, + "speedup": 0.09311699865085397 + }, + { + "group": "cuda", + "case": "normal_pdf", + "n": 100000, + "fastdist_s": 0.00019290001364424825, + "baseline_s": 0.0004199999966658652, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 4.61377907101304, + "baseline_noise_pct": 13.73809601353744, + "max_abs_diff": 5.551115123125783e-17, + "speedup": 2.177293763392061 + }, + { + "group": "cuda", + "case": "normal_cdf", + "n": 100000, + "fastdist_s": 0.0002103000006172806, + "baseline_s": 0.00089640001533553, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 5.6585845297434085, + "baseline_noise_pct": 2.8670227048185284, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.262482228741714 + }, + { + "group": "cuda", + "case": "normal_logpdf", + "n": 100000, + "fastdist_s": 0.00020140002015978098, + "baseline_s": 7.789998198859394e-05, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 1.9364305354930857, + "baseline_noise_pct": 0.6418909200137636, + "max_abs_diff": 0.0, + "speedup": 0.38679232468195324 + }, + { + "group": "cuda", + "case": "exponential_pdf", + "n": 100000, + "fastdist_s": 0.00018709999858401716, + "baseline_s": 0.00031350000062957406, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 5.3981928247779845, + "baseline_noise_pct": 5.550244351489606, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 1.6755745751050717 + }, + { + "group": "cuda", + "case": "uniform_pdf", + "n": 100000, + "fastdist_s": 0.00018610002007335424, + "baseline_s": 9.449999197386205e-05, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 5.427167867623094, + "baseline_noise_pct": 0.4232836822970162, + "max_abs_diff": 0.0, + "speedup": 0.5077914120407585 + }, + { + "group": "cuda", + "case": "normal_pdf", + "n": 1000000, + "fastdist_s": 0.001994499994907528, + "baseline_s": 0.004949299996951595, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 4.667835625534197, + "baseline_noise_pct": 2.9701172585114914, + "max_abs_diff": 5.551115123125783e-17, + "speedup": 2.481474058454967 + }, + { + "group": "cuda", + "case": "normal_cdf", + "n": 1000000, + "fastdist_s": 0.002145700011169538, + "baseline_s": 0.009828000009292737, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 5.904833092713909, + "baseline_noise_pct": 2.182539716030776, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 4.580323418060606 + }, + { + "group": "cuda", + "case": "normal_logpdf", + "n": 1000000, + "fastdist_s": 0.001963999995496124, + "baseline_s": 0.0013095999893266708, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 9.1853363639648, + "baseline_noise_pct": 4.780086401664021, + "max_abs_diff": 0.0, + "speedup": 0.6668024400864899 + }, + { + "group": "cuda", + "case": "exponential_pdf", + "n": 1000000, + "fastdist_s": 0.001884900004370138, + "baseline_s": 0.00374549999833107, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 6.074592735906896, + "baseline_noise_pct": 6.672006707362558, + "max_abs_diff": 2.220446049250313e-16, + "speedup": 1.9871080639010734 + }, + { + "group": "cuda", + "case": "uniform_pdf", + "n": 1000000, + "fastdist_s": 0.001867399987531826, + "baseline_s": 0.0014153999800328165, + "baseline_name": "fastdist-cpu", + "fastdist_noise_pct": 6.554569487655535, + "baseline_noise_pct": 24.98234061216291, + "max_abs_diff": 0.0, + "speedup": 0.7579522274194586 + }, + { + "group": "sample", + "case": "normal_sample", + "n": 100000, + "fastdist_s": 0.026889899978414178, + "baseline_s": 0.0008866999996826053, + "baseline_name": "numpy", + "fastdist_noise_pct": 3.3495848615913655, + "baseline_noise_pct": 0.28194315250906066, + "max_abs_diff": null, + "speedup": 0.03297520631889305 + }, + { + "group": "sample", + "case": "uniform_sample", + "n": 100000, + "fastdist_s": 0.024024999991524965, + "baseline_s": 0.0002314999874215573, + "baseline_name": "numpy", + "fastdist_noise_pct": 2.5315296725926038, + "baseline_noise_pct": 0.2159969495663906, + "max_abs_diff": null, + "speedup": 0.009635795525628345 + }, + { + "group": "sample", + "case": "normal_sample", + "n": 1000000, + "fastdist_s": 0.2852374000067357, + "baseline_s": 0.010226899990811944, + "baseline_name": "numpy", + "fastdist_noise_pct": 0.6886894911030368, + "baseline_noise_pct": 1.7864653360456118, + "max_abs_diff": null, + "speedup": 0.0358539938681619 + }, + { + "group": "sample", + "case": "uniform_sample", + "n": 1000000, + "fastdist_s": 0.25460909999674186, + "baseline_s": 0.0035176000092178583, + "baseline_name": "numpy", + "fastdist_noise_pct": 4.839614921837846, + "baseline_noise_pct": 4.602570218368746, + "max_abs_diff": null, + "speedup": 0.013815688477995766 + } + ] +} diff --git a/benchmarks/run.py b/benchmarks/run.py index 44f744e..e9a6f4d 100644 --- a/benchmarks/run.py +++ b/benchmarks/run.py @@ -21,6 +21,13 @@ compiled code, and neither pays per-element Python overhead. This is the honest headline comparison. + prims fastdist's *_cpu entry points against the numpy / scipy.special + expression a user could write by hand for the same quantity. This is + the harder baseline and the honest ceiling: most of the margin over + scipy.stats is that library's generic distribution machinery -- + argument validation, broadcasting, masking -- rather than faster + arithmetic, and this group shows what is left once that is removed. + scalar fastdist's *_scalar entry points against SciPy called on one value at a time. Both sides pay Python call overhead per element, so this measures the cost of a single call rather than throughput. It is @@ -67,10 +74,13 @@ from harness import Result, environment, measure, write_report # noqa: E402 try: - from scipy import stats as sps + from scipy import special as spec, stats as sps except ImportError: # pragma: no cover sys.exit("benchmarks require scipy: pip install scipy") +SQRT_2PI = np.sqrt(2.0 * np.pi) +LOG_SQRT_2PI = np.log(SQRT_2PI) + import fastdist._fastdist as core # noqa: E402 SIZES = (1_000, 100_000, 1_000_000) @@ -134,6 +144,53 @@ def batch_cases(sizes): ] +# --------------------------------------------------------------------------- +# Primitives: fastdist vs the hand-written numpy / scipy.special equivalent +# --------------------------------------------------------------------------- +def primitive_cases(sizes): + """The same quantities, expressed directly instead of through scipy.stats.""" + rng = np.random.default_rng(20260905) + + for n in sizes: + x_real = rng.normal(0.0, 1.0, n) + x_pos = np.abs(rng.normal(2.0, 1.0, n)) + 0.05 + k_count = rng.integers(0, 20, n).astype(float) + k_binary = rng.integers(0, 2, n).astype(np.int32) + + yield from [ + ("normal_pdf", n, + lambda x=x_real: core.normal_pdf_cpu(x, 0.0, 1.0, 0.0), + lambda x=x_real: np.exp(-0.5 * x * x) / SQRT_2PI), + ("normal_cdf", n, + lambda x=x_real: core.normal_cdf_cpu(x, 0.0, 1.0, 0.0), + lambda x=x_real: spec.ndtr(x)), + ("normal_logpdf", n, + lambda x=x_real: core.normal_logpdf_cpu(x, 0.0, 1.0, 0.0), + lambda x=x_real: -0.5 * x * x - LOG_SQRT_2PI), + ("exponential_pdf", n, + lambda x=x_pos: core.exponential_pdf_cpu(x, 2.0, 0.0), + lambda x=x_pos: 2.0 * np.exp(-2.0 * x)), + ("exponential_cdf", n, + lambda x=x_pos: core.exponential_cdf_cpu(x, 2.0, 0.0), + lambda x=x_pos: -np.expm1(-2.0 * x)), + ("uniform_pdf", n, + lambda x=x_real: core.uniform_pdf_cpu(x, -3.0, 3.0, 0.0), + lambda x=x_real: np.where((x >= -3.0) & (x <= 3.0), 1.0 / 6.0, 0.0)), + ("uniform_cdf", n, + lambda x=x_real: core.uniform_cdf_cpu(x, -3.0, 3.0, 0.0), + lambda x=x_real: np.clip((x + 3.0) / 6.0, 0.0, 1.0)), + ("poisson_pmf", n, + lambda x=k_count: core.poisson_pmf_cpu(x, 4.0, 0), + lambda x=k_count: np.exp(spec.xlogy(x, 4.0) - 4.0 - spec.gammaln(x + 1.0))), + ("poisson_cdf", n, + lambda x=k_count: core.poisson_cdf_cpu(x, 4.0, 0), + lambda x=k_count: spec.pdtr(x, 4.0)), + ("bernoulli_pmf", n, + lambda x=k_binary: core.bernoulli_pmf_cpu(x, 0.3, 0), + lambda x=k_binary: np.where(x == 1, 0.3, 0.7)), + ] + + # --------------------------------------------------------------------------- # Scalar: per-call cost # --------------------------------------------------------------------------- @@ -256,6 +313,12 @@ def run(sizes, sample_sizes) -> list[Result]: results.append(measure("batch", case, n, fd, sp, "scipy", inner=inner, repeat=15)) print(f" batch {case:<18} n={n:<9,} {_fmt(results[-1])}") + for case, n, fd, prim in primitive_cases(sizes): + inner = 50 if n <= 1_000 else 1 + results.append(measure("primitives", case, n, fd, prim, "numpy/scipy.special", + inner=inner, repeat=15)) + print(f" prims {case:<18} n={n:<9,} {_fmt(results[-1])}") + for case, n, fd, sp in scalar_cases(): # More rounds than the batch cases get. These loops are dominated by # per-call Python overhead, which the interpreter varies far more than diff --git a/examples/release_benchmarks.ipynb b/examples/release_benchmarks.ipynb new file mode 100644 index 0000000..36b15ff --- /dev/null +++ b/examples/release_benchmarks.ipynb @@ -0,0 +1,328 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "id": "95d090aa", + "metadata": {}, + "source": [ + "# fastdist release benchmarks\n", + "\n", + "Reads the JSON reports under `benchmarks/results/` and renders the numbers behind\n", + "[`BENCHMARKS.md`](../BENCHMARKS.md). Nothing here is typed in by hand: every figure comes from a\n", + "recorded run, tagged with the commit and machine that produced it.\n", + "\n", + "To measure your own machine first:\n", + "\n", + "```bash\n", + "pip install scipy matplotlib\n", + "python benchmarks/run.py\n", + "```\n", + "\n", + "## How to read these numbers\n", + "\n", + "There are two baselines, and they answer different questions.\n", + "\n", + "**vs `scipy.stats`** is what you would replace: `scipy.stats.norm.pdf(x, 0, 1)` and friends. fastdist\n", + "wins here by a wide margin, but most of that margin is SciPy's generic distribution machinery —\n", + "argument validation, broadcasting, masking — not faster arithmetic.\n", + "\n", + "**vs primitives** is the honest ceiling: the same quantity written directly with numpy or\n", + "`scipy.special`, e.g. `scipy.special.ndtr(x)` for the normal CDF. Against that, fastdist is close to\n", + "parity, sometimes behind. If you are already writing vectorised numpy, this is the comparison that\n", + "matters to you.\n", + "\n", + "Quote the first number and you will mislead someone. Both are shown below for every case." + ] + }, + { + "cell_type": "code", + "execution_count": 1, + "id": "5d39dd98", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-12T01:36:20.207740Z", + "iopub.status.busy": "2026-09-12T01:36:20.207512Z", + "iopub.status.idle": "2026-09-12T01:36:20.553572Z", + "shell.execute_reply": "2026-09-12T01:36:20.552834Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "report 0.1.0_20260912T013406+0000_da5f274059.json\n", + "fastdist_version 0.1.0\n", + "git_commit da5f274059956eb4a82cd491daa3a6383b89ac79\n", + "timestamp_utc 2026-09-12T01:34:06+00:00\n", + "processor AMD Ryzen 7 7700 8-Core Processor\n", + "platform Windows-11-10.0.26200-SP0\n", + "cuda_available True\n", + "toolchain python 3.14.2, numpy 2.5.2, scipy 1.18.1\n" + ] + } + ], + "source": [ + "import json\n", + "from pathlib import Path\n", + "\n", + "import matplotlib.pyplot as plt\n", + "\n", + "RESULTS = Path(\"..\") / \"benchmarks\" / \"results\"\n", + "report_path = sorted(RESULTS.glob(\"*.json\"))[-1]\n", + "report = json.loads(report_path.read_text(encoding=\"utf-8\"))\n", + "env = report[\"environment\"]\n", + "rows = report[\"results\"]\n", + "\n", + "print(f\"report {report_path.name}\")\n", + "for key in (\"fastdist_version\", \"git_commit\", \"timestamp_utc\", \"processor\", \"platform\", \"cuda_available\"):\n", + " print(f\"{key:18} {env[key]}\")\n", + "print(f\"{'toolchain':18} python {env['python']}, numpy {env['numpy']}, scipy {env['scipy']}\")" + ] + }, + { + "cell_type": "markdown", + "id": "8553b789", + "metadata": {}, + "source": [ + "## Both baselines, side by side\n", + "\n", + "`speedup` above 1 means fastdist is faster. `max abs diff` is how far the two implementations\n", + "disagreed on the same input; the suite checks this before timing anything, because a speedup on a\n", + "wrong answer is not a speedup." + ] + }, + { + "cell_type": "code", + "execution_count": 2, + "id": "dcd75276", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-12T01:36:20.555353Z", + "iopub.status.busy": "2026-09-12T01:36:20.555046Z", + "iopub.status.idle": "2026-09-12T01:36:20.559480Z", + "shell.execute_reply": "2026-09-12T01:36:20.558968Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "\n", + "n = 1,000\n", + " case fastdist vs scipy.stats vs primitives max abs diff\n", + " bernoulli_pmf 2.0us 24.77x 1.04x 2.2e-16\n", + " exponential_cdf 4.5us 6.83x 1.04x 1.1e-16\n", + " exponential_pdf 4.1us 7.22x 0.97x 0.0e+00\n", + " normal_cdf 6.8us 4.50x 0.63x 2.2e-16\n", + " normal_logpdf 1.9us 16.86x 0.97x 8.9e-16\n", + " normal_pdf 5.2us 6.09x 0.88x 1.1e-16\n", + " poisson_cdf 9.9us 7.12x 3.63x 2.2e-16\n", + " poisson_pmf 40.7us 0.95x 0.37x 2.0e-19\n", + " uniform_cdf 2.1us 15.19x 1.58x 0.0e+00\n", + " uniform_pdf 2.0us 15.74x 1.41x 0.0e+00\n", + "\n", + "n = 100,000\n", + " case fastdist vs scipy.stats vs primitives max abs diff\n", + " bernoulli_pmf 290.9us 12.02x 0.16x 2.2e-16\n", + " exponential_cdf 360.5us 4.67x 1.00x 1.7e-16\n", + " exponential_pdf 309.1us 4.34x 0.82x 0.0e+00\n", + " normal_cdf 891.0us 2.42x 0.94x 2.2e-16\n", + " normal_logpdf 96.8us 18.28x 0.56x 8.9e-16\n", + " normal_pdf 413.8us 3.63x 0.68x 1.1e-16\n", + " poisson_cdf 1308.6us 4.63x 3.72x 2.2e-16\n", + " poisson_pmf 4024.2us 0.78x 0.58x 2.0e-19\n", + " uniform_cdf 99.4us 17.07x 0.51x 0.0e+00\n", + " uniform_pdf 93.6us 16.60x 0.64x 0.0e+00\n", + "\n", + "n = 1,000,000\n", + " case fastdist vs scipy.stats vs primitives max abs diff\n", + " bernoulli_pmf 3507.1us 10.88x 0.31x 2.2e-16\n", + " exponential_cdf 4264.6us 4.50x 1.46x 1.7e-16\n", + " exponential_pdf 3785.8us 4.50x 1.31x 0.0e+00\n", + " normal_cdf 9670.1us 2.29x 0.95x 2.2e-16\n", + " normal_logpdf 1346.7us 15.72x 2.25x 8.9e-16\n", + " normal_pdf 4863.4us 4.15x 1.25x 1.1e-16\n", + " poisson_cdf 14042.7us 4.44x 3.58x 2.2e-16\n", + " poisson_pmf 41920.1us 0.90x 0.65x 2.0e-19\n", + " uniform_cdf 1480.6us 12.86x 1.98x 0.0e+00\n", + " uniform_pdf 1427.6us 12.84x 0.99x 0.0e+00\n" + ] + } + ], + "source": [ + "def by_group(group):\n", + " return {(r[\"case\"], r[\"n\"]): r for r in rows if r[\"group\"] == group}\n", + "\n", + "batch, prims = by_group(\"batch\"), by_group(\"primitives\")\n", + "sizes = sorted({n for _, n in batch})\n", + "\n", + "for n in sizes:\n", + " print(f\"\\nn = {n:,}\")\n", + " print(f\" {'case':<18} {'fastdist':>10} {'vs scipy.stats':>15} {'vs primitives':>14} {'max abs diff':>13}\")\n", + " for case, size in sorted(batch):\n", + " if size != n:\n", + " continue\n", + " b = batch[(case, n)]\n", + " p = prims.get((case, n))\n", + " prim = f\"{p['speedup']:.2f}x\" if p else \"-\"\n", + " print(f\" {case:<18} {b['fastdist_s'] * 1e6:9.1f}us {b['speedup']:14.2f}x {prim:>14} {b['max_abs_diff']:13.1e}\")" + ] + }, + { + "cell_type": "code", + "execution_count": 3, + "id": "a810c006", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-12T01:36:20.560798Z", + "iopub.status.busy": "2026-09-12T01:36:20.560562Z", + "iopub.status.idle": "2026-09-12T01:36:20.705540Z", + "shell.execute_reply": "2026-09-12T01:36:20.704895Z" + } + }, + "outputs": [ + { + "data": { + "image/png": "iVBORw0KGgoAAAANSUhEUgAAA94AAAG4CAYAAACzXz9FAAAAOnRFWHRTb2Z0d2FyZQBNYXRwbG90bGliIHZlcnNpb24zLjExLjIsIGh0dHBzOi8vbWF0cGxvdGxpYi5vcmcvgI3uAAAAAAlwSFlzAAAPYQAAD2EBqD+naQAAiwtJREFUeJzt3Qm8zNX/x/EPkSVZQvYilFZSZMkSkZSolEqSovXXIlIKUbK0aJFUllZCJNpIhQhFooWoZEko+y4y/8f79P9O35k7c+/cuXO57ryej8ete+fO8r3Hd+Z7Pud8zufkCAQCAQMAAAAAAJkiZ+Y8LQAAAAAAIPAGAAAAACCTMeMNAAAAAEAmIvAGAAAAACATEXgDAAAAAJCJCLwBAAAAAMhEBN4AAAAAAGQiAm8AAAAAADIRgTcAAAAAAJmIwBsAEmDq1Kl29tlnW758+SxHjhy2cuXKw96uAwYMcMeydevW4G3/+9//rHDhwpadZcW/8frrr7eyZcseMa93qI83Ef/GWfHfHQAAD4E3AGTQ5s2b7aqrrrJ69erZli1bLBAIWPny5RParh07drRixYol9DmPpNfParJCe2SFY0D2ceDAAZs2bZo7r4oWLeoG7caPHx/1/uvXr7cbb7zRihcv7gYca9SoYZMmTcrwfQ/la2X0uAAgPQi8ASCDvv32W9uxY4cLvvPmzZul2/OFF14ImQHPjpLhb0RK/LtnPGvnySeftFq1atljjz2W6n137dplDRs2tO+//95mz55tGzdutCuuuMIuv/xymzhxYtz3PZSvldHjAoD0IvAGgAxSh000YwIAR6JLLrnEPvnkEzfjfdxxx6V63yFDhtiyZcvs9ddft1NOOcWOOeYY6969uzVq1MjuvfdeO3jwYFz3PZSvldHjAoD0IvAGgAw499xz7ZprrnHfK01R6ZnNmjVzP/ft29f97H0VKlTIGjdubJ9//nnIc6jzp9nyUqVKuc7fWWed5Wae9u7d635/4YUX2ogRI2zTpk0hz+cF/DJ69Gg77bTT3Iz76aefbu+9917E4420DjYRrx8u1r9d3nzzzRTHHml9eqzPGelv9NYs6/natm3rHqvgQkHG7t27M709PH/++addeeWVVrBgQZcmfsstt9j27dtT3O/jjz92SxcKFCjgjuH888+3Dz/8MPj7WI8hlr83o8ebqPPcn0HSqlUrl+6cJ08ed068/PLLaR5rRv/dY33tWP+OWKT3+LKKCRMmWKVKleyMM84IuV0zxqtXr7avv/46rvseytfK6HEBQHoReANABixYsMDefvtt9/38+fPd+u4pU6a4n3v06OF+1pfWTy5atMh19DSztHTpUncf3d6kSRPbuXOnzZo1ywVSY8aMcevGlfopn376qd18880uGPCeT1/e+l4Fruq0X3bZZa6omwI0rc2cM2dOmsefiNePJJa/XV577TW74YYbXHrnqlWr7KOPPnJrLCMde6zPGY1msO688063pvP33393r60Bi169emV6e3ivf+utt9rtt9/uXl+vPXnyZLv00ktDZtfGjh3r/qYzzzzTfvrpJ1u+fLlVr17dWrRoYaNGjYr5GGL5exNxvIk6z2XmzJlWu3ZtF/TOnTvX/vrrL3vooYesS5cuLsCPR6ztEMtrx/p3ZMbxRTJ8+PCQQY/UvmbMmGGJovRszRKHO/XUU4O/j+e+h/K1MnpcAJBuAQBAhrz99tsBfZzOnz8/pvsXL1488OCDD7rvf/rpJ/fYsWPHpvqYm2++OVC0aNEUt//zzz+BMmXKBBo0aBBy+99//x044YQT3HNv2bIlePudd94ZKFSoUPDnjL5+evn/dh176dKlAxdccEHIfQ4cOBCoUKFCimOP5Tkj/Y3Stm1b93wfffRRyO0dOnQIFCxYMNPbw3v9iRMnhtw+ZswYd/u7777rfj548GCgbNmygerVq6d4jlq1agVKlCjh2ietY4j1783o8Sb6PD/11FMDZ511VvBv9PTu3TuQN2/ewObNm6P+G2fk3z3W147174hVRv+dhg0b5h4fy9f06dPT/Zn2zjvvpPjdrl273O907OEWLlzoftevX7903zeSzHqtjB4XAMSDGW8AyCRKyb3//vvt5JNPdmnU3syTZtJ++eUXd59y5cq51FLNbmnmXDNo6aG017Vr17rZbr/cuXNb8+bN03x8Rl8/I3+7jv2PP/5wM7l+Rx11lF188cVxPWdqcuXKZU2bNg25TWmmel4vRTuz2kNy5szpZov9WrZs6f4GLy1bbaJZT6W7hlPK94YNG+yHH36I6fVi+XszeryJPM9//fVXN0Ouv13ngJ9S65XKHU/6byztEOtrZ8b5kZF/J6Wk+zMeUvtSIbFE0r9xrL9Lz30P5Wtl9LgAID0IvAEgk6gT/8Ybb9izzz5r69atcyml3lZj+/fvd/fJnz+/K2ik25RqqlRhrRkdOHCg7du3L83X8Dr+JUqUSPG7SLeFy+jrZ+Rv9479+OOPT/H4SLfF8pyp0XOGB1VauyzeWvLMag9RwKYgy0+Bqo7BC7C8NilZsmSKx3u3xRI0x/r3ZvR4E3mea2snefTRR93r6tj1pQEArXH3t096xNIOsb52ZpwfGf13OtTUBiokqa0Tw3nH6xVnS899D+VrZfS4ACAeBN4AkAk0k/vZZ5+59aGaeS5SpIibQdEaUc1o+p1zzjluXbg6fFpn2qBBA3vwwQftgQceSPN1tMZXNBMaLtJtkWTk9TPyt3vHrgJe4cJvS097ZnQGK9Ht4dE6YB2vn4I1zWx6beF19lP794x17+6MztjFcryJPM+9v+uJJ55wj//nn3/clxfI6+u6665L998RSzuk57UTfX5k5N/pcK3x1oy8sjPCeWv6NRgRz30P5Wtl9LgAIL0IvAEgE6lQk5+KJoUHMx7NwNSvX98GDx7sOvdffPFF8HeqnhxpRk3FgUqXLm3vv/9+yO16DRUqS494Xj8jf7t37P5q3aKAxytQl97nTKREt4eCuPC/VYXkFNSpCrjXJmXKlIlYlf7dd991WQyqtB3vMST6eBN5nutvV7q6CgMe6q2c4nnt1M6P7E5ZDlpGEL7sQeeo0vFr1qwZ130P5Wtl9LgAIL0IvAEgEyigVGf8ueeec1sUaZZQnfpXX33VKlasGLyfZsyuvfZamz59ukvf1RZCqhytWZcLLrggZHZGlZR1PwWmwQ/xnDnt8ccfd7NZqsCsWVFVB7/pppusatWqaR5nRl8/I3+7d+yaMe3Zs6eb5dY2Ptqyygsu0/ucGZUZ7eHR9lPaM1h/744dO2zatGl2zz33WJ06dYJr9NUmmnVVtXxtj6X1+5pV1r7CqvSu33np3/EcQ3rEcryJPs9feeUVVxVdW/QtXrzY3U/nhJ5T26tlxt+ZnteO9e/Q82iWWf9umelwrfHWuVm5cmVr3769q7q/a9cutwWg1v4PGjTIncfx3DdSu2XWa6XnvgCQEHGVZAMApFnVfNWqVYFWrVoFChcu7L6uvvrqwIYNGwKnnHJKoGXLlu4++/fvdxWSGzdu7CpUFyhQwFVWfuqpp9zvPPv27Qu0a9cucNxxxwVy5MjhXu+vv/4K/v6NN94IVKlSJXD00Ue7/48fPz7Qv3//NKuaJ+r1w8Xyt3tef/314LGfdtpprpK2KknrNXbv3p3u54xW3VrV36NVhf75558ztT2811+/fr07Vj2vHqvK5JEqt7///vuBOnXqBPLnzx/Ily9foHbt2oFJkyaF3Ce1Y4j1703E8SbyPJcffvghcO211wZKlizpzony5csH2rRpE5g9e3bwPumpap6edkjrtWP9O2bNmuWef+DAgTG1c6zHl5l07qRWFT383/2PP/5w55/aIU+ePIFzzjknarX7WO8brd0y47XSe18AyKgc+k9iQngAABKjU6dONm7cONu2bRtNiiOOMjmeeeYZW7FiRbBQGmg3AMmNPBoAQJaidctaW6yiVcCRSOn53bt3J+im3QAgiBlvAMBhozW0mh2844473HpLzRAqYJk6darNnj2bAkcAACBbYMYbAHDYqHpwjRo1XJEoVeyuW7eu7d271xWvoqowAADILpjxBgAAAAAgEzHjDQAAAABAJiLwBgAAAAAgE+XKzCc/kh08eND++OMPO/bYYy1HjhyH+3AAAAAAAFmIdubesWOHlS5d2nLmTH1Om8A7CgXdKvoDAAAAAEA0a9assbJly1pqCLyj0Ey314gFCxZMtREBAAAAAMll+/btbrLWix1TQ+AdhZderqCbwBsAAAAAEEksS5MprgYAAAAAQCYi8AYAAAAAIBMReAMAAAAAkIlY453BLcf+/vvvxP1rAAjKnTu3HXXUUbQIAAAAjngE3nFSwP3bb7+54BtA5ihcuLCVLFkypoIVAAAAQFZF4B3nRunr1q1zs3EqH5/WZukA0v8e2717t/3555/u51KlStGEAAAAOGIReMfhwIEDLigoXbq05c+fP/H/KgAsX758rhUUfB9//PGknQMAAOCIxVRtHP755x/3/6OPPjrR/x4AfLyBrf3799MuAAAAOGIReGcA606BzMV7DAAAANkBgTcOudWrV7s18gAAAACQDAi8ccjdfffd9vjjjx8xAT8DBQAAAAAyguJqCVT+wQ/tUFo54BI7Ep144olWokSJwxLwly1b1l544YVD8jgAAAAAEALvJLF+/XpXif2kk04KuV0Vo7dt22aVK1cO7k++du1aFxinVbE9rfvu2rXLva4C7Vy5/jvVunTpYrlz5w7+vHLlSlfBWs+j41HVeFWM92zYsME9V1rHntpx/fHHH7Zjxw7bvHmz/fDDD+42PU4/b9q0yf183HHHhbxuao/LkydPutoKAAAAQPIi8E4SM2fOtE6dOrkg1tumSTp06GBFixa1N954w83oPvzww+7nLVu2WIsWLezFF1+0AgUKpHi+1O6rAP+uu+6yt956ywWl+vnJJ590rxVpBrljx452zDHH2C+//OICaQXUTZo0sbFjx7rnmzNnjrVr184F8f5juf76690+6iNGjAjeNnToUHvwwQfdcW3dutWaN2/ubnv55Zdt/vz5LuD/7rvv3H3fe+89GzdunI0ePdr9rNfNmzevvfrqq9a4cWN3W7THTZs2LeLrHHvssZn0LwggO2QqZbcMJgAAEBvWeCeJyy67zP1/8uTJwds2btxon3zyiQtgNaOrgHjChAm2YsUKNwvcrFkzN6MbLq373nHHHTZr1iz78ccf3fpo3UfBeWref/99e+KJJ+z333+3VatW2fLly+3RRx91v1NQr4BbgbhHz/vZZ5/ZTTfdFLxNQfv//vc/dz/vuC699FJbs2aN9enTxxo1amTXXnutm7nWV6VKleyhhx4K/qzA+/7777e2bdu6GXaJ9LjixYtHfR0AAAAACEfgnSQ0y33FFVfYqFGjgrcpcCxWrJib3VWgGQgErEiRIu53OXPmtOuuu85OOeWUFM+V2n0VYL/55puueJoCWylYsKDdd999qR6fjuGSS/6d8SlVqpR169bNhg0b5n5WmvqNN95oI0eODN7/tddecynfdevWDd6mmfWDBw8Gj0tbUV1zzTV22mmnpdk+f/31ly1ZssQaNGgQklYeSUZeBwAAAEDyIfBOIprZnjJlSnBNs1LBNZN71FFHuZTtrl272vnnn28XXXSRDRgwwH777beIz5PafZUurqD07LPPTtexhQetp59+ukvh9o5V6ehz5861ZcuWuaBfgbeXuu5RwP7AAw9Yw4YNrWnTpta/f383I52aTz/91A0Q6KtVq1ZuAOGff/6JONOfkdcBAAAAkLwIvJOIUqaVJq11zQoU582b54Jxj9ZhK4VbAe33339vp556qs2YMSPic0W7r9ZIe7PC6bFnz56Qn/V4zSR7RcsUGGs2WrPe06dPd699ww03pHgeDQLodzfffLNLdVdAr5T0SDRAcPXVV7u175qpV3r74sWL3UCEfpea9LwOAAAAgORG4J1ElBKulGilm+tLwXL16tXd71ShWxSYe/c577zzQtaEe1K7b5UqVVx18A8++CDkMfv370+z+Jtmsj2ff/65S133F4LTrLeKwL3yyiuumJlmnpUivnTp0hTH1aZNGzejr1T0SZMmuds1KOA/Dq3pVsDdsmVL1zbyxRdfpDjW8Mel9ToAAAAA4EfgnWRUOExVwlWB2z/brZlepY4riF6wYIELJr/99lurU6dOiudI7b6q/j1w4EB77LHHXAq20sOVFn7VVVelelyagdfs+ZdffmlDhgyxZ555xnr27BlynyuvvNL27dvn1qZ7RdVUdbxGjRrue808KwDW8ei4VK1c//f+Bg00qOibjklruAsXLmwVK1a0Hj16uNn/MWPGWPv27YNBuCf8cfpbU3sdAAAAAPDLEfBPMx4mqkatitAVKlRw20r5aTZTa27DRdpz2b/3sgpkhc9aesW+YrF9+3YrVKiQOzYVB/Pbu3evW9Os4/VSqw/HtjTxbj9z8cUXu/b+8MMP3R7bHgXk2hJMKdfaBkyp3NEC5rTu+9FHH7mgWP8WVatWtUceecStDZd77rnHPUYVxeXCCy9091F7K11d+3grhVtBcLhbbrnFzSyr+rmCfM1+Dx8+3L7++mv3ewXQCty1Fvz44493gwualRetGe/evbstXLjQFYjTtmBKKdex6e/QDLoqsvft29dVM9eWZtEep4rw0V4HiRPtvQYkCtuJAQCAeKUWM2apwFtVpAcNGuQCGRXR0tpdFazyq1mzZsh6YQVlCna07ZO2n4rktttus7fffjsY6IkqYE+cODFTA2/ER4H3ueee69ZNp0X301r1aP/2yF54ryGzEXgDAIBDEXjnssNI63q1NlhBtNYGR+LNZHq0d3Tr1q2tXbt2qT63ZivHjx+f0OPF4aPBlqlTp7pU7/QMoAAAAADA4XZYA+/bb7/d/V9pw7EaMWKE1apVy84888xU76cUYq0b1ghE0aJFM3ysyDzKHChZsmSq97nzzjtd5XOtw/ZnMgAAAABAVndYA+/0UoCuWc9hw4aleV/Niqrgldbili9f3l566SWrX7/+ITlOpE8s/57abxsAAAAAjkRHVFVzVcdW8TVt4ZQaBdia7dY+y9ouSvs/t2jRwhUUi0bVspWj7/8CAAAAAOCQz3irFtv8+fPdfsdeirhSfxXsqvBVjhw5MnxQ0V731Vdfteuuuy5F5fNwuo8nT5489txzz7mtr959911XUTsSbX2lStYAAAAAAByWGW9VE3/hhRdcdXCtsdY+0Cp8pi9tK6Xq4yeffLLbYkn3TTRVPNcsdqdOndL92KOPPtpt+ZTajLe2i1I1Ou8rtfsCAAAAAJDwGe+zzjrLFSnr1auXXXLJJSkKlmkttfaF1p7KCsR//PFHSyQVVTv77LPtnHPOSfE77RWtLce0T7dmxhX4a49nz6pVq9zXKaecEvX5NTOuLwAAAAAADkvgrXRtbdEVTbFixax9+/bua9q0aTE959atW126+oYNG9zP2htbz6PZaX3576c0ce35HYkGA+bNm+e2mtq/f7+bfb/77rvt9NNPd+u8e/fu7Wbj27ZtG+ufCwAAAADAoU019wfd69evj/m+ae3jfc0117h11wqSn376afdz+P7bSjPXbLV/7bZfmTJlXAq8l1aux6ui+b333uvWhV9//fVuXXr+/PljOi4AAAAAABIlR0C52emUM2dOt092dqaq5toDXOu9CxYsGPK7vXv3utl57T+dN2/ew3aMOLJ07NjRSpUqZY899phlR7feeqsdd9xxrlBhoh7Dew2ZrfyDH2aJRl454JLDfQgAACCBMWNC9vEuXbq0rV271s00w6d3oUPbHL230fyZQEsX/vnnH3v88ccT+ryqg5CdB2o2bdqU7l0N4nkMAAAAcKSJK/BWCrfSw1955RU3WwVkJ2PGjLGnnnoq4c+rAoFHHXWUZVf6PFA2DAAASB5kDgGxiauXPHjwYJswYYIVL17czX6XL18+5AtZz+uvv+4qwoevLOjcubPdcMMN7nuti2/ZsqVbT3/BBRfY6NGjoz7ftddea4888ojbhk37t6vi/BNPPBHy/BdddJG9/PLLIY+78847rVu3biHP8+CDD7qBHB3fmWeeac8++6wrqPe///3PqlSpYuedd16Kdf96XM+ePa1Lly5WvXp1q1q1qjsvPW+99ZY7pvAlESq6d+ONN0b9u37++We3ldyFF14Y8fdptdE777xjjRo1cjUHrrjiClu2bFnwdw888ICrY5AZf8PChQutVatWwePSYz3t2rWzhx9+2O6//37XxtqhQG0cTm2sx2p3ANVp0C4F4VTksHHjxu7vu/zyy23p0qXB3+lcGDhwYPBn/W1ly5Z1X3pNbQWYVn0IAAAAIDuKK/BWh3rYsGEuqHr00UetR48eIV/IehQEL1682GbNmhW8bd++ffbaa6+5QFFraZs2bWpnnHGGvf/++24d8pQpU+zbb7+N+Hx//fWX9e3b14455hh788033TmhQFyBp0fV6nfs2JEitXjz5s0hz/Pkk0+6tc+jRo1ywbYGAxQgag39e++95wJUFdbTunr/4/r16+eq2L/xxhsumH/ooYds5MiR7vfa8u6nn36yTz75JPiYnTt3ut9ffPHFUdtJf7uCz0iF+NJqo+eff946dOhgV155pU2aNMkFvGoXf6r5li1bEv43/P333+64NEjhHdenn37qCgp6rzNgwAC35Z4GYBQgK51eM9Qefa9MFg2MfPzxx3bTTTe5Nv/oo4+C99E2gfqbFODr71Pw7//79G/r//s0oKDdBvSl80z/7i1atEgx+AMAAABkd7niLRKFI0vJkiVdgK3gtn79+u42zWgq+FagqG3dFDTddddd7r7afu38889PtYiegj1voOXUU0+1Sy+91G0ld/XVV6d7UECz3qLgUdvGqcq9AjfvNqV+f/HFFy4Y9+g1FeyKguFffvnFBbIKGosUKeL+LgWpzZo1c/cZO3as26tdgWM0H3zwgbVu3Tri71JrIwW/Cmb79Onjglc57bTT3Ox4ahLxN2gfewW9GrTQ7HKkfzvNUL/wwgtuPbVeZ8WKFe51brnlFreeXTPiGkjTLL13f81ma2a8efPmbnBA/9b6G/X3e3+fAuloChcu7L5Ex6UBGhWd0LZ/ymwAAAAAkkWGFmSqs+/NqiHr07ZqmpFWkCgKwi+77DI79thjXUCrlHEF088884wtWrTI3Se1NbsK4PxKlCjhZnXTS0G2n5YwhN+m/d3Dn7tu3bohP9erV89+/fVX27VrV3CAaPLkycEZdm0tp73cFbhGomqEs2fPdgMIkaTWRkuWLHGP1yCCX1prnhPxN5xwwgkuHV/BuQYtNAOvWWX/a+t1/EXM9DqrVq1ylRiXL1/u2lYBtZaKnHjiie45FahrIECUMq9Bh/T8fSrAqEEIZS/o+TQgoMGAlStXptomAAAAQHYTV+Ct9beacVMwVLNmzeDtum3u3LmJPD4kkGYzNcOtVGIFiZrxVjAuKvqloFPrkJWSrlRmrctdvXp11OeLVCgsnjTiSM8Ty3Nrz/ZIP3sDCw0bNnQBnwYYFFx++eWXbiY5mqlTp7rZdT0m2nFGa6MDBw64+0QL6qNJxN+g4HfmzJkuTf377793Keo6Ln+AG+11dD4ohV6U7q6/T68xZ84c++6779z/JZ6/T8eh4Pu5556zGTNmuJRzPV6vCQAAACSTuAJvrUVV59lfOEruuOOObLtHcXZQoEABN8OtIE6FtDTT7aUwi4IizaZqPa5mQ2XIkCFxv57SisPXeHvPmwgKMv0UDBctWtSlaHtuvvlml6qtLxUqq1atWqpp5tFmu9NqI6Vm58qVK90ZIIn6G3RcWpOtGXEdl47FX6gt0utoz0FlF+jYFYjr/ewVQ/O+lFIvKrim+8T69/3555/uNbQ/t9LeTzrpJBfge0E+AAAAkEziCrwVoGg9qFJH/WrUqOFm3pB1KWhUAS79+7Vp08YFaKL0ZA2aaPmAKK1YQbMCs3gpLVvV772CW5pR1axnouhc86qKKyVaVdVvv/32kPuoAJjWFCs4Tm22WynQygRILfBOrY0UxCpA1lppr9jaunXr3JrvzPgbtC5cFcxFM9N6HS8VX8elFHL/v51msdX+yhrQ+m4VW/NeRwMy+r53797B969m3PU+99af6z6qSq5iat988427TRXK9ZhINHCgwntqU1Gq/G233ZZqWwAAAADZVVzF1dSx92bk/OtGVXGZfXyzNs1wK4jSzKUXVImCuIkTJ7pUawVne/bssWuuuSZYSCseSn1WEKpZU72mirppK6pEueqqq9w6ZGVa6NzTz3rN8HXnKgCm6twadIjGGxCoVatW1Puk1UZKqVY2iGZ4NaChGf9I23Yl4m9QYK00btGMtY5Hxc40gKDj0vOoSrlHRdpUuVxbkWmwQD/7K5KreJ3+jbRFmNLK9XwqxuffHkzrx3Pnzm0NGjRwaffKmNBtkeh+mqFXsK3BChVn03r1r7/+OtX2AAAASGbsi5595QjEsShXgYWqIWv/Z3XAVRVZVFVZ+yBrreyRToGNZjG1FloBlJ/SZbW1lYpt5c2b1440mhnV36BU4kg0O6lq1KkNoug5lHrsbxut/de5oFRpPwWUSoVWMKbn1mCNN3AT6Xm0/ZXu779Nqcva4kvBoWifbc2oa+ZWgaaCRQWCkSiY1L/T22+/HfXvUbCrquWaFY5Fam2kY9F5E94OminX+8Wr9J2Rv0HBs75Kly6d5nFpsEWF8BRcp/U6Ctx1nMcdd1zEdfap/X16bb2u9/eJPl70b6zb9O+vCuz6t8+XL1/Ux2Sn9xqyPjo4AMDnaFbCdSn7xIwJmfHWuk1tMaTtndSx1syZgm2tIyXVPOtTUbzUKOiK5zmiBU9esBzpuSM9T6T09uOPPz7qsXhBXCQ//vijS5nWuZoa3Sc9e9Cn1kaa7Q4PSiXSbfH+DQqcIwXPaf3bpfY6oiA4reUF0f6+SK+tQRb/84UPFMRyrgEAAABJucZbWxHNmjXLzZ4pzXXMmDFuRkoVzf1VzoHDSTUHqlev7tK4a9eunep9tRbZ28P6SP0bAAAAAGRNcc14q6CSvlQdO9rvgMykwZ7wLbLCqYic0tPTSvuQMmXK2JH+N0Tz1ltvuTRvAAAAAEfQjHdqlZrTquIMJIJS1NMKRlXULSMBa3b5G/Q6WnsCAAAA4AgKvKNZsmRJmuuHAQAAAABIJulKNdfsW6TvvWrIybZXbxwF4QHwHgMAAECSSVfgre2IpF27dsHvPVpDWr58eTvvvPMsu/O2Wfr777/TrBINIH67d+92/2eNOgAAAJIm8L7++uvd/5VOrr2Bk5W2U1LBK+03rYAgtf2uAcSXTaKgW/u3a5u6aHuKAwAAANm2qnnVqlVt0KBBdt9997mfBw8ebAMHDrSKFSu6CsrlypWz7Ex7E5cqVcp+++03W7Vq1eE+HCDbUtAdvqwFAAAASIrAu0uXLtaqVSv3/dq1a61bt27Wt29f+/LLL93vxo0bZ9mdtoHSHuZKNweQeMomYaYbAAAASRt4T5s2zYYOHeq+nzJlijVt2tQF3G3btrWzzjrLkoVSzPPmzXu4DwMAAAAAkIXFtTj5wIEDwaJHCsIbN27svlcQun///sQeIQAAAAAAyTbj3aBBA+vUqZPVr1/fJk+ebP3793e3z5kzx+rWrZvoYwQAAAAAILlmvIcMGeK20Ro/fry9+OKLVqFCBXf7yJEjrVevXok+RgAAAAAAkmvGu0yZMvbOO++kuF2BOAAAAAAA+A8bUAMAAAAAkNVmvOWbb76xCRMm2OrVq12xNb8xY8Yk4tgAAAAAAEjOGW/t012vXj1bvny5jRo1ynLlymWLFi2ysWPHUtUcAAAAAICMznj37dvX3nzzTbvyyistR44c9tZbb9nBgwetc+fOtnnz5nieEgAAAACAbCmuwFsz3c2bN3ff586d2+3pnT9/fuvRo4dVqVIlXc+1a9cue/vtt+2nn36y22+/3SpWrBjye6Wzz507N+S2UqVKWZcuXVJ93r/++sulvG/YsMHOPPNMa926tR111FHpOjYAAAAAAA5Lqvm+ffvcdmJSunRpFzTL3r173e9ipe3HKleubFOmTLGnn37a1qxZk+I+06ZNs08++cRKliwZ/CpatGiqz/vrr7+6YPu9996zf/75x7p37+4GCvQ9AAAAAABHRHE1T6tWrax9+/bWpk0bmzRpkjVq1Cjmx1arVs2WLl1qO3bscDPb0Zx88snWtWvXmJ/3gQcesEqVKrmgPWfOnHbrrbe659AMeNu2bWN+HgAAAAAADsuM9/Tp04Pf9+/f35o1a+ZmratWrWrDhw+P+XmqV69uhQoVSvN+K1assF69erlZ8a+//jrV+6rC+ocffugCbAXdUr58eWvQoIGbAQcAAAAAIEsG3v4Z5wIFCgS/V8r5k08+abNnz3ZB9/HHH5/QA1TxtmOPPdZ9v2TJEqtfv77de++9Ue+/atUql/IevlZcP2ttejRKkd++fXvIFwAAAAAAhyzVfNCgQS7AViBco0YNCwQCdih069bNKlSoEPz52muvtSZNmthll10WMa1dhd6kYMGCIbdrZl2F3KLRzH2fPn0SeuwAAAAAAMQceJcrV87eeOMNq1Onjvv5l19+iXpfra9OFH/QLRdeeKGVLVvWZs2aFTHw9mbjt27dGnK7fvZmziNRAbb77rsv+LNmvPU3AwAAAABwSALvfv36uSJl3qyxqpFHk9mz4VrHHa16+gknnGDHHHOMLVu2zK0996jy+qmnnhr1OfPkyeO+AAAAAAA4LGu8VaxMs8bell+//fZb1K9EBthaO+43btw4W79+vUs397zzzjuu8Jpor+4rrrjCXn/99WBw/sMPP7jnueqqqxJ2bAAAAAAAJHw7sVy5crk0bwW6qhSeUQsWLHBbfO3cudP9PHToUPvggw+sadOm7kvryR955BHbv3+/nX766bZ69Wr77LPP3G0XXHBB8HmmTp1q8+bNsy5durifBw4caPXq1bPzzjvPVU7Xc2pt+OWXX57hYwYAAAAAINP38W7durUlgiqilyxZ0n2vwm3h67Q1e61AW1uIffvtty7YfvHFF+3EE08MeZ6rr77azj///ODPpUqVsu+++84F3Bs2bLAbbrjBGjZsmJBjBgAAAAAg0wPvRNEstr7SUrNmTfcVjWbHw+XPn98F5AAAAAAAHBFrvAEAAAAAQPoReAMAAAAAkNUC73Xr1tmgQYOCPw8ePNgVXWvQoEGw6jkAAAAAAIgz8Fb1cAXasnbtWuvWrZt17tzZihYtGqwsDgAAAAAA4gy8p02bZhdddJH7fsqUKa64mQJuVRyfMWMG7QoAAAAAQEYC7wMHDtju3buDQXjjxo3d93nz5nV7bgMAAAAAgAxsJ6a13J06dbL69evb5MmTrX///u72OXPmWN26deN5SgAAAAAAsqW4ZryHDBli+fLls/Hjx7v08goVKrjbR44cab169Ur0MQIAAAAAkFwz3mXKlLF33nknxe0KxAEAAAAAwH/YxxsAAAAAgKww412sWDH3/40bNwa/j0b3AQAAAAAA6Qi8X3jhhYjfAwAAAACABATe11xzTcTvAQAAAABAdKzxBgAAAAAgExF4AwAAAACQiQi8AQAAAADIaoF379694/odAAAAAADJJq7Au0+fPnH9DgAAAACAZJPQVPMlS5akucc3AAAAAADJJObtxKRkyZIRv5eDBw/a5s2b7bbbbkvc0QEAAAAAkEyB91NPPeX+365du+D3nty5c1v58uXtvPPOS+wRAgAAAACQLIH39ddf7/6vdPJmzZpl1jEBAAAAAJDca7yrVq1qgwYNCv48ePBgK1u2rDVo0MDWrFmTyOMDAAAAACD5Au8uXbq4QFvWrl1r3bp1s86dO1vRokXd7wAAAAAAQAYC72nTptlFF13kvp8yZYo1bdrUBdwvvviizZgxI56nBAAAAAAgW4or8D5w4IDt3r07GIQ3btzYfZ83b17bv39/Yo8QAAAAAIBkKa7m0VruTp06Wf369W3y5MnWv39/d/ucOXOsbt26iT5GAAAAAACSa8Z7yJAhli9fPhs/frxLL69QoYK7feTIkdarV69EHyMAAAAAAMk1412mTBl75513UtyuQBwAAAAAAGRwxhsAAAAAACQ48C5WrJj78n8f7StWf//9t40ZM8YaNmxoJUuWdGvEwy1ZssStJz/99NPd/uF33nmnrV+/PtXnfeCBB9zz+b/0GgAAAAAAZNlU8xdeeCHi9xnx8MMP2+rVq+2WW26xtm3bukDc759//rE2bdrYvffe6/YJVyV1bVumKuoLFixw68wj2bZtm9WoUcOGDRsWvC137twJOWYAAAAAADIl8L7mmmsifp8RAwcOtJw5c9rvv/8e8fdHHXWUfffdd5YjR47gbSNGjLDKlSvbV199leosdp48edxMNwAAAAAASbvGW0F3WvxBt+zbt8/9/+ijj071cTNmzLCTTjrJzj77bLv77rtt06ZNGTxaAAAAAAAycca7QIECMT/pzp07LTMEAgF78MEH7eSTT3ap5NGUKFHC7S2u/cbXrl3rHqP9xb/99tuo6ekK6L2gXrZv354pfwMAAAAAILnEHHirCJpn8eLF1q9fP1f0zAuA58+f79ZUP/TQQ5lzpGbWrVs3++KLL2zmzJmprtnu3bt3cKZcQfrkyZOtXLly7m/o0KFDxMcoUO/Tp0+mHTsAAAAAIDnFHHhfeumlwe+feOIJGz16tLVs2TJ4m4qjXXDBBTZo0CBXNC3RevToYS+//LJNnTrVqlWrlq70dM2An3jiibZ06dKoj+nevbvdd999ITPeCtYBAAAAADgkgbffokWLXJAdTre1a9fOEq1Xr172/PPP25QpU6x27drpfvyePXvsjz/+SHWrMxVj0xcAAAAAAIe9uFrhwoVtwoQJKW4fP368FSlSxBJJ6d/PPvusC7rr1KkT8T7aYsyrcK512rfffrvbpkw2b95sN910kyvklqhq7AAAAAAAZOqMt4Jhre9+//333RpvFT3Tvtr6Wdt9xWrs2LF2zz332MGDB93PV1xxhatW3rVrV/eloFnrtfPmzet+56eU9uuuuy64b/fGjRvd95q1Pvfcc+3CCy+09evX2/79+13ArrXhJ5xwQjx/LgAAAAAAhzbwVoGyKlWquJlorfWW0047zWbNmmW1atWK+Xkuu+wyV3k8WgV1zZ6vW7cu4mMLFSoUEoQrwPbcfPPN7kvV1Y855pgUa74BAAAAAMjSgbdmofUVab2197tYaGuvaNt7iQLmkiVLpvk8BQsWzPAWaAAAAAAAZJk13qltu8WWXAAAAAAAZHDGO5olS5akWjkciVf+wQ+zRLOuHHDJ4T4EAAAAADjyA29/2nd4CrgKpKkY2m233Za4owMAAAAAIJkC76eeesr9X3t1e997cufObeXLl7fzzjsvsUcIAAAAAECyBN7XX3+9+7/SyZs1a5ZZxwQAAAAAQHIXV6tatarbwsszePBgK1u2rNsabM2aNYk8PgAAAAAAki/w7tKliwu0Ze3atdatWzfr3LmzFS1a1P0OAAAAAABkIPCeNm2aXXTRRe77KVOmWNOmTV3A/eKLL9qMGTPieUoAAAAAALKluALvAwcO2O7du4NBeOPGjd33efPmtf379yf2CAEAAAAASLZ9vLWWu1OnTla/fn2bPHmy9e/f390+Z84cq1u3bqKPEQAAAACA5JrxHjJkiOXLl8/Gjx/v0ssrVKjgbh85cqT16tUr0ccIAAAAAEByzXiXKVPG3nnnnRS3KxAHAAAAAAAZnPEGAAAAAACZOOMt33zzjU2YMMFWr17tiq35jRkzJt6nBQAAAAAgW4lrxnvcuHFWr149W758uY0aNcpy5cplixYtsrFjx1LVHAAAAACAjM549+3b195880278sorLUeOHPbWW2/ZwYMHrXPnzrZ58+Z4nhIAAAAAgGwprsBbM93Nmzd33+fOndvt6Z0/f37r0aOHValSJdHHCAAAAABAcqWa79u3z20nJqVLl7affvrJfb937173OwAAAAAAkMHiap5WrVpZ+/btrU2bNjZp0iRr1KhRRp8SAAAAAIDknvGePn168Pv+/ftbs2bNbMqUKVa1alUbPnx4Io8PAAAAAIDkmPHu2rWrPfXUU+77AgUKBG9XyvmTTz6ZOUcHAAAAAECyzHgPGjTIAoGA+75GjRqZeUwAAAAAACTfjHe5cuXsjTfesDp16riff/nll6j3rVSpUmKODgAAAACAZAm8+/XrZ7feeqvt2rXL/Vy5cuWo9/VmxgEAAAAASHYxB95t27Z1lcvXr1/vZr9/++23zD0yAAAAAACSbTuxXLlyWdmyZe2dd96x8uXLZ95RAQAAAACQzNuJtW7dOvFHAgAAAABANhRX4A0AAAAAAGJD4A0AAAAAQHYPvFesWGFTpkyxzZs3R73P0qVLbcaMGbZhw4aYnzeexwAAAAAAcNgD73Xr1tmgQYOCPw8ePNgVXWvQoIGtWbMm5ueZM2eONWvWzOrVq2cXX3yxfffddynuo+3LLrroIrd/+P333++Kug0YMCDV543nMQAAAAAAZJnAu0uXLi7QlrVr11q3bt2sc+fOVrRoUfe7WK1atcruuecemzt3btT79OrVy5YvX24///yzzZ8/39577z3r3r27zZ49O6GPAQAAAAAgywTe06ZNczPKohTxpk2buoD7xRdfdKndsbr22mvdTHfOnNEP44033rCbb77ZihUr5n7W65599tn2+uuvJ/QxAAAAAAAc9n28PQcOHLDdu3dboUKFXBDeuHFjd3vevHlt//79CTu433//3TZu3OiCZj/9vHjx4oQ9Rvbt2+e+PNu3b8/w8QMAAAAAENeMt9Zyd+rUyZ544gmbPHmytWjRIrhmu27duglr1a1bt7r/H3fccSG3K6V9y5YtCXuM9O/f3w0keF/lypVLwF8AAAAAAEh2cQXeQ4YMsXz58tn48eNdenmFChXc7SNHjnTrqxPl6KOPdv/fs2dPyO2abfd+l4jHiNaAb9u2LfiVniJxAAAAAAAkNNW8TJky9s4776S4XYF4ImnW+aijjkoRBCudXJXKE/UYyZMnj/sCAAAAACBL7eO9d+/eFF+Joln1+vXru6rkHs1Gf/rpp24bMs+PP/5oX375ZboeAwAAAABAlp3x/uWXX+z22293wW54SrcEAoGYnkdbkX3//feuGJp8/fXXLnCvVKmS+5J+/fpZw4YN7a677rLatWvb0KFD3ay2qpZ7nnnmGZs3b5798MMPMT8GAAAAAIAsG3grgFUF87Fjx1qRIkXifvElS5bYs88+G9zy6/PPP3df119/fTDwrlWrltvnW8GzXk+F3e677z7Lnz9/8HnOOOMMy5Xrvz8llscAAAAAAJBlA+9vvvnGfvvtNytevHiGXrxJkybuKy3aCuyVV16J+vt777033Y8BAAAAACDLrvFWcTXt5Q0AAAAAADIh8Nb67vvvv9927doVz8MBAAAAAEgacaWaq5jZ6tWrbdy4cVaqVCnLkSNHyO9XrlyZqOMDAAAAACD5Au+ePXsm/kgAAAAAAMiG4gq8O3bsmPgjAQAAAAAgG4prjTcAAAAAAMjEGW9vS7EJEya4td7hFc7HjBkT79MCAAAAAJCtxDXjraJq9erVs+XLl9uoUaMsV65ctmjRIhs7dqzt378/8UcJAAAAAEAyzXj37dvX3nzzTbvyyitdRfO33nrLDh48aJ07d7bNmzcn/igBAAAAAEimwFsz3c2bN3ff586d23bv3m358+e3Hj16WJUqVRJ9jAAAAAAAJFeq+b59+yxfvnzu+9KlS9tPP/3kvt+7d6/7HQAAAAAAyGBxNU+rVq2sffv21qZNG5s0aZI1atQoo08JAAAAAEByz3hPnz49+H3//v2tWbNmNmXKFKtataoNHz48kccHAAAAAEDyzXg3bNgw+L1Szp988slEHhMAAAAAAMk94+3ZtGmTzZ8/P3FHAwAAAABANhNX4L1161a3trtYsWJWs2bN4O26be7cuYk8PgAAAAAAki/w7tatm6tevmzZspDb77jjDnvssccSdWwAAAAAACTnGu8PPvjA5s2bZyeccELI7TVq1LCZM2cm6tgAAAAAAEjOGe8tW7ZYkSJF3Pc5cuQI3r5z507LmTNDy8YBAAAAAMhW4oqSzznnHJs4cWKKwHvgwIFWp06dxB0dAAAAAADJmGquvbubN29uX3zxhQUCAevZs6dNnTrVvv/+e1LNAQAAAADI6Ix3vXr1bNasWbZnzx6rXLmyjRkzxipUqOAqmvurnAMAAAAAkOzimvGWatWq2ahRoxJ7NAAAAAAAZDNUQgMAAAAAIKvNeO/YscMGDRpks2fPdhXOwy1YsCARxwYAAAAAQHIG3h07dnTB9VVXXWWFCxdO/FEBAAAAAJDMgfdHH31kixcvtpNOOinxRwQAAAAAQLKv8T722GOtYMGCiT8aAAAAAACymbgC706dOtlDDz1k+/btS/wRAQAAAACQ7KnmHTp0sOrVq7vtxMqWLWs5cuQI+f1PP/2UqOMDAAAAACD5Au8bb7zRSpYsadddd12mF1dTIbe9e/emuP2CCy6wm2++OeJjXn31Vfvss89CbitXrpz1798/044TAAAAAICEBd7z5s2z5cuX2wknnGCZrUmTJrZ///7gz6tXr7aHH37YGjVqFPUxX331lTu+u+++O3gb1dcBAAAAAEdM4F2mTBnLkyePHQpt2rQJ+fmRRx5xxd3Cbw+nQYHrr78+k48OAAAAAIBMKK7Wtm1b69q1q+3evdsOpYMHD9prr73mUtyPOeaYVO+7ZMkSu+WWW+z++++3yZMnH7JjBAAAAAAgwzPer7/+ukv5Hjt2rJUqVSpFcbWVK1daZpg2bZp7XVVVT03OnDntzDPPtGrVqtnatWvthhtusIsvvtjefvvtqI9RhXZ/lfbt27cn9NgBAAAAAMkprsC7Z8+edjiMGDHCzj77bDvnnHNSvV+fPn2sePHiwZ9btWpl5513npupv/TSSyM+RoXX9DgAAAAAAA574K1K44faxo0bbdKkSfbss8+meV9/0C01atRwa76//vrrqIF39+7d7b777guZ8VYldAAAAAAADnngfTi8+eablitXLre+Ox67du1K9fcqFneoCsYBAAAAAJJHXMXVDoeRI0fa1VdfbYUKFYqYgq4Za9HWYxMmTAj5/QsvvOBmzC+55JJDdrwAAAAAABwxgbf25f7hhx+iFlWbO3euvf/+++77o446ysaPH29VqlSxyy+/3K0H177fQ4cOdeu8AQAAAAA4lI6IVHNtHaaK5HXq1Im65lwF1LyK5rqvKqsvXrzYihQpYmeddZYVLlz4EB81AAAAAABHSOB9xhlnuK9oatWqleK28uXLuy8AAAAAALJVqnmBAgUS/ZQAAAAAAByxEh54p1U9HAAAAACAZJKuVPNoe2ADAAAAAIAEBN5Tp061Fi1a2NFHH52ehwEAAAAAkLTSFXircnjr1q2tTZs2Ue8zduzYRBwXAAAAAADJt8b79ttvt5deeinV+xQtWjSjxwQAAAAAQHLOeDdq1MgqVqyY6n02btyY0WMCAAAAACB5q5qfeOKJmXMkAAAAAABkQwnfTgwAAAAAACQw8H7++edt8ODBGX0aAAAAAACypXSt8Q63detWe+CBB9z37dq1s8KFCyfquAAAAAAAyBYyNOM9atQoK1eunJUtW9ZGjx6duKMCAAAAACCbyFDgPWLECLvxxhvdl74HAAAAAAAJCry//fZbW7x4sd1www3Wvn17W7RokfsCAAAAAAAJCLw1w920aVOXZq6vJk2aMOsNAAAAAEAiAu+9e/e6Nd0dOnQI3qbvteZbvwMAAAAAABkIvN999133/5YtWwZva9Wqlfv/xIkT43lKAAAAAACypZzxpplfd911lidPnuBt+v7aa68l3RwAAAAAgIzu4z148GArU6ZMitv79etna9eujecpAQAAAADIluIKvE877bSItxcqVMh9AQAAAACAdKaaDxkyxPbv35/m/f7++293XwAAAAAAkI7Ae8qUKVapUiXr06ePLVy40A4cOBD8nQLyr7/+2nr06GEVK1a0jz/+mLYFAAAAACA9gff7779vw4YNsy+//NLOPfdcy5cvn5UsWdJKlChh+fPnt1q1atmCBQts5MiR9sEHH9C4AAAAAACkd41306ZN3dfGjRtt7ty5tmbNGsuRI4eVLVvW6tSpY0WLFqVRAQAAAADIaHG1YsWKWYsWLeJ5KAAAAAAASSWufbwBAAAAAEBsCLwBAAAAAMhEBN4AAAAAAGQiAm8AAAAAALJacTXPrFmzbOnSpe770047zc4///xEHRcAAAAAAMk7471q1SqrUaOGNWjQwB555BH3Vb9+fatZs6atXr06oQd42223uS3L/F9nnHFGmo975pln7MQTT7S8efO6Y509e3ZCjwsAAAAAgEwLvDt27GjHHXecrVy50tatW+e+9H2RIkWsU6dOlmhXXnmlBQKB4NcPP/yQ6v1HjBhhDz/8sA0ZMsTWrl1rjRo1smbNmiV8UAAAAAAAgEwJvDV7rOD2hBNOCN6m73XbF198YYfbk08+6QYHLr30UitatKgNGDDAChcubEOHDj3chwYAAAAASDJxBd5lypSxgwcPprhdt5UtW9YSbcqUKZY/f34rVaqUXX311W52PZrNmzfbsmXLrGHDhsHblJ6un+fMmZPwYwMAAAAAIOGBt9LJb7rpJluxYkXwNn3foUOHhKeaV6pUyd5++21bv369ff7557Zt2zarV6+e+38kup8UL1485Pbjjz8++LtI9u3bZ9u3bw/5AgAAAADgsFQ1f/HFF9166YoVK7q13lp3vWXLFve7X375xf3ek9rsdCy6du0a/L5gwYI2duxYN/Ot/99yyy0xP4+OUTPf0fTv39/69OmToWMFAAAAACAhgXfPnj3tcNFabaWz//zzzxF/X7JkSff/v/76K+R2/VyiRImoz9u9e3e77777gj9rxrtcuXIJO24AAAAAQHKKK/BW4bLDZevWrfb7779b6dKlI/5eM/CnnHKKTZ8+3a644orgbPeMGTOsXbt2UZ83T5487gsAAAAAgMO+xvtQ0brryy+/3L7++mvbtWuXLV261Nq0aWPHHnustW3bNmQgwL+3t9LTR44caR9++KErtvbggw+6gF17ggMAAAAAkOVnvAsUKJDq73fu3GmJoBloBdUKpL/99lu3T7gKq3311VeuWFo0eoxSxW+//XbbsGGDnXnmma4y+oknnpiQ4wIAAAAAIFMD77feeivFNmJac/3UU09Z586dE9r6l1xyiftKzfDhw1PcpvXa/jXbQCzKP/hhlmiolQNSP+cBAAAAZPPAu1WrVhFvP+ecc2zAgAH20EMPZfS4AAAAkh4DwgCQPSR0jXft2rVtwYIFiXxKAAAAAACOaAkNvN977z233RcAAAAAAMhAqvm5556b4rYtW7bYypUr7cUXX4znKQEAAAAAyJbiCrxbt26d4jZVHK9bt27Itl4AAAAAACS7uAJv7YsNAEdKYSCqxAMAACDbrPEGAAAAAABxzngXKFAg1rvazp07Y74vAAAAAADZWcyB95gxY4LfL1682Pr162edOnWyGjVquNvmz59vw4YNYw9vAAAAAADiCbwvvfTS4PdPPPGEjR492lq2bBm8rW3btnbBBRfYoEGD7OGHH471aQEAAAAAyNbiWuO9aNEiF2SH023ffvttIo4LAAAAAIDkDbwLFy5sEyZMSHH7+PHj3bZiAAAAAAAgA9uJ9enTx63vfv/9990a70AgYAsWLHA/jxgxIp6nBAAAAAAgW4or8O7QoYNVqVLFnn32WbfWW0477TSbNWuW1apVK9HHCAAAAABAcgXeUrt2bfcFAAAAAAAyIfCWTZs22YoVK4JbigEAACAb6l3IsoTe2w73EQDAoSuutnXrVmvVqpUVK1bMatasGbxdt82dOze+IwEAAAAAIBuKa8a7W7dutm/fPlu2bJmdcsopwdvvuOMOe+yxx+yjjz5K5DECAIAjRPkHPzzch2ArB1xyuA8BAICMB94ffPCBzZs3z0444YSQ25VyPnPmzHieEgAAAACAbCmuVPMtW7YE9+vOkSNH8PadO3dazpxxPSUAAAAAANlSXFHyOeecYxMnTkwReA8cONDq1KmTuKMDAAAAACAZU8379+9vzZs3ty+++MICgYD17NnTpk6dat9//z2p5gAAAAAAZHTGu169ejZr1izbs2ePVa5c2caMGWMVKlRwFc39Vc4BAAAAAEh2ce/jXa1aNRs1alRijwYAAAAAgGwmQ5XQNm3aZPPnz0/c0QAAAAAAkM3EFXhv3brVWrVqZcWKFQtJLddtSjcHAAAAAAAZCLy7detm+/bts2XLloXcfscdd9hjjz0Wz1MCAAAAAJAtxbXG+4MPPrB58+bZCSecEHJ7jRo1qGoOAAAAAEBGZ7y3bNliRYoUSbGP986dOy1nzgwtGwcAAAAAIFuJK0o+55xzbOLEiSkC74EDB1qdOnUSd3QAAAAAACRjqnn//v2tefPm9sUXX1ggELCePXva1KlT7fvvv8+UVPP169e76um5cuVyQf/xxx+f6v1V4C18/blm6Fu2bJnwYwMAAAAAIOEz3vXq1bNZs2bZnj17rHLlyjZmzBirUKGCC3j9Vc4zSkF9hw4d3HMOGzbMnn76afc6Q4cOTfVxr7/+uvXu3dtmzJgR/GLbMwAAAADAETPjLdWqVbNRo0Yl9mgiBN4NGjRwQbdmu2XEiBF26623WrNmzVwQHs25555rr732WqYeHwAAAAAAmRZ4i2a9ly5d6r4/7bTT7Pzzz7dEUqG2G2+8MeS2Fi1aWMeOHd3rphZ4b9q0ycaNG2eFChWy6tWrW/HixRN6bAAAAAAAZFrgvWrVKmvdurV98803VqJECXfbhg0b3Czz+PHjU2wzlkhTpkxxAfkZZ5yR6v20xlsp8GvXrrUff/zRpalrpjwa7UuuL8/27dsTetwAAAAAgOQU1xpvzTgfd9xxtnLlSlu3bp370vcqYNapUyfLLL/88ot17tzZ7r777lSD+3bt2rnjeffdd+2rr76yJ5980u68805btGhRqgXjNDvufZUrVy6T/goAAAAAQDKJK/CePXu2W2vtD371vW5TpfPMsHr1arvwwgutYcOGLpBOTd26de3oo48O/nz77be7QQHNlkfTvXt327ZtW/BrzZo1CT1+AAAAAEByiivVvEyZMnbw4MEUt+u2smXLWqIpCFbAffbZZ7v0ca/QWnrkz5/frfuOJk+ePO4LALKy8g9+aFnBygGXHO5DAAAAyN4z3konv+mmm2zFihXB2/S9tv5KdKr577//7oLuqlWrumJpuXPnTnGfOXPm2KRJk4LBv9Z1+yndXMF7rVq1EnpsAAAAAABkyoz3iy++6FK/K1as6NZ6a9uvLVu2BNdh6/cerbWO1969e61Ro0a2c+dOa968ecj2ZUon1x7iMnLkSJs3b561bNnS/vnnH2vSpIkL1k8//XR3nC+99JJdeeWVdvnll8d9LAAAAAAAHLLAu2fPnnYoHDhwwOrUqeO+//LLL0N+d9JJJwUDbwXhJUuWdN9rRnzhwoU2evRo+/bbb93ablVaVzAOAAAAAMAREXirqvmhUKBAAXvttdfSvJ9S3P3y5s3rUuEBAAAAADgi13hr+7BBgwYFfx48eLArqtagQQOqgQMAAAAAkNHAu0uXLsHq5Spk1q1bN7e/dtGiRd3vAAAAAABABgLvadOm2UUXXeS+197YTZs2dQG3iqrNmDEjnqcEAAAAACBbyhlv0bPdu3cHg/DGjRsH11bv378/sUcIAAAAAECyFVfTWm7t112/fn2bPHmy9e/fP7iftiqMAwAAAACADMx4DxkyxPLly+e26VJ6eYUKFYL7affq1SuepwQAAAAAIFuKa8a7TJky9s4776S4XYE4AAAAAADI4Iw3AAAAAACIDYE3AAAAAACZiMAbAAAAAIBMROANAAAAAEAmIvAGAAAAACATEXgDAAAAAJCJCLwBAAAAAMhEBN4AAAAAAGQiAm8AAAAAADIRgTcAAAAAAJkoV2Y+eXawaNEiK1CggGVV+9b/YlnBwoULLTugPbNnm2aX8zOrtGd2alPaM3u2aXY5P7NKe8rCPP9YlpCN/m2ziyxzjmaTc4P2PLLs3Lkz5vvmCAQCgUw9miPU9u3brVChQof7MAAAAAAAWdi2bdusYMGCqd6HGe80zJw5M0vPeF/y/CzLCj68u55lB1mmPfM8bFnCrV9kizbNLudnVmnP7HSOZpn25BylPbP6OZpN3vPIxudoNvkcpT2PvBnvBg0axHRfAu80VKtWLc3Ri8MpT8l1lhVUr17dsoMs0555j7IsIQH/rlmhTbPL+ZlV2jM7naNZpj05R2nPrH6OZpP3PLLxOZpNzg3a88jLko4VgTeA7K93Flk20nvb4T4CAAAAHAYE3gAAAEgK5R/80LKClQMuOdyHAOAQYzsxAAAAAAAyETPeAAAge2F5CQDwOZrFEHgDAAAAhxKDQ0DSIfAGAOBwoxMOAEC2RuANAAAA4MiWFQYw2b0EqaC4GgAAAAAAmYjAGwAAAACAZE81X7BggQ0dOtQ2bNhgZ555pnXt2tWKFi2a8McAAAAAAJB0M95z5syxunXr2rHHHmvt2rUL/rxr166EPgYAAAAAgKSc8X7ooYfskksusWeffdb93Lx5cytdurQNHz7c7rnnnoQ9BhlEQQsAAAAAOPJmvPfs2WOzZs2yVq1aBW/TLHbjxo3tk08+SdhjAAAAAABIyhnvNWvW2MGDB61s2bIht+vn6dOnJ+wxsm/fPvfl2bZtm/v/9u3bLSs7uG+3ZQXbcwQO9yHoHyvDT0F7Zs82zRLnZzZqz+zUprRnYtszq7Rpdjk/s0p7Zqc2pT0T2560Ke2Z6bZn7VjMixUDgcCRHXj//fff7v/58uULuT1//vzB3yXiMdK/f3/r06dPitvLlSsX17Enmyywc6LZgCxxFAmRZf6SbNKmWeavyCbtKVnmL8kmbZpl/grak/aMgnM0sWjPbNqm2eQzVLLMXzIgyxxJqnbs2GGFChU6cgPvIkWKuP9v3rw55PZNmzYFf5eIx0j37t3tvvvuC/6sWXM9hyqh58iRI0N/R3ankR4NUCjboGDBgof7cI54tCftmdVxjtKeWRnnJ22a1XGO0qZZHedo7DTTraBb9cTSkqUD7zJlyljx4sVt4cKFrliaRz/XqFEjYY+RPHnyuC+/woULJ+TvSBYKugm8ac+sivOTNs3qOEdpz6yOc5T2zOo4R2nPwyGtme4joriatG/f3kaMGGF//vmn+/mjjz6yRYsWuds9gwYNsk6dOqXrMQAAAAAAHApZesZbtO76xx9/tMqVK1vFihVt6dKl9sQTT7h9uT1LliyxefPmpesxAAAAAAAcClk+8FZRNM1YL1++3DZs2GCnnnqqFStWLOQ+Xbp0CVYhj/UxSByl6D/yyCMpUvVBe2YFnJ+0aVbHOUp7ZnWco7RnVsc5SnseCXIEYql9DgAAAAAA4pLl13gDAAAAAHAkI/AGAAAAACATEXgjJosXL6alEmj16tW2detW2jRB9u/f74ooInF4zyfW+vXrXc0RJMY///xjP/zwA82ZQN99953bjxaJsWXLFluzZg3NmSA6N3WOInFUnFr9Jxw6BN5IlTaEv/rqq61Nmza2e/duWisBhg0bZtWrVw+pxI+MXTiqVatmzzzzDM2YAPv27bOOHTvaZZddZn/99RdtmgDjxo2zM88806ZPn057JsDKlSutVq1a1qtXL9ozQYMYKlLbrFkz++2332jTBJgyZYor7PvBBx/QngkauGzUqJHdddddDA4lyKOPPmoNGzZ0fSgcOhRXQ6ruvfdeW7FihY0fP96OPvpoWiuDFixYYE2aNLH58+dbpUqVaM8EqFKlit16663WuXNn2jMB+vbta9OmTXM7QxxzzDG0aQb9+uuvdtZZZ9mcOXOsatWqtGcCnH/++a7DqHMVGffSSy+5rxkzZljhwoVp0gzSgGX58uVd8F2vXj3aMwFatWplpUqVsiFDhljOnMwZZpT69A888IC7LpUoUYJz9BDi7EWaF5AiRYpYrly5bOTIkbZ9+3ZaLI7ZBH97Hjx40EqWLGkLFy50HR3E355em6o9lbqvcxTpp3PS354FChRwQffo0aPdTAPiP0c3b97sUvl0jmo5xMcff0xzZqA9vXNUncVdu3bZyy+/THsm4LqkrZgUdE+ePNl++eUX2jQDdu7c6TIE9Z5XBsHEiRNpzwycn945evzxx7vPUg0Shf8e6WtTtacGMIoXL26fffYZS8sOIQJvpKp9+/Y2atQol8r75ptvsi45nZT+rJFaT+PGje3YY491MzYtWrRwnXLEbtmyZVaxYkX7/fffQ87Rbt26ubQ+ZRRwQU6fiy++2M0ieG644QYXHJ599tn27LPP8p5PpxEjRriUSM+5555rlStXtgsvvNDN0rLOO330Xj/ppJNs+fLlIe/5xx57zGW7fPnll7Znz570/jMltWuvvdb69esX/Pm6665zA8F6zz/88MO2bdu2w3p8R7oKFSq4me6WLVtazZo1XU0XxE6D6CeffLJ9/fXXIe95XY9OP/10++STT5gESqe7777b7rvvvuDPrVu3tj/++MNdn2677Tbe84eS9vEGornhhhsCOXPmDJx55pk0UhwWL14cOOqoowIfffSR+/mdd94JFClSRNVrAosWLaJN0+mff/4JnHPOOYHrrrvO/bx27dpA1apV3TnatWtX2jMOTz31lDsnN27c6H6+9957A7ly5QqULVs2sH//fto0nX799ddAnjx5AqNHj3Y/T506NXD88ce79/ynn35Ke8bhggsuCDRv3tx9v2XLlkDt2rXde/7GG2+kPeMwfPjwQP78+QNr1qxxP/fp0ydw9NFHBwoXLhzYuXMnbZpBc+fOdZ+fOXLkCIwZM4b2jEPr1q0DtWrVChw8eDCwd+/ewIUXXuje85deeintGYf33nvPXdeXLFnifn7++efdZ4CuVRs2bKBNDyECb6Tqs88+C3zzzTfuDTthwgRaKw633XZboEqVKoG///47sGzZssDy5cvdBeXyyy+nPeMwa9Ys16H58ssvXSD+8ccfu4vIscceG1i/fj1tmk46L08++eTAHXfc4X6eOXOmuzgfc8wxgZdffpn2jMODDz4YKFeuXGDXrl2BVatWBb777jsXONavX5/2TMAApv4/atQoFyz+/PPPtGk66XOzevXqwQFMBYq//PJLoFixYoHHH3+c9sygv/76K/DVV1+5iQtNWhw4cIA2TafffvstkDdv3sAbb7zhfp4yZYq71iv4nj9/Pu0ZBw1eXHTRRe579etXrFgRKF++fKBz58605yFE4I0Q06ZNCzz22GPuQux39913BypUqOBGHhE7dbwVICooHDRoUPB2XZQVPGpgA7HT6Lc6iM2aNQuce+657mfRzOzpp58e6NChA82ZThrtHjJkiAtsvv/+++Dtjz76aKB48eKBrVu30qbpsGfPnsCCBQtcENOrV6/g7Rp0y507d2Ds2LG0Zzqpg3j11VcHBzA9559/fqBFixa0Zzopu0UBjTeA6XnxxRcDBQoUcJlESN9n6LPPPus+R3fv3h28/Y8//nDtqduRPhqwvOWWWwKlS5cOycK47LLLAnXr1qU500mZQpMmTXIDF5MnTw7ePm7cOHdd+umnn2jTQ4TAG0EKWtSxufjii92b8+233w7+bvPmza4j2a9fP1osRpo5KFSoUKBy5cquPZXGp5FwD6Ph6aPBoIoVK7oBIM3GKnX31VdfDRk0Ujsr6EHatm/f7tL5jjvuuECZMmVcezZu3DgkgDzxxBMD9913H80ZI2VeKG1f73kNZOTLl891ID1qS7Wp2hZp03Kc0047zbWZBi91jj7zzDPB3+u9rvf8J598QnPGQEGhrju6FqlN1Z7+AUzNzJ511lnuPojNwoULXXCogFBZLmeffXbI+1t9pqJFi7o+FNKmDBYtJ1Oqvrcs7+GHHw7+XgPvynTxlvIgdZqU+N///uf6ouo7qT11fdq3b1/wPg0aNAgu5UHmI/BOcrrQaiRMaw/V6fbejLpYqKOjEdvw0XD/bYhMKbr6kFu9enUwaNTF99Zbb00xGq52Rer+/PNPtx7JWy+nc1YpUyVLlnQBpIfR8NhpqcMVV1wRfM8PHjzYXZQnTpyYYjRcs7VInWaydT56qc8KGtUhb9OmTfA+yh7Qem9lEyB1O3bscB3vl156yf2sWa8rr7wyxQDmTTfd5LJdqEcQ2+B6kyZNgrOyr7/+uhu48A9gfv75524mXFlZiG7btm0u+0JZF17tBp2Xes/7640oS/Ckk05yWYNInd7DGrzo27dvsO30/lbKuVLPPd26dQsu5UHq7r///sB5550XzFzTrLcGLp588sngfXSt0ueAt5QHmYvAO8mDQ6WSqrOtzuDIkSNDAvIzzjgj0L59+5DbtF6J0fC0aQSxR48eIbdpjbxmwfxF1TTAoUwCRsNTN2zYMLcWyU8XklKlSgUeeOCB4G0KehgNjy2o0bk4e/bskNs7derksgr8S0q0LpnR8LS1bNnSzSz4aSmJPl9Vl8DzyiuvhBS2QmQqRKlsDG82VtTR1vnpH8BUXQcNEivbAKmv61YGxgcffJCiY67PUX0meDQg5xW2Qii9b/V5qPe1zk+dj34aHA4frHz33XdDClshsunTp7vrkn82VsG46hFo0M2jwfYSJUqELOVBZMpme+2110JuGzhwYKBgwYIhRdWU1h++lAeZg+3EktDs2bPtrbfecnshfvDBB+5L27EsXrw4eJ+jjjrKbd3wxhtv2Pz580Nu27FjB1s2+WhLsPfee8/mzp0bsi9y+FZhV1xxhdtW6N577w3epu0dtFUb242k3AfVT+2pPeT9W4UVKlTI7rjjDndOrlixwt1WqVIle+CBB9h7OszPP/9s7777rv3666/u5xw5ckQ8R7t37+7uozb1PPfcc67dtS8t/jsftYd0+Dka3p7aVuy8886ze+65J7hX+s0332wXXHCBrVy5kub0UXtqMsDfnrpt3759wdvy58/vPj+HDx9u3333nbtN+3n36dPHNm7cSHv6rFq1yiZMmOD2jk/tHL3//vvdnr6PP/548LannnrK8uXLx3aXPjq/vv/+e2vWrJnVr1/fbW+nLQJ17f7zzz+D92vTpo3VqlUrZOumyy+/3F3/ec+HUt9y0qRJ7vzzzk9da/zb2eXKlcudozqXZ86c6W7TlqwDBgzg/Azz999/h3xeRnvP33nnne7/2jrQ07dvX/dZun79ej5HM1smBfTIolauXOlGXjVSq0q7/tnvSLMwrVq1clu3MPIdmVL0NGOt6rBKL/OKgKhAndrYP4sgV111lRspHz9+fCb86x75VFFbazrVRieccEKwzoDWdSkVKrwwldKmdF+dp0hJswUdO3Z06+VUtdy/PlYFaiIVptLsob7WrVtHk0bw3HPPuWUjOu/q1KkT/BxVqr4+Q/1p0F4Wge6rLZyQkqrrak2n2kip+l4lfc3GaOYwvLK+PiN0X20xhpR0rda1SG2pWSwvbVcuueSSiJX1lbarbYW0FR4i0/tY13RlA3iUsh8pjVzrvnW9UiVupLRp0yZ3HiqDUhW1ld3iZbTo2hNeWf/HH39073ltHarMDYRSTQFlW+k9rP69+pnedUhZq8peDW831SLQOapzFYcWgXcSuueee9yHmL/oj96UeiO2bds25L66ECsNTVtgIZSqwaqj/e2336ZoGqWOK41fAbn3gaeLitZ9q/Pz0EMP0Zw+qqatdfBa8qB1xRogUhqZLgxvvfVW8AKizqR/cEjFqrTWu1q1am7NHUI98sgjrm20Jj6cClKFF6jTRVipaQrKWe8VSuniQ4cOdcWntJ2NzlmlnGodsgaG9P5WAKN9Zr31xkqZVOdS97vzzjs5PX3UZiqOpve8zkFdj5544gnXcdTghnet0ppu//VH6+NVj0SdSfafTUkDQEp/jrS14rx589xnqn8ATv8OCijVplTcj+733393RT3Dd86IlkauAc+77rqL93wE6gOp9kWkbdY0aaElEf6AUHVwtJZetRx0vuJfSgvXe1p9IxVK1fIGb/JC1ykNDKlauZbf+QvU6bNBxdZUSJktQw89Au9sTAGf9jpVcKKtQ7y1G+qEa5Y2vMDPF1984YqqzJkzJ+R21nxEHwHXFjfRqOCKAnNlDCjQVgfcvzYR/83QaB9pdf7CByRUREUdb63n1pfWemlAQzM6avtoHUz8S4Ggv3ZDOA1u6D1/7bXXus8JDWxo4AOh1KHR2kMF2fqc9CiwVlVobW8nGjxSh0bnqc5l/U4dTDKGUtLnot7zmpX1U/CtmRsFOhrMqFevnmtTzYy1a9fOndP+QksIpQF0tWE0Tz/9tHvPa+ZW67s10EY2RigFdxq01NfSpUtDgkJ9BoRvsahBC29/ZA/9psh0vdaArz/jMrzdNHipvpNmcdXP0nruH374gbd6GPXv1U5ar+3PrtTkhD5be/fu7X7WwKauXwq0VRNH9XJS+4xA5mKNdza1d+9ea9KkiQ0aNMg2bdrk1sWdf/75bu1M4cKF7bHHHrMnnnjC1q1bF3xMvXr17KqrrnLrEf1r7XLnzm3JwluHGes6ZK2piaZx48Zu3Xzt2rXt999/t549e7p19Qil9cYDBw5065COOeaYkN+pzfRvMm7cOLemW/UJevToYX/88YedeeaZ9s0337h1Sckkkeeo1sZOnTrVrZ3VGu6PPvrIfQbg3zWdXn2Lk08+2Tp27GhbtmwJOUePPvpotzZ2ypQpbq2n1nZq7bE+e/Xz3XffbW+//bY7xxFK159I7/nOnTtbsWLFXH0RnZeffvqpW8+pdbTly5e3hQsXuv8jvve81h5rrazaeOvWrTZ+/HhXdwD/mjx5srtm//LLL/bZZ5/ZWWed5c5F6dq1q7sOPfrooyHNpZoYOk9VLycZ+03p4dXGiHaOqt1UM2fIkCHu/DzuuOPcdf70008/xEea9V1zzTVWpUoVt67b/zlatmxZ149/5ZVX3M833nijzZs3zypUqODW048YMcKtm8dhksmBPQ4xpThqm6Drr7/epfN4ac6aPdCaWa39EKX4aL1MeIVypfspdSoZTZ061bWJv6JmapT+pJmZ8BlXpfeocnGy+/jjj0OWM6TlwgsvdDOE4Zo2bepmuxBwo9W33357zE2h7dVUnThS2jT7nUenpQ7KTtFnprf1ktbMKfvCn7LnfZYqfVefH8lO2QDpqdysbACtkQ1ff6glOprdRvpp+6VTTjklRRqv1slOmzaNJo1CS2uU+aOldf6t1JQZqFoD3m4kWo8caYtF1SNh6UPalP2jLItIS2+0vV2kZVGITtdyhXLvv/9+yO3KXNXtLMHLegi8syFdPPSGC1+74a3p9FJ8tHWDUs6+/vrrw3SkWa/gh9JzBgwYENP9lQZ54oknuoDR65yLCq1om7Bkv7gqiE4tFT+cUsmUDjVixIiQ25Wiz17n/9LFVW20ePHimNpUHUgFhUqR9KhjU6NGDVfUCqFUr0Gpugq4vQJqXrqeaG1sgQIFQjrdGtRUG7P2MODSbZV2G6vVq1e79Zza3sZPhdPY6zw+WtOpAeF77703uMRB1yr9u6gPgMg0MKG12gq8/dSGqnnh386qYcOGbmID8XnppZfcZ6Z/azudt9rOikAx/TSAeeqppwaL+3rb12rpGMXosh4C7yNEet48Wt+hdR9asxmuUqVKIZ0czYCHd3qSvTBNeio6a9ZQa44VgCvLQEUtNEOrjk6yU/E5Dez418SmRaPgKgTSp0+fwOeff+6KhuiCwij4f7SeOD0VnTVooWD9vPPOczOJOl/9gTj+pTVyKvTlfR5qL/Mbb7zRfZYqQPQyinQ+li5d2s3OqOOoAabwgkvJHrykJ2tK62jVCVc9B73nVZBKn6dr167N1GPNzkaPHu0+R1VYUQVTdb4qEEfqdO5pNluD8H5vvvmm+xzwBjI0+636IuE7GCA2akfVdlD/QNcz1RvQpMfEiRNpwjh4A5i6FmlwXgVpNYCkAQ5kPTn0n8OV5o7YaF9Drc/W+qIrr7wypsdoDZL239Vek0WLFg1Zx33RRRe5dbLeetGcOVnq72/rqlWrWs2aNW3kyJExtbXW0I8ePdqtQdSevc2bN6dN/991111nP/30ky1YsCCmNtGaT+11XqBAAatYsaJba6fzvkiRIjH9WyQDtafWt48dO9btDRuLH3/80a2bk0svvdSd4wilddrab3fHjh1u71jZv3+/1ahRw0499VS3Vlu0Jl57+VarVs1KlSplrVq1sk6dOrGO+/9pXfuHH35oS5YssTx58qR5mu3Zs8dOOeUU19Zq5+rVq7v1h8lWuyGaX3/91V599VVXd+V///ufO+dioTXKWr+ttbS65uvahNSphoOuP9rnWPUvPFrrffHFF7vaOd51jL7Tf1R7ZcaMGVa3bl274IILYj7N9Jjp06e7dfOqLVKuXDlO0X+XAKf7etK7d293zmqPedVvuOWWW9z7HlnQ4Y78Edsst7ZVibT/ZjRKfdasQZMmTQLbt28PplEqBS3WNNVkpXVwmoHRlkHIGC/7YtiwYenKOlAVY9bLRafPA21Np1lZJIbWaGsGJrxisXaE0KVy9uzZwduUZqp0UwQibqWoNP3+/fvH3DxaH6trE+n6oT788EOXhaF129rrXDPXfC5mriFDhrisDW/NrPphWr7n378b/1FlfF2LVE9EmVUvvPACzZOBTAB9biorTdXz0zNjrT6/dnx48MEHaf8sjsA7i9KHvi603n6wSoPcuHFjSOcvLdqTUx1GvYm1JYu2EFN6JP6zYsUKt9bdG5zw6CJSp04dmioBtEZWncdY126pKJD267z55puTvv11IdZabBVK8Q/EeVsCPv7440nfRuldnqN9T1WAUstClIb7888/Bz9jtU9v+DYrGqjU56g+j71UU+0rrZRUFVpCSup8ay38H3/8EXPzaJ/eli1b0pyBgFvCoHNTAY23jZKKfmqbOv9aYySerj+qK6JBuJo1a7rif1raE55+nuyfrRpgU8q90pu9pXUqKqslDuGF5xAbXXtUi0GTPhoA0gSQiv7FigHMIwOBdxbjdew0qq29+Z599tng71R0Kr2dGc2Sq5CS1tmGB5fJQHtwzpw5M8XtujAouFYbay2MghitSfZo5kUXEK2Vi0V4BdlkC2Z04R00aJC7aKhugD+rQiOxKlbVpUuXdGcdJHMBMF1wNQChPUyVAaAOjr/zp7Xb6fk8SJZzVLSeNVIlfA1Gaibhueeec50UBdNaW+h1FLXeWBka/vNXGRgauFSg7a8cq/NZ+6Hu2bMnkIwUCOocVE0LBczaw9x/rp1xxhmuRkOs9F7Xe57K2wFXAFWzhxqMCC+WqIBQg8WxSKb3fCR//vmnK5aqz4L07AP92WefucE2BT+qW4D/6HNTmRdqHw2ohw9U6rMy1sJzyX5+epmoXbt2deep6odoQMOjfczTm9nGAGbWR+CdhajIhD+15Mknn3SdRK+AhzcSG74FWFpv6mTe6kadY6Xc+zvH6kgrC0Dtq4BRgx3qqCsA9xfxUrEfpe6kVihNaam6qF988cWBZEgbV0EZzfb5jRkzxgXWrVq1Crz22msuLVIDGqpS6r+PApfwx0ai++qcVwVpVYxORkpr1JZAkydPduenqpVqtktt7NHnwVlnnZXm54G2d1K1aZ3vycLb8kcDb/5BIw1iqFCSRx0aFaBSZ1H+/vtv9zmsmW8VWlLWhSrDqg0vvfRSVxDIowwOdTqTMd1fAxCVK1d2QbcGhLX0IW/evCE7ZHz66acuSIxlyY7O899++y0wdOhQ0s3/n7YN1IBw+Pml97u2vUwtaFFbamY8mdNOFWjrmq40cS250/mp/lCslBVD1tt/NDGhTAxdl7SFlQYn1I9SZobfwoULXZ9T24mmRo9XfzbWQaTsSJ93GsRQ0VMNrmvQV9cgj4LwWDPbVIxSlcx1jqu4GrIuAu8sRJWG9Sbzgj+9AdW58e/bq6qv6sz495lMS6dOnVxapZe2nkwUGCuo9qc564LsrZNTUKP0KHWuNcihDqRHGQLqqPfq1SvF86oT7z1O+/0mS3VTjab6q+WrHRs0aBCyBEIdbV1EwmsS6LEKXlKb8dJ9NLPrD9qTkTo23gVY567OYc0uaBbM36FJ7fNAnyPe4zTDm2zvf63B9g+I6VzVLE14WykbSLd72RUKaPTevuaaa9y+6V5GgQY99Fma7JSSr/e8fzDXC3LUkfYHhGqz2rVrpzkopMyOZMxuUVCtTrU+85TWrBR9L+tNS8t0TXrqqadCHqPzUZku6rSH0wCd9pn3OuvJOCik9GedU7r++Ktk6/2s8yzWWVYtQ1PdgVGjRgWSmc4pZVppMFLZQbrmeBToaT283sd+HTt2dLO3ka45alcNLGsQP1mrmGuwUm2kiR1vBx0NWCjLMjxo9jLbou3y4H2G6D3ft2/fQ3L8yBgC7yxEs7JKXfRv++Ht2+vtve2NxNaqVSt4gU6NLjJKt1LnW535ZOkY+ukDSwFd+AeX7qc9uNXpUWqp9j0Pv4joA1IX8fCOumbJ1LFPtkJ1Opd0oYj0d+sCrRlBjeBqJlCj3krtDU8lDc++UECki5Aep/aO5bzObqL9zXrf6+Ksba3UEdd7X3ud+js0Su1XdoB/UEiZMxoUuu2229zjkpE64PrsVIEq77NQmRjhgYyoQzl8+PCoz6WZHs2Y+dOp8W+6ubKKvJRTtZHSc/1LdhS8hC/Z0aCQBjmTdVDI6zBri7/mzZu7AKZHjx6u4633rEdtE6nQpNJS/f0EfX6ow162bNlAu3bt0rUcLbsW9ytcuHDI7evXr3fvf2W8xUKfF927d3f/Psnq6aefdn1SnauaGPIPUHq0vCy8j6TzVeeitlv19w8eeughFyD269cvKQeFJk2a5PrhuiapLcOzBZRyruxBLc8Lz2zT+zqctm1UDQJltyjLBUcGAu8smiLpn/HTh1qjRo1SjMSq2m5q1Fk8+eSTXepJsuwrrYBDawv1YbVq1Sr3QaaZQ6XmaZ9tP81Uq229Tp9XRMl/EQkPiPR8uhCNGzcukN2p3ZSerA6L18nW7KsyMDRz4O8sq73r1q3rZri8NUqacVCav/8iovPbqzWg51cQpHQ1tWsy1iDQTKHaSQM+Sof02toLoHVR9a+h0x7TOkf9gbb/HFWbat2yigH5B+uSld7jSo30sge0Zk4zs/4MFf1OwY1//bYXHLZu3doFR1pnl1bqZHakc1CdO82kRFrrqrWc6nh7561mshT0+Ncpvvfee8HBH/+gkD5HknVQSNQOen/762aMHz/evb+9wUl9xipbLbVMC73/1T/QjLkKB+LfyuTKBAoPRvRZqkG21Aql6Zy84447XDCj4DBaXZPsSueTBspeffXVwNlnnx2csPAmhu6+++6Q+3///fdugDP889OfMq0+qwJxLZNIxkEhtalmpXUt8moJKRtLfVU/9a+UZdmnT5+Q2zXgO2PGjJA213teAXkyp+ofqQi8swAF2fpg8iiNz58iqRlYdczVAfJoJFYzhBpFDKd1jVqnqFSfZFvbrQ7gscce60apNbKqDzAFjPpw0oXYP2OlDy1/AKMLjW5TJ/zXX3+N+Pxq22QppuQVSNKaba3B1LKH559/PtjG+t6jEVx1aPwXW3XINcMdfhHxqFOkjrpXVTrZqPOngQnN8is9X50cBcxeIK0Oijrh/nXKbdq0cffTV7RZ8vSsYzzSqCOsOgKxUoCtmS/N3HhtrgwCBSkamNDvNcOo973/3PXXNdDnbzJmYehzU4OQygjS4JBmY5Xt41HKvgaJ/RlGOj/1ntf6+EjU3irElqyDQv5zTG2kwbZwWjfvv/7rGq42Te19reyOZDtH9fcqS0UDvspAU8aAd2326uEoUyj8nNZ17M4770zxfBrk0Gy4sjAUXPrrvSQTBX9qA2WvqE5DpImh8KrlGqhQu0b6DPX+rXSOJiP1k5RZqf66/5xSv19tqcE2P53TWuuta0806v8rBZ3idEcmAu8sQCNX/rWv+oDShdZLkRRdCDQ67qXnqLOjWUdtMxRe6EsfmqqGnozpe+qcKOBWkB2eDq0RbHW4vQ6KV3hNmQH6ENPMqwKg8FT1ZKa1rgr+NFLrrzasmVetP/RmrFQoRYND3gyD1oHpPFQhK/0O/9Lsv9Zx6v2sjrc6i/60XHV2/Fv+aeBDnXC1p2ZrNZimi3cypumJOnzqrKQn40TV9jWYpmUSXjtr6YnOa52zKr7k/Q7/BofqfKsDqKrk3uellpH40yC9DCGvQ63PTn2G6j2vgTqEdr6VaaFBH52P3nmpwcrwbDRdi5QR4KfstkiD7MlMGWwqOKlinBpA1/VIy7+8fo+uO5GK++nc1GeqfyZb1zb1pzTQFL5eORl51/1PPvkkxe/UxuHp9xrQ1Fr4ZBv8iYVXbC5SpXf115VN5Z/M0XmppaT+IqDIXgi8s9Abc8qUKcHb1Mnxp0h665a0Nsbj/5BT6kmyFfqKREGztg/T6KsCFT8FhQpsvBkzfdgp9UkdIl1IsvNMYbyuvvpqN7uqWYXwmYNKlSq5kW7vXNTsjc5RVYJVm0baxi2Zad2bAmd17lRwSh0bbybWozVwGhn3Bn8UaCpIVJqeUnPZS/bfzkr4TgWp0Weo1sX7K5J7n6nJsgQnFnoPax2nslkUsGgA019IyUuD1JZr/oJVWjerqvC6vwYxkfJ8VYAYvp2VamXky5cvRUaQPhO0NArRqZinMjD8SxU0AKTru/8zNZbK5Ppc1bVM62+zKw2WRVoqo+u4ZrG1pMx/vVbwpywCZbBEq52RjEtv4qXdMfTZqfb202SZBivDq5Yzk529EXgfYkov0f7R4TNW4VUgNQOjFElvdNxbt6Q0vUj0OAXfyRjMaDBCnRv/SLVGtTWbFd7ZUcVXBYXMaken4M6fLaG0+0gFktRR0QXYO+90UdFMpEa+k3G9djRKzVfavdYLewNnCnJatGjhUpz9My+a1VLgrVQyROZ1VhQkpmdPdA1uJmu6Y2p0LiqQUXtq8Ewz2Zph1XvbvxRHNAuuYHH16tXBDqIqEytLg0GhlPSZqHaMtqWaru/6vf6v81rBj65PbAcUuS09KoSqdgqnHTeUlZGeyuSqXRAeEGUnWqqkWj/K7Am/XRMUGgRW1qUG25Th4l2PdC7qtkh1A7Q8R9utJiNtceoVRPTTZ6K2BVO/XdfwRx99NBhAq58arbCfzuXUqpYj+yHwPsSUJupVdfRToK10SKWIe3Qx1pvYS4MkjSeUOoi6+OrDToGNOoT+9HLNLGptjZ8CmzJlyriZRYTSukulOGkmVm3kX8+p/WAj7Wmu9Yhqe6Ski61mErzZA7Wrf92WLtRayxVeTVtBjzqL/roPSNlZ0fY26dnnXQOeSuNF6Oehiv1oUEKDGf7gOVIdEZ3LKt4XaSYsmUQruKUsCi0l0QCvBtl1zVZBKhVH08yiKrmrWJrSor3BNc3Q6tqvzwcFjcm+fVU41bZR9p+3S4MoBVpB4fLlyyNuDeifCdfj/UUrk43qhkSqs6LU+p49ewZ/VsallvH4i3lqxwz/8jxPshWc8yibVOehik36B2sUNKsvqu0nlTmpfryu7f4sK/UFIhX2U1uOHDky6vp4ZD8E3oeBty9feHVHfQgq0PZSxb0KxaT0pEzN1xotpT/71xJqRlEziN4oo1eU7u233w5W11WxEFWHjFY8LRmpLTRgoU6fCsx523upI+51WJQhoAuL/0Kt2zRboNH0ZO7YhFMnRcsaNFChzrXOR80eqEMYnorbu3dvl4Lm32dej1fnnTWd0XmDGeHbsaT174JA8NqiYFADZ6JMKg32+DuTXvaFgkg/zY7rsyJZi09pBlCBoP89KxrA0IClzkl1pL3tO9VeqoyvlGe1pa5BKqiozwMvS0sZRuFbhuHfDAstswmv3Kz20tpYBYZ+KkSnAfjsPIOdlvDzUstAtGuIJnAUGCrbUgVNdf6F94N69erlBoG8AXbtDKP29NcdSWYaPNP56F2DlKUimv0Pn4DwdijwqpHrM1d9pUiF/ZBcCLwzkdJuNdqo6prac9N7k3r78imtJzwdUm9U/z6edBZT0mi2ZgyUOeCngQwNaAwdOjR4m0Z61aHUGi511JNlL/P00IVZa7P9WQDq2KiwV4cOHYK3qdiHRsTVGfJS+rTOlnP0PxqAUPqeCtBoD16/SLMHWnundrz//vsz+V/5yKSOoWo16BzVOedfwqAOjWa9/AUmkTqdb5pRVUdc56PXwfb22w7folI/a91s+LZMyTw7ozbUoJoKUPk/LzUTltYWnx4NWijF3L9tKFLS8jsNBkeiGVqvgr6CSg0C6xqv7KxkpUkb9Y382Wq6bus9rIF0FZTVe17ZVuprhtdh0eBPeFE1ze6GZ2gmC/UXvYkbUYaVJnO0LEe1b7yJCH2WKvMynCaH/NlB2nJN7/vwTA0kFwLvTKKOjUa39KGnQFCzhSqY4s1iqWiNOo3+vflUGE3FrDQynowVydNDQY0uut5aQ48uEArI/bMxamtv70SkpMBF56I/xcyrCqs29q9P1IVGnRsFisxyR6blDeHbgMnKlSvd7EH4dliqyqstmyio8h8F2Jqd0XtZ55pmEFWgStta+dtJWS41atRIc/BHs0DM2vy3Y4bOw/AZa7V3+HIStauWn2hHCPxHn32aNdR72hsM9ld4VwCjDrvWdIav3dSMudrUK0yZbKKlKWsmVgMXyhLyBsi1llZZGcpW0w4Qeq9rRlb9JA3+aJ2tsgzU9krj7d+/f9IPBCsAVIAtGhjX0gatL1ZNET9NCOnz00/vdxWto0jiv5566qmQLFT1NzWwrvPNX3VcBSc18BaeaaF13uGFEv0FK5GcCLwzgS7I4euPdIHWSKR/Zksz23pT64KiD0VdVHQxYQbxX2oHzSoorU8dRV2AvXWvGpjQBUUp0eEXb227prV0SEnF5rSmULPW/pQ0pUfqQhzeKWrVqlWaVWGRso01qh2+dtsbuIhU3I+gO5Q6Ncq2UOVn0Xk5YMAA1+HxZxLoc9W/U0E4PU7/DmpzpVbTzv/umKF21HpkP285idJN/bRmkR0KUlI2m+oGeK644go3UKRZWtUfaNCggeukKyVa1yudfxrYUEd82LBhSfmxqRoXFStWTDHjp+u6MqxUl0X1G7wlZMp4UeEvLbnTFowKCJUZqPWz3paC6idooIP39n99ILWv0u5FmQAa7FGf1J+yr9lXfQ74615oEFj91PB09WSl/rjWa3vXaw0Mqb+uQR5tqepRSr4yhsILfiobI1pBZCQvAu9M4FV7Dg9iNKOorVq8iuZ6U+vNqy0vVEgt1u1xkoU+tHTBVdqUghkNTiio9j4EdWHR7I3WfPupmIr2RWcA4z9qC6Xqaj2x9j9VJ0cpzrpgeClU6iz60yf9Fc3Hjh17CP7Fj6z2fOmll9ygkNpNg0L+qvpax6VOd/gIuGYTtUaMquWx03tfGQFKN9dgpX//eP9OBeFr4rW2Vp8felz4Z0Qy0NImZQqoMrl/dwevkKeK/YRTWq8GOb3PBUS3YMECd/3x0nJ1PVcwo0whb69zzXb713Ij4LZH1LXIT0tzYq2SrQBbBeu0hhaRaYnd8ccfHxJAt2vXzg36+AcotI2VAnINrmvQQ/0DZmRTZrfofFPxWY8yWdSX92f96XNW/X4Nruu+qtOiibVoOxogeRF4Z5AuuhqJ1VZfXnq43nS62Krj56dRSN0evsUVUlJHReuOvXRItV2zZs1coKMRXI+CcXWskbrnn3/eBS/eLJdSeVU46ZJLLgneR6O1ulh7tQg806ZNY7/jMNq+ToMX6mRrFkEdF52vXkaGKpdqVDxScKP7h6ehJzPNaqvToiU54YOPeq+r86LzV4MdChr1Gaq9dz0KuP0DQ5pV09Y5Gljyr89LJhqUVCda24Np8EEBon8pibJewgcwRG2sdYlU1g6lNfCaYQ0fwFFWhqpDR1saptlE1RdJ5jXx4fQe1nvevwOJ0nS9LdR0bdLkhT4T/Nd60Zp4LXtQtgEz3NF5tUO8lHNvEEiDxP4aOKJ/B30eaJAu/NqPf2lQQpkXHp17et/7a+CIsqv0flc76zHJOOCLtBF4Z4A6f1rHrf+rk33NNdcEf6e0caWa+We99QGn0UWql6ZNHT9VgtXstkbCFRDq4hDegVFav9YkMSObOgXd2m9XNFOgEVwV/fDPbCno0e3+izVS0ii3Ahml4PpnsnUhVqaFf1Rca+tYCx+dUiA1W610PA0E6T3vzxJQx8bfudFAnNpeMwv+GQg/dSy15s6bdUw2CqY1G+PfZ1YZVRqw8NJP9TmqQUz/AEaybxUUic7Fm2++2Q3+aKBN554y1LysNb231dYaGPKMHj3apaeqo67lUAyypaTzzj9goeUNSh/XjKz+r88BDRipbZWNpbW2ujYpTV0DxAxk/Gfu3LluNlv9z2+++SYkbVzFUP1p/SqUFl4DB6HUD1I2iz9bwNshR9mU/gmJ8Bo4QCwIvOOgNCmtLVaBFK9zp6IqCgC9GRa9GZWiq7WzGmlUyq4C8RtvvDGel8y2NMOiDqIqF6v4kZcernQntZ/ScrUm0dvLXDRrq5Q+jx7PYEYgJAgMX0NcpUoVV7VcnUEVqfJXd/dXitZ2Gaq8zWxCdF999ZULYsK3A9TAhm73Cv6pDdVZT9YiSqnRzIo63Rqg9DKDNAumATatofN30DUD6wWDqt2g4kFaouPfEx2BkPewApZwGtxQe3u0VlYDGP6ZR4TSIKSu895npM5Vpen79+fVuejPHtBsrbILPvzwQ4qkRhFpwELXfLWZl5Wlz09V4vbWciOU+kqaoFBxOaWWq8aAdnXxL2tQNqB/MFhBpWoOaJcdRK4toveyltvo3PPvmKHMVi119AbdRDUedH0C0oPAO0bq+GkUW51Dfcipg62f/TQzq+IpXjCu/To1Qqv76iKjdZ2M1Ia2qYp3aVsGdWSUnuPNtqqdlCql9PLwfweN7iq1HylnZ3Rx0IyBRmL9M9caKFK2hQrX+Ge0dPFWpWh/ihkzXoGQ2Sudo1qz7Q3uaLZAo99KzfPT+17vdY2Ee77//vuQQaNkp2JoSn9WO2nmRQNCfiqS5t++SgOWyiZSwSrN0KojqTR+/EuDF5rl8qeMqzCV3v/hs1qa7Va7+89HDQx7mTBISSnQ4dXwlRKt9vW2AtPnrtJLGWCLTH0mDY6rVou/7kr4gEWkJTn6vWa8EUrtWbt2bRdE+7PWVAXeXwxVqc46V7X1mscbGMJ/VDVfkzmqy6JrtvqfmiRTFpbXN9KAkK5Z/q3VtARFE27h+8wDqSHwjoFGXjWypW2C1HFUB/vEE09Msb5Do+K6UCvN0U8ddgLu/+jiq1mrm266ybWrN7uq9fIKDr3ZWH0Q6qKhFKqvv/7afbhdfPHF7jEEh//RuaWUXRWeUvq4ghbNZmnU1hugUCdRGQS6j9fe6jDq4hJeGR7/nqNqF6Xoq4N41llnuW3UvEE1Df7oM8Af3KgDpMCGPTpT0megUsMVPCvtedmyZW5AQ0G2v7q22l0ZF/7tq5RZoCrQGtCgYOJ/9B5Xmr4G2jSw61WCVoqkZr6UVuqn9bI6PwlkItPnYXgxRM16+avoiz4/NcuoKvv+YEbp0WwDGqp3795uuY1SyBWgqI28gZ9IAxYff/yxu2Zp9lZtTKGv/+izz7t2e+u1/Rksohojus77azToOqaBS87NyNQuGgBWf8lfE8RbeucvBKiaDfps9We7aXAYSA8C7zTccMMNboZGVTf9wZ7WFOtCEt7J1iyY3sDh+0sjlLZWUScwvDKpAnKlQnttre1DdNHWWiVdpDVSS9AdShdZdb51kfAP8CgjQ2mRXkdHMzW6KKsugQIbDRK1b98+RWcz2elCq0wLBd1ee2pWRkGOlwatVEl1DLUOUZ1DpaAqJZWlJKF0bqljo4EKzVzr/e0viqYMofBqxlqzGL71Df4zY8YMN3utjCCdd+qQK5tKAY43+6V1sHqv676eBx980H22IuXSHA1I6nqu64zOUW8wSIO+uv6EX3OUJaT29GNQKJQGyzQ46S0J0eCvUnXVl/LaytsBRrOMotReDXQqAArfpSDZqRCqv1in3uPqa4ZvC6jPAi3R03ntDXo2btyYATcftZl2cPAPYqo/qnXx4ct29JmggWLRwIe2XdXnBRAvAu80aMsgvSHD9+fzLr7+9TOiC4oCdb2REZ23pY22XvDztrVK1n1O46FOoTrU6tT4KeDRSLcGOfwj4grIdQGn4mZkGgzSez58yy+lnGqGVlkvoswCVdXX+aqUPw0KsTb+P1o7rBlupUWqrdR24cW8lMqvgEcpe34K1Lt27RrnOyJ7U60Qdbj9nXBddzT7dfXVV7ufdR4qgFRQo0EkpaVqwE3FKPEvrdVUvQZlCamt1DbapUDtpEE3DRopS0Mdb/91yluf7F9SgpQUYId/hqr2jT5bvewM0VaMCgyROmVeaHDNW/IUaUZWVN9FKdLhfSv8R1mVGtz17zykbEr16SOdx/7dX5Q5xHInZASBd4SRMP9Iq4IajXiHrzUWVTVWKrRXKRYpaSRbW/soAFT6mIrQeVSlXNVi//rrr5DHPProoyn2oMS/lOqsQR1lAvjPU11I1KHROjo/b69zfwVupE0X2/CiKQpulCp51VVX0YSpUMdEQYlmZP0ZLdpvW+/r8Jks7eurwl9+LM2JTp+hek+r0rOfZgv9S3VE2RiaQdTMDZktoUaMGOECaA0A+7ex0yCRbvNSyV955RXX3lpqpqJUytJgUCiUBnrCdxFQPYZIVfP12erPDNLnhQY72P0hdFAovPaCBtJVJV9V9qPNyPoHNFWLCNFpwE0Za172hTIy1JbhWZj6vNUApjKxgEQg8P5/CkzU0VbHRYWTbr311uCaGKU86natMw6n2UR9GLJ+JiUVl1MHpn///u4CoQuuUqK9dvS2tAlP21EnSGm7mo1AIGT9m9LDVSlbAxYKYvydbF1INFMTnvKorAzte4r0zdbqYhueuaK19PosmDlzJs0ZhQrRaXbbv72iKOBWer6q6/upOJgCG2YQY6frjoKV8OuOiqX5l+ogOm/XAc0ihtNMrT+DSNsLKdhWwcovv/ySZv1/Os+0jlsDGHoPK9tPWVWi27W8xL9rhqiQquo7+JGmH0rnWPiMrH8g3b9tmPpVKqqG6JRB9cILL4TcpuWgyhzSAJync+fOLnvNPxAn0bauBOJB4P3/s7IqojJ8+HAXDCoFV+s5/VWhtSZWwWD4BUIp0+EjZPiXOjX+dEhdpJVSpkIWXkqutg9JbU9e/EupuLpIeNkV6syo86Kqryq04r+QhFfbVkeILYPST4NvOlfDZ1/VAfdvZ4dQWg+v89JLew7frkVBudc596j+QLTqxoi+VCd8Zw1vqQ7XpNgHh8N3IvAyBXQ7WQKRqR+koFCFZDWzrT6UBic0eaH1xUqH1vu5aNGiLuPNu95rVlyfqYMHD+ZtnYZoA+lKiVaWUPggMYXoolPWjwaBwlPEdf6WKFEiODikCuaKBcILUwKJROD9/wVUvDeaClJobYy2DfBXLfXWKKrjiNhoZFbFU/yUEqWRXAXc/guJ9pdGdDoXte1a+Ayiitf4q8LqQqJZ8fBZBqSflkAos0BVuBGZBim1ZETr3/0p5FqPqEEgbz2iR51IrTf2Vy1HfKIt1VEQxAxi7LTMQbsW+Ge5VLBSSyUQfUZW13edf/4Cs7ruqAiql0qunUqU5aaUXq1FVnaglpyxlCRtKpQYaSBdg5QaFPJX4FYGFvVFovMq6P/vf/8LuV3nqwYq77///uBtmoDTtqxAZsnWgbeC5FhmUrVNkD7EVNFQF9vrr78+OIvopzWKGs3lohFIEUxrsEJrivwXYaXqRVoLp4uvf0RRj9HMA/4THjg//vjjLgUqUkCuQNujGQW1L6PfifHMM8+4ziX1BkKpk6dtapQZpMBFA5UaBPLe/956RH9hP4+Wmmi/1PB9ppE+0ZbqIP3ZRKoArwEhDQhrEEnLePxVj5GS3vcKAMP7Smo/taeKfHntq76TghkyhdKnV69e7vru7SUtynLRbepfsaQkfXt1axnpDz/8EHK7MjBU4JPCkzhUsnXgrXUv/rWtWsetzqLWwfkLV1x++eVujZIKJ2kNp58+5LxOt2Z0WHccSmu3lZqjNTTquOgDzBuJVXCt9XPKFvDTBYOq5ZGpY6IAWx0adaq9NcY6L3Wbqu76vfvuuy4Tw4+LceIogPSvo8e/lAWgDovXIVRaqapqn3766cGBSc12RSvsxzmaGPp8UEo/M9wZ88ADD7jPV1WMb926NQPB6RiwCE8b1x7HakuWN2WcMjA1SKmsQC3P0Ta2Ku6nyY4lS5Yk4BWSi2oQaDmEVxtDxWjVb1VaP3VbcKhk68Dbq1Ko7WwUaGsthy6wmoXRyNddd93l7qdRbl0odD8/dbi1Tjm80AL+3X9TwbaCbn/HWoWTdDHWRUHtpo64Zr40YKGK8Rr51swYM4gpZwEVVGsbEBVR+/XXX935qbVb3rpuVdavWLFicF9ur1CNf6sL4FBQgBK+VZA6hvpc1YyXR9utRdqiBchKNDur7A0VBEPs1J9SdsC6deuCt6nwl/oA4ftLIz6aodUgvPqoKpxIde2Mpe+r7oCWPqj2gGICJtNwqGXrwNtfpVCzrP5qpJop0FrjcePGuZ+VXq6ZQ6VMa9ZRFxSllUeqZJ7MvCqb3lZq4enPmslSQRBlFXij3xqtVVvr/qqwrQ8//EcDFBrAUPGP559/PqRpbrjhBrf8QWuU1JaqZKw0s3vuuccF4lq35O0rDWQGnXfaUsU/S629dzXwFk6Btr+omtL3lHVEJxxZnVLLtaZWRSqRvgELXYfUl9IEh4LE8EE5ZIwyWpisSAz1PzVBpOV7kZaUApkt2wfeXpVCzRyGp+PddNNNrqiKt2ZR+6JqRFF7Tmv/SQLElGtk/FusKHNAgxXhFwSlkauqsZ/u4635wn/LGN544w33/dChQ92Itn9rC9GFQVkbkydPDq791uO0rvOll15yqWhAZlDKuDIqFIzo81MpeV7tAa94moJyP31uhm8VBBwJ1D/QcgmlnSJ9Axa6dtWsWdMNrGsPaQBAZDn0H8vmXnnlFbv11lvt22+/tWrVqgVvnz17ttWrV882btxoRYsWPazHeCR45plnbPz48fbll1+6n//66y+rXLmyPfDAA9a9e/fg/T788ENr06aN7dy58zAebdZXtmxZe/755+2KK66wf/75x8455xx32wcffBByv6pVq1rbtm2tW7duh+1YkTz27NljEydOtB9++MF+/fVX9/mp93KjRo2sRo0a9tZbb9nu3bvtjDPOsBNPPNEmT55sxx57rG3atMnOPfdc69Wrl3Xo0OFw/xlAus2fP99+++03u/rqq2m9GKkLed5551nFihXt7bffpt0AIBU5LQl07NjRBdx9+/YNub1AgQKWM2dO94WUF9PXXnvNDVgsWbLE3bZu3TorWbJk8D7Fixd3nezevXvbJ5984m5TADly5Ehr2bIlTRpm1apVrn3k4MGDtmHDBitVqpT7+aijjrJnn33WPvroIzdw4dH9//zzTzfAARwKy5cvt3bt2tmQIUNs+PDhVqhQIStTpoz7PBg9erTNmjXL8ufPb++9954tW7bMTj31VLvyyitdIH7ZZZfZjTfeyD8UjkgaWCLoTp8cOXLYc889Z2PHjnWTGQCA6JJixltmzpxpDRs2tEceecQFiwcOHLD27dvb9u3bQwId/GfatGn21FNPuf9ffPHFtnXrVjv77LPthRdeCN5n//79duaZZ7rOes2aNV0wefLJJ7uLcOHChWnO//fNN99Y06ZNrWDBgnbvvffapZdeapUqVbIVK1ZYhQoVgu3UunVrmzJlij366KN21llnueBHAx5ffPGFHX300bQnDgkNuGkAbcuWLW6A0nP99de7gbgFCxa4Actt27a5LBj9/8ILL3TnLIDko8+GpUuXuqwBJjMAIMkDb7nqqqtcJ/G4445zM7oXXHCB61xqRgf/zq4q7f744493o9ie7777zgXgmu1SOpmCQgWImqUVzdJecsklLiVVqftVqlShOSPQIM+wYcPc7IDS9Pfu3etSeU866aTgfVauXOlmEJVyrjRenaOdO3d2M4zAoeItI9Eg5X333Re8fe3atXbKKae47AxlEgGA99nw7rvv2h133BHsGwAAkjjw9oKaQYMG2TXXXGNFihQ53IeUJSjt+emnn7aBAwe6dZrlypWzESNGWJMmTULup4Bc65C1xltp5uqQay2ngsLmzZvbvn377LPPPjtsf0dWonRcrZPNlSuXS7097bTTgr9TtoXWxes8zJMnj5sp6Nq1a3DAokePHi6tV2m8xxxzzGH8K5DMdH4+9thj9vPPP1uxYsWCt2vJzssvv+zWwur8BgAAQNqSanFz+fLl7dNPP7Wbb76ZoNtHAbSCRK3PUvCsYFprtNXh9s+GKyhXp3vNmjUuFbV///52//33u+wBFV5TsKkR72SmttCIv5Yx5M2b1w32VK9e3aZPnx68j4KVEiVKuKUPKkyl+5x++unWqlUrl6rnFapT+wKHy1133eXO0549e4bcrkEinc8E3QAAALFLqhlvpKT1w1prrMJfmtVS5fe7777bVTbWrLcqanv3K126tAu6lQYtOnX8KekK4PU8EyZMSNqm1tIFrYFXTQFVetbaV6Xfeyn7ni5dutgff/wRrAKrdtfaOA165M6d21WO7tSpkwvENWAEHA4ff/yxtWjRwhYuXMj6bQAAgAwgTzDJqSiaCn7JbbfdZu+//75LL1V6tL9AioqmeenmHn/QLf369Uv6AmAadFA7KkVcAxeaLVSl5/CK+uvXr3eziR4VrdOXR9uHac2cgnfgcFFRRS3LURFACqcBAADEj8A7yamitoJArX2/6aab7KeffgoGezt27HDBo6pwb9682e0jnVplbaVWJzutl9dyBhVRUzVoVSj3AhbNcCtrwBvI0NY10WhQQ+vAgcNN2RcAAADIGFLN4ba50uzqvHnzgkG31nQrEFfQqLRoxEZF6rQGdujQoW7m208VylX5XdWi58yZY3Xq1KFZAQAAgCRA4A1XRK1WrVpu7bZmtZVirsBRqdBjxoxha5B0UJaAKpifcMIJrtCc2lC33XLLLa6CuaqVAwAAAEguBN4IBt+PPPKIffPNNy5YbNeundunN3wdN9K2aNEit657y5YtboswbQum9tT+3VSCBgAAAJIPgTeQCfbu3Wuff/65m+2uXbu2mwEHAAAAkJwIvAEAAAAAyET/7RcFAAAAAAASjsAbAAAAAIBMROANAAAAAEAmIvAGAAAAACATEXgDAAAAAJCJCLwBAAAAAMhEBN4AAAAAAGQiAm8AAAAAADIRgTcAAAAAAJmIwBsAAAAAgExE4A0AAAAAQCYi8AYAAAAAIBMReAMAAAAAYJnn/wAQuJ90aRDO+QAAAABJRU5ErkJggg==", + "text/plain": [ + "
" + ] + }, + "metadata": {}, + "output_type": "display_data" + } + ], + "source": [ + "n = 100_000 if 100_000 in sizes else sizes[-1]\n", + "cases = sorted({c for c, size in batch if size == n and (c, n) in prims})\n", + "\n", + "fig, ax = plt.subplots(figsize=(10, 4.5))\n", + "width = 0.4\n", + "positions = range(len(cases))\n", + "ax.bar([p - width / 2 for p in positions], [batch[(c, n)][\"speedup\"] for c in cases],\n", + " width, label=\"vs scipy.stats\")\n", + "ax.bar([p + width / 2 for p in positions], [prims[(c, n)][\"speedup\"] for c in cases],\n", + " width, label=\"vs numpy / scipy.special\")\n", + "ax.axhline(1.0, color=\"black\", linewidth=1)\n", + "ax.set_xticks(list(positions))\n", + "ax.set_xticklabels(cases, rotation=30, ha=\"right\")\n", + "ax.set_ylabel(\"speedup (>1 means fastdist is faster)\")\n", + "ax.set_title(f\"fastdist against both baselines, n = {n:,}\")\n", + "ax.legend()\n", + "fig.tight_layout()\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "id": "9fee72dc", + "metadata": {}, + "source": [ + "The gap between the two bars is the cost of SciPy's distribution layer, not a difference in the\n", + "mathematics. Where the second bar sits below 1, hand-written numpy beats this library." + ] + }, + { + "cell_type": "code", + "execution_count": 4, + "id": "2ade9262", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-12T01:36:20.707164Z", + "iopub.status.busy": "2026-09-12T01:36:20.706952Z", + "iopub.status.idle": "2026-09-12T01:36:20.712339Z", + "shell.execute_reply": "2026-09-12T01:36:20.711684Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "GPU path against this library's own CPU path (>1 means the GPU wins)\n", + "\n", + " case n gpu cpu speedup\n", + " exponential_pdf 1,000 25.3us 4.3us 0.17x\n", + " normal_cdf 1,000 22.8us 7.1us 0.31x\n", + " normal_logpdf 1,000 28.5us 2.4us 0.08x\n", + " normal_pdf 1,000 23.9us 6.7us 0.28x\n", + " uniform_pdf 1,000 24.7us 2.3us 0.09x\n", + " exponential_pdf 100,000 187.1us 313.5us 1.68x\n", + " normal_cdf 100,000 210.3us 896.4us 4.26x\n", + " normal_logpdf 100,000 201.4us 77.9us 0.39x\n", + " normal_pdf 100,000 192.9us 420.0us 2.18x\n", + " uniform_pdf 100,000 186.1us 94.5us 0.51x\n", + " exponential_pdf 1,000,000 1884.9us 3745.5us 1.99x\n", + " normal_cdf 1,000,000 2145.7us 9828.0us 4.58x\n", + " normal_logpdf 1,000,000 1964.0us 1309.6us 0.67x\n", + " normal_pdf 1,000,000 1994.5us 4949.3us 2.48x\n", + " uniform_pdf 1,000,000 1867.4us 1415.4us 0.76x\n", + "\n", + "The GPU must be warmed before timing. An idle NVIDIA card drops its clocks and\n", + "downtrains its PCIe link, which made the same call measure 4.1x slower; run.py now\n", + "runs sustained GPU work first.\n" + ] + } + ], + "source": [ + "cuda = by_group(\"cuda\")\n", + "if not cuda:\n", + " print(\"This report came from a CPU-only build, so there are no GPU numbers.\")\n", + "else:\n", + " print(\"GPU path against this library's own CPU path (>1 means the GPU wins)\\n\")\n", + " print(f\" {'case':<18} {'n':>11} {'gpu':>10} {'cpu':>10} {'speedup':>9}\")\n", + " for case, n in sorted(cuda, key=lambda k: (k[1], k[0])):\n", + " r = cuda[(case, n)]\n", + " print(f\" {case:<18} {n:>11,} {r['fastdist_s'] * 1e6:9.1f}us {r['baseline_s'] * 1e6:9.1f}us\"\n", + " f\" {r['speedup']:8.2f}x\")\n", + " print(\"\\nThe GPU must be warmed before timing. An idle NVIDIA card drops its clocks and\")\n", + " print(\"downtrains its PCIe link, which made the same call measure 4.1x slower; run.py now\")\n", + " print(\"runs sustained GPU work first.\")" + ] + }, + { + "cell_type": "markdown", + "id": "c11a3ffd", + "metadata": {}, + "source": [ + "## Where this library is and is not the right tool\n", + "\n", + "Read off the tables above rather than from memory, but as of this report:\n", + "\n", + "- **Good fit:** evaluating PDFs/CDFs over arrays when you would otherwise call `scipy.stats`, and\n", + " scalar calls in a Python loop, where fastdist's thinner binding layer is worth much more than the\n", + " arithmetic.\n", + "- **No advantage:** against hand-written vectorised numpy, where the two are close and numpy\n", + " sometimes wins.\n", + "- **Wrong tool:** bulk random sampling. Every variate crosses the Python/C++ boundary individually,\n", + " so numpy is one to two orders of magnitude faster. See the `sample` group in the report.\n", + "\n", + "Reproduce everything here with `python benchmarks/run.py`, compare two runs with\n", + "`python benchmarks/compare.py --latest`, and see `BENCHMARKS.md` for the full log." + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.14.2" + } + }, + "nbformat": 4, + "nbformat_minor": 5 +} diff --git a/requirements-dev.txt b/requirements-dev.txt index 93b4e19..7ff56ce 100644 --- a/requirements-dev.txt +++ b/requirements-dev.txt @@ -14,3 +14,7 @@ nvidia-ml-py twine mypy pybind11-stubgen +matplotlib +jupyter +nbformat +scipy From b4975294152ec3a6b0c6ca93c39f2fbe2fca9391 Mon Sep 17 00:00:00 2001 From: ghosteau Date: Fri, 11 Sep 2026 21:38:30 -0400 Subject: [PATCH 19/20] Document the validation contract, and correct the docstrings that predate it Adds a README section stating the rule at each layer: parameters are refused at construction (ValueError for a bad value, TypeError for a bad type, non-finite included), a non-finite input yields nan rather than raising, CDFs are clamped to [0, 1], and the C++ API signals invalid parameters by return value because it has no exceptions. Every example in the section was run and its output checked before being written down. The Raises sections in bernoulli, exponential and uniform documented only the range conditions, which was already incomplete for finiteness once the checks went in. Thirteen occurrences updated. Co-Authored-By: Claude Opus 5 --- README.md | 34 ++++++++++++++++++++ python/fastdist/distributions/bernoulli.py | 20 ++++++------ python/fastdist/distributions/exponential.py | 4 +-- python/fastdist/distributions/uniform.py | 2 +- 4 files changed, 47 insertions(+), 13 deletions(-) diff --git a/README.md b/README.md index 2d59db8..629b4cc 100644 --- a/README.md +++ b/README.md @@ -86,6 +86,40 @@ logit, Euclidean/Manhattan distance, cosine similarity, coefficient of variation --- +## Validation and error handling + +One rule at each layer, the same for every distribution. + +**Parameters are checked when you construct a distribution.** A value that cannot describe a +distribution raises `ValueError`; the wrong type raises `TypeError`. Nothing is constructed, so no +later call can quietly return nonsense. Non-finite parameters are refused too -- `nan` passes every +range comparison, so it is rejected explicitly. + +```python +Normal(0.0, -1.0) # ValueError: sigma must be positive +Normal(0.0, float("nan")) # ValueError: sigma must be finite +Normal(0.0, "1.0") # TypeError: sigma must be a real number +``` + +**A non-finite input is not an error.** `x = nan` or `+/-inf` yields `nan`, matching the C++ core and +numpy's elementwise behaviour, so one bad value in an array does not abort the whole call. A string +is still a `TypeError`, even though numpy would happily read `"0.5"` as a number. + +```python +Normal(0.0, 1.0).pdf(float("nan")) # nan +Normal(0.0, 1.0).pdf([0.0, float("inf")]) # array([0.3989..., nan]) +Normal(0.0, 1.0).pdf("0.5") # TypeError +``` + +**Outputs stay inside their mathematical range.** CDFs are clamped to `[0, 1]`, so accumulated +rounding cannot hand back `1 + 1e-16` to code that treats the result as a probability. + +**The C++ API has no exceptions**, so it signals invalid parameters by return value: `NaN` from +anything returning a real number, `-1` from the integer samplers, and `INT_MIN` from +`discrete_uniform_sample`. + +--- + ## Reproducible sampling Every `*_sample()` call draws from one shared Mersenne Twister engine. Seeding it makes a run reproducible: diff --git a/python/fastdist/distributions/bernoulli.py b/python/fastdist/distributions/bernoulli.py index 66df527..256ee0c 100644 --- a/python/fastdist/distributions/bernoulli.py +++ b/python/fastdist/distributions/bernoulli.py @@ -64,7 +64,7 @@ def __init__(self, p: SupportsFloat): TypeError If `p` is not a real number. ValueError - If `p` is outside [0, 1]. + If `p` is outside [0, 1], or is not finite. """ self._validate_params(p=p) @@ -97,7 +97,7 @@ def _validate_params(p: SupportsFloat) -> None: TypeError If `p` is not a real number. ValueError - If `p` is outside [0, 1]. + If `p` is outside [0, 1], or is not finite. Notes ----- @@ -336,7 +336,7 @@ def mean(self, p: Union[SupportsFloat, None] = None) -> float: Raises ------ ValueError - If `p` is outside [0, 1]. + If `p` is outside [0, 1], or is not finite. """ if p is None: @@ -362,7 +362,7 @@ def variance(self, p: Union[SupportsFloat, None] = None) -> float: Raises ------ ValueError - If `p` is outside [0, 1]. + If `p` is outside [0, 1], or is not finite. """ if p is None: @@ -388,7 +388,7 @@ def stddev(self, p: Union[SupportsFloat, None] = None) -> float: Raises ------ ValueError - If `p` is outside [0, 1]. + If `p` is outside [0, 1], or is not finite. """ if p is None: @@ -479,7 +479,7 @@ def sample(self, p: Union[SupportsFloat, None] = None) -> int: Raises ------ ValueError - If `p` is outside [0, 1]. + If `p` is outside [0, 1], or is not finite. """ if p is None: @@ -511,7 +511,7 @@ def _pmf_scalar(cls, k: int, p: SupportsFloat) -> float: Raises ------ ValueError - If `p` is outside [0, 1]. + If `p` is outside [0, 1], or is not finite. TypeError If `k` is not an integer. """ @@ -540,7 +540,7 @@ def _cdf_scalar(cls, k: int, p: SupportsFloat) -> float: Raises ------ ValueError - If `p` is outside [0, 1]. + If `p` is outside [0, 1], or is not finite. TypeError If `k` is not an integer. """ @@ -569,7 +569,7 @@ def _mgf_scalar(cls, t: SupportsFloat, p: SupportsFloat) -> float: Raises ------ ValueError - If `p` is outside [0, 1]. + If `p` is outside [0, 1], or is not finite. TypeError If `t` is not a real number. """ @@ -598,7 +598,7 @@ def _cgf_scalar(cls, t: SupportsFloat, p: SupportsFloat) -> float: Raises ------ ValueError - If `p` is outside [0, 1]. + If `p` is outside [0, 1], or is not finite. TypeError If `t` is not a real number. """ diff --git a/python/fastdist/distributions/exponential.py b/python/fastdist/distributions/exponential.py index 181a2b2..c4a36f6 100644 --- a/python/fastdist/distributions/exponential.py +++ b/python/fastdist/distributions/exponential.py @@ -39,7 +39,7 @@ def __init__(self, lambda_: SupportsFloat): TypeError If lambda_ is not a real number. ValueError - If lambda_ is not positive. + If lambda_ is not positive, or is not finite. """ self._validate_params(lambda_=lambda_) @@ -73,7 +73,7 @@ def _validate_params(lambda_: SupportsFloat) -> None: TypeError If lambda_ is not a real number. ValueError - If lambda_ is not positive. + If lambda_ is not positive, or is not finite. Notes ----- diff --git a/python/fastdist/distributions/uniform.py b/python/fastdist/distributions/uniform.py index c953bfd..569338a 100644 --- a/python/fastdist/distributions/uniform.py +++ b/python/fastdist/distributions/uniform.py @@ -186,7 +186,7 @@ def _validate_params(a: Union[SupportsFloat, None] = None, b: Union[SupportsFloa TypeError If `a` or `b` is not a real number. ValueError - If both `a` and `b` are provided and `a >= b`. + If `a` or `b` is not finite, or both are provided and `a >= b`. """ if a is not None and not isinstance(a, Real): From af65ea0fbbfaaf27480afcccd680d6410dd6b584 Mon Sep 17 00:00:00 2001 From: ghosteau Date: Fri, 11 Sep 2026 21:39:13 -0400 Subject: [PATCH 20/20] Point BENCHMARKS.md at the release benchmarks notebook The running instructions covered run.py, compare.py and table.py but not the notebook, which is the readable view of the same data. Co-Authored-By: Claude Opus 5 --- BENCHMARKS.md | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/BENCHMARKS.md b/BENCHMARKS.md index 0bec272..11ba615 100644 --- a/BENCHMARKS.md +++ b/BENCHMARKS.md @@ -38,6 +38,11 @@ gate a change. To render a report for this log: python benchmarks/table.py --latest ``` +For a rendered view of a recorded run -- both baselines side by side, with a +chart -- open [`examples/release_benchmarks.ipynb`](examples/release_benchmarks.ipynb). +It reads the JSON rather than re-timing, so it costs nothing to open, and it +ships with its output already in place. + --- ## Method, and what the numbers do not say