]> Piment Noir Git Repositories - freqai-strategies.git/commitdiff
fix(quickadapter): make causal k-NN weight availability exact (#151)
authorJérôme Benoit <jerome.benoit@piment-noir.org>
Wed, 29 Jul 2026 13:57:20 +0000 (15:57 +0200)
committerGitHub <noreply@github.com>
Wed, 29 Jul 2026 13:57:20 +0000 (15:57 +0200)
Make the causal availability of the label-WEIGHT column exact for adaptive
k-NN Gaussian-fill bandwidths. Compute production 1D k-th-neighbor distances in
O(M log M), and propagate the earliest complete confirmation group at which each
clipped k-NN sigma is provably stable, over the confirmable Zigzag suffix
geometry: a confirmed pivot needs at least five slope observations, so
successive pivots are at least six candles apart and the last future candidate
is n-6. Initial-orientation replay is handled atomically via a position-only
successor bound (last replayed pivot + 6); later groups use the exact
confirmation frontier. Floor/ceiling/interior/singleton stability is decided on
the effective rank min(k, pivot_count-1) jointly with that geometry.

Truncate causal Gaussian fills to the finite support ceil(4*fill_sigma_candles)
tracked by availability; causal_mode=false keeps the legacy dense unbounded
Gaussian bit-identical. Pure-Gaussian uniform pivot centers keep their own label
availability (fixed, constant-clipped, adaptive); only adaptive off-center bands
additionally wait for sigma availability. Thread each label column's
weighting_config to availability.

Add the first quickadapter test harness, locking the Zigzag geometry invariants
the availability proof relies on (min pivot spacing, first-future bound, k-NN
availability bounds), runnable in the freqtrade container.

Fixes #131

README.md
quickadapter/user_data/strategies/QuickAdapterV3.py
quickadapter/user_data/strategies/Utils.py
quickadapter/user_data/strategies/tests/test_causal_weight_availability.py [new file with mode: 0644]

index 83d1424b52db5f388d495bb0734b0317bc83dbca..539b42d156f2435b9fa10e3dc4479b027383ffc3 100644 (file)
--- a/README.md
+++ b/README.md
@@ -81,7 +81,7 @@ docker compose up -d --build
 | freqai.label_weighting.fill_method                             | `zero`                        | enum {`zero`,`epsilon`,`gaussian`,`epsilon_gaussian`}                                                                                                                                                        | Off-pivot weighting scheme. `zero` hard-zeros off-pivot rows; `epsilon` applies the epsilon floor `fill_epsilon * <fill_epsilon_baseline>(pivot_weights)`; in causal mode, each row's baseline uses only pivot weights available with that row's label. `gaussian` applies per-pivot Gaussian bumps; `epsilon_gaussian` sums the `epsilon` floor and the `gaussian` bumps. Pivot rows take the max of their raw weight and the off-pivot field at their index (no-op for `zero`). Switching away from `zero` may require retuning tree-leaf regularization (`min_child_weight`, `lambda`) and resetting any prior Optuna study. Changing this parameter requires deleting trained models. |
 | freqai.label_weighting.fill_epsilon                            | 0.000001                      | float [0,1]                                                                                                                                                                                                  | Off-pivot fraction of the pivot baseline. Ignored when `fill_method` not in {`epsilon`,`epsilon_gaussian`}.                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                               |
 | freqai.label_weighting.fill_epsilon_baseline                   | `mean`                        | enum {`mean`,`median`}                                                                                                                                                                                       | Pivot baseline statistic. `mean` tracks central tendency; `median` is robust against pivot-weight skew. Ignored when `fill_method` not in {`epsilon`,`epsilon_gaussian`}.                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                 |
-| freqai.label_weighting.fill_sigma_candles                      | 25.0                          | float >= 0.5                                                                                                                                                                                                 | Gaussian standard deviation in candles for the per-pivot bumps. Acts as the upper bound on per-pivot sigma when `fill_bandwidth == "knn"`. Lower bound 0.5 prevents severe underflow in the Gaussian tail. Ignored when `fill_method` not in {`gaussian`,`epsilon_gaussian`}.                                                                                                                                                                                                                                                                                                                                                                                                             |
+| freqai.label_weighting.fill_sigma_candles                      | 25.0                          | float >= 0.5                                                                                                                                                                                                 | Gaussian standard deviation in candles for the per-pivot bumps. Acts as the upper bound on per-pivot sigma when `fill_bandwidth == "knn"`. Under `causal_mode=true`, bumps are zero outside the finite support `ceil(4 * fill_sigma_candles)` tracked by exact availability; non-causal baselines retain the legacy unbounded Gaussian tails. Lower bound 0.5 prevents severe underflow inside the causal support. Ignored when `fill_method` not in {`gaussian`,`epsilon_gaussian`}.                                                                                                                                                                                                     |
 | freqai.label_weighting.fill_sigma_min_candles                  | 0.5                           | float >= 0.5                                                                                                                                                                                                 | Lower bound on per-pivot sigma in candles when `fill_bandwidth == "knn"`. Clipped to `fill_sigma_candles` when larger. Ignored when `fill_method` not in {`gaussian`,`epsilon_gaussian`} or `fill_bandwidth != "knn"`.                                                                                                                                                                                                                                                                                                                                                                                                                                                                    |
 | freqai.label_weighting.fill_bandwidth                          | `fixed`                       | enum {`fixed`,`knn`}                                                                                                                                                                                         | Per-pivot Gaussian bandwidth selector. `fixed` applies a constant `fill_sigma_candles` to every pivot (legacy behavior). `knn` adapts each pivot's sigma to local pivot density via `sigma_p = clip(fill_bandwidth_alpha * d_k(p), fill_sigma_min_candles, fill_sigma_candles)` where `d_k(p)` is the index distance to the `k`-th nearest pivot neighbor (Loftsgaarden & Quesenberry 1965; Silverman 1986, §5.2). Mitigates the crushing of weaker pivots by stronger neighbors in dense clusters. Ignored when `fill_method` not in {`gaussian`,`epsilon_gaussian`}.                                                                                                                    |
 | freqai.label_weighting.fill_bandwidth_neighbors                | 1                             | int >= 1                                                                                                                                                                                                     | `k` for the k-nearest-neighbor bandwidth selector. Ignored when `fill_method` not in {`gaussian`,`epsilon_gaussian`} or `fill_bandwidth != "knn"`.                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                        |
@@ -101,7 +101,7 @@ docker compose up -d --build
 | _Feature parameters_                                           |                               |                                                                                                                                                                                                              |                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                           |
 | freqai.feature_parameters.label_period_candles                 | min/max midpoint              | int >= 1                                                                                                                                                                                                     | Zigzag labeling NATR period.                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                              |
 | freqai.feature_parameters.label_horizon_candles                | `label_period_candles`        | int >= 1                                                                                                                                                                                                     | Conservative fixed purge used by causal train/test guards and as the default `timeseries_split` gap. Zigzag labels additionally expose their exact row-wise confirmation time; centered smoothing composes the maximum availability time across each kernel support. When unset, falls back to `label_period_candles`.                                                                                                                                                                                                                                                                                                                                                                    |
-| freqai.feature_parameters.causal_mode                          | true                          | bool                                                                                                                                                                                                         | Causal split guard toggle. When `true` (default): rejects `data_split_parameters.shuffle=true`, `shuffle_after_split=true`, `reverse_train_test_order=true`; for `timeseries_split` auto-sets `gap=label_horizon_candles` when unset/`0` (rejects explicit `gap<label_horizon_candles`); for `train_test_split` applies the same fixed purge; both split methods additionally remove train rows whose exact Zigzag confirmation (and, under active weighting, one-pivot-later weight availability) plus smoothing availability reaches the test boundary. `false` is deprecated; acausal baselines only.                                                                                  |
+| freqai.feature_parameters.causal_mode                          | true                          | bool                                                                                                                                                                                                         | Causal split guard toggle. When `true` (default): rejects `data_split_parameters.shuffle=true`, `shuffle_after_split=true`, `reverse_train_test_order=true`; for `timeseries_split` auto-sets `gap=label_horizon_candles` when unset/`0` (rejects explicit `gap<label_horizon_candles`); for `train_test_split` applies the same fixed purge; both split methods additionally remove train rows whose exact Zigzag confirmation and active weight availability, including k-NN bandwidth confirmation, plus smoothing availability reaches the test boundary. `false` is deprecated; acausal baselines only.                                                                              |
 | freqai.feature_parameters.min_label_period_candles             | 12                            | int >= 1                                                                                                                                                                                                     | Minimum labeling NATR period used for reversals labeling HPO.                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                             |
 | freqai.feature_parameters.max_label_period_candles             | 24                            | int >= 1                                                                                                                                                                                                     | Maximum labeling NATR period used for reversals labeling HPO.                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                             |
 | freqai.feature_parameters.label_natr_multiplier                | min/max midpoint              | float > 0                                                                                                                                                                                                    | Zigzag labeling NATR multiplier.                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                          |
index 2caff75003b6e2d427c37b49b49e034f745b2a07..113c7de5539551612f4d81b6ba2d60a79454121e 100644 (file)
@@ -950,6 +950,7 @@ class QuickAdapterV3(IStrategy):
         causal_mode = get_causal_mode(
             self.freqai_info.get("feature_parameters", {}), logger
         )
+        finite_gaussian_support = causal_mode
 
         for label_col in LABEL_COLUMNS:
             label_params = self.get_label_params(pair, label_col)
@@ -988,6 +989,7 @@ class QuickAdapterV3(IStrategy):
                     indices=label_data.indices,
                     metrics=label_data.metrics,
                     weighting_config=col_weighting_config,
+                    finite_gaussian_support=finite_gaussian_support,
                     logger=logger,
                     known_at_lookahead=(
                         label_data.known_at_lookahead if causal_mode else None
@@ -1000,6 +1002,7 @@ class QuickAdapterV3(IStrategy):
                         known_at_lookahead=label_data.known_at_lookahead,
                         indices=label_data.indices,
                         fill_radius=weight_fill_radius(col_weighting_config),
+                        weighting_config=col_weighting_config,
                     )
 
             if label_col == EXTREMA_COLUMN:
index a839820965b27d2939b89bde4e3a55e19987a0ca..9ace98cbe1cef18de9d2616196211fab95301983 100644 (file)
@@ -2251,6 +2251,65 @@ def _causal_impute_weights(
 _GAUSSIAN_FILL_CHUNK_BUDGET: Final[int] = 50_000_000
 _GAUSSIAN_FILL_DENSITY_WARN: Final[float] = 0.1
 
+# Gaussian fill support: rows within k*fill_sigma_candles of a pivot receive
+# its bump. sqrt(2*ln(100))=3.0349 sigma contains every value above 1% of the
+# pivot peak; k=4 retains a rounding margin and truncates only below exp(-8).
+_WEIGHT_FILL_RADIUS_SIGMA_MULTIPLIER: Final[float] = 4.0
+_ZIGZAG_CONFIRMATION_ALPHA: Final[float] = 0.05
+# With all slopes successful, the one-sided Binomial(0.5) p-value is 2**-m.
+_ZIGZAG_MIN_CONFIRMATION_SLOPES: Final[int] = math.ceil(
+    -math.log2(_ZIGZAG_CONFIRMATION_ALPHA)
+)
+
+
+def _compute_pivot_kth_neighbor_distances(
+    pivot_indices: NDArray[np.floating],
+    neighbors: int,
+) -> NDArray[np.floating]:
+    """Distance from each pivot to its k-th nearest pivot neighbor.
+
+    After sorting, binary-search the split of the k neighbors between the
+    monotone left and right distances: O(M log M) time and O(M) space.
+    """
+    pivot_count = pivot_indices.size
+    sorted_idx = np.argsort(pivot_indices, kind="stable")
+    sorted_positions = pivot_indices[sorted_idx]
+    k = min(int(neighbors), pivot_count - 1)
+    distances = np.empty(pivot_count, dtype=float)
+    for i, position in enumerate(sorted_positions):
+        min_left = max(0, k - (pivot_count - i - 1))
+        max_left = min(k, i)
+        left = min_left
+        right = max_left
+        while left < right:
+            left_count = (left + right) // 2
+            right_count = k - left_count
+            left_distance = (
+                position - sorted_positions[i - left_count] if left_count else 0.0
+            )
+            right_distance = (
+                sorted_positions[i + right_count] - position if right_count else 0.0
+            )
+            if left_distance < right_distance:
+                left = left_count + 1
+            else:
+                right = left_count
+        distances[i] = min(
+            max(
+                (position - sorted_positions[i - left_count] if left_count else 0.0),
+                (
+                    sorted_positions[i + k - left_count] - position
+                    if k - left_count
+                    else 0.0
+                ),
+            )
+            for left_count in (left - 1, left)
+            if min_left <= left_count <= max_left
+        )
+    result = np.empty(pivot_count, dtype=float)
+    result[sorted_idx] = distances
+    return result
+
 
 def _compute_pivot_sigmas(
     pivot_indices: NDArray[np.floating],
@@ -2281,26 +2340,7 @@ def _compute_pivot_sigmas(
             f"supported values are {', '.join(FILL_BANDWIDTHS)}"
         )
 
-    sorted_idx = np.argsort(pivot_indices, kind="stable")
-    sorted_positions = pivot_indices[sorted_idx]
-    k = min(int(neighbors), M - 1)
-
-    d_k_sorted = np.empty(M, dtype=float)
-    for i, position in enumerate(sorted_positions):
-        left = max(0, i - k)
-        right = min(M, i + k + 1)
-        candidate_distances = np.abs(
-            np.concatenate(
-                (
-                    sorted_positions[left:i] - position,
-                    sorted_positions[i + 1 : right] - position,
-                )
-            )
-        )
-        d_k_sorted[i] = np.partition(candidate_distances, k - 1)[k - 1]
-    d_k = np.empty(M, dtype=float)
-    d_k[sorted_idx] = d_k_sorted
-
+    d_k = _compute_pivot_kth_neighbor_distances(pivot_indices, neighbors)
     sigmas = float(alpha) * d_k
     sigma_max = float(sigma_candles)
     sigma_min = float(sigma_min_candles)
@@ -2319,11 +2359,15 @@ def _gaussian_fill_weights(
     bandwidth_neighbors: int = 1,
     bandwidth_alpha: float = 1.0,
     sigma_min_candles: float = 0.5,
+    finite_support: bool = False,
     logger: Logger | None = None,
 ) -> NDArray[np.floating]:
     """Per-row max of per-pivot Gaussian bumps.
 
-    Out[i] = max over p of ``w_p * exp(-(i - p)**2 / (2 * sigma_p**2))``.
+    ``Out[i] = max_p w_p * exp(-(i - p)**2 / (2 * sigma_p**2))``. When
+    ``finite_support`` is true, values outside
+    ``ceil(4 * sigma_candles)`` are zero to match causal availability
+    propagation. The default retains the legacy unbounded Gaussian tails.
 
     With ``bandwidth == "fixed"``, ``sigma_p == sigma_candles`` for every
     pivot. Clustered pivots within ``~sigma_candles`` then let the strongest
@@ -2346,7 +2390,7 @@ def _gaussian_fill_weights(
             f"Invalid pivot_weights min={float(pivot_weights.min())!r}: must be >= 0"
         )
     pivot_indices_array = pivot_indices.astype(float)
-    pivot_weights_row = pivot_weights.astype(float)[np.newaxis, :]
+    pivot_weights_array = pivot_weights.astype(float)
     pivot_sigmas = _compute_pivot_sigmas(
         pivot_indices=pivot_indices_array,
         sigma_candles=sigma_candles,
@@ -2355,7 +2399,6 @@ def _gaussian_fill_weights(
         alpha=bandwidth_alpha,
         sigma_min_candles=sigma_min_candles,
     )
-    inv_two_sigma_sq_row = (0.5 / (pivot_sigmas * pivot_sigmas))[np.newaxis, :]
     M = pivot_indices_array.size
     if (
         logger is not None
@@ -2370,29 +2413,54 @@ def _gaussian_fill_weights(
             M,
             n_values,
         )
-    chunk = max(1, _GAUSSIAN_FILL_CHUNK_BUDGET // max(M, 1))
-    if logger is not None and chunk < n_values:
-        logger.debug(
-            "gaussian_fill: N=%d, M=%d, chunk=%d, ~%.0f MB peak buffer, "
-            "bandwidth=%s, sigma=[%.2f, %.2f]",
-            n_values,
-            M,
-            chunk,
-            chunk * M * 8 / 1e6,
-            bandwidth,
-            float(pivot_sigmas.min()),
-            float(pivot_sigmas.max()),
-        )
+    if not finite_support:
+        pivot_weights_row = pivot_weights_array[np.newaxis, :]
+        inv_two_sigma_sq_row = (0.5 / (pivot_sigmas * pivot_sigmas))[np.newaxis, :]
+        chunk = max(1, _GAUSSIAN_FILL_CHUNK_BUDGET // max(M, 1))
+        if logger is not None and chunk < n_values:
+            logger.debug(
+                "gaussian_fill: N=%d, M=%d, chunk=%d, ~%.0f MB peak buffer, "
+                "bandwidth=%s, sigma=[%.2f, %.2f]",
+                n_values,
+                M,
+                chunk,
+                chunk * M * 8 / 1e6,
+                bandwidth,
+                float(pivot_sigmas.min()),
+                float(pivot_sigmas.max()),
+            )
+        out = np.zeros(n_values, dtype=float)
+        for start in range(0, n_values, chunk):
+            stop = min(start + chunk, n_values)
+            positions = np.arange(start, stop, dtype=float)
+            buf = positions[:, np.newaxis] - pivot_indices_array[np.newaxis, :]
+            np.multiply(buf, buf, out=buf)
+            np.multiply(buf, -inv_two_sigma_sq_row, out=buf)
+            np.exp(buf, out=buf)
+            np.multiply(buf, pivot_weights_row, out=buf)
+            np.max(buf, axis=1, out=out[start:stop])
+        return out
+
     out = np.zeros(n_values, dtype=float)
-    for start in range(0, n_values, chunk):
-        stop = min(start + chunk, n_values)
+    support_radius = math.ceil(
+        _WEIGHT_FILL_RADIUS_SIGMA_MULTIPLIER * float(sigma_candles)
+    )
+    for pivot, pivot_weight, pivot_sigma in zip(
+        pivot_indices_array,
+        pivot_weights_array,
+        pivot_sigmas,
+    ):
+        if pivot_weight == 0.0:
+            continue
+        start = max(0, int(pivot) - support_radius)
+        stop = min(n_values, int(pivot) + support_radius + 1)
         positions = np.arange(start, stop, dtype=float)
-        buf = positions[:, np.newaxis] - pivot_indices_array[np.newaxis, :]
-        np.multiply(buf, buf, out=buf)
-        np.multiply(buf, -inv_two_sigma_sq_row, out=buf)
-        np.exp(buf, out=buf)
-        np.multiply(buf, pivot_weights_row, out=buf)
-        np.max(buf, axis=1, out=out[start:stop])
+        np.subtract(positions, pivot, out=positions)
+        np.multiply(positions, positions, out=positions)
+        np.multiply(positions, -0.5 / (pivot_sigma * pivot_sigma), out=positions)
+        np.exp(positions, out=positions)
+        np.multiply(positions, pivot_weight, out=positions)
+        np.maximum(out[start:stop], positions, out=out[start:stop])
     return out
 
 
@@ -2719,6 +2787,7 @@ def _compute_gaussian_bumps(
     weights: NDArray[np.floating],
     label_weighting: dict[str, Any],
     *,
+    finite_support: bool,
     logger: Logger | None,
 ) -> NDArray[np.floating]:
     """Per-row max of per-pivot Gaussian bumps.
@@ -2735,6 +2804,7 @@ def _compute_gaussian_bumps(
         bandwidth_neighbors=label_weighting["fill_bandwidth_neighbors"],
         bandwidth_alpha=label_weighting["fill_bandwidth_alpha"],
         sigma_min_candles=label_weighting["fill_sigma_min_candles"],
+        finite_support=finite_support,
         logger=logger,
     )
 
@@ -2745,6 +2815,7 @@ def compute_label_weights(
     metrics: dict[str, list[float]],
     weighting_config: dict[str, Any],
     *,
+    finite_gaussian_support: bool = False,
     logger: Logger,
     known_at_lookahead: pd.Series | None = None,
 ) -> NDArray[np.floating]:
@@ -2753,8 +2824,10 @@ def compute_label_weights(
     Returns an array with positive values at pivot ``indices`` (scaled by
     strategy) and off-pivot values controlled by ``fill_method``.
     ``known_at_lookahead`` enables a per-row causal epsilon baseline; ``None``
-    preserves the global non-causal baseline. Callers must skip invocation when
-    strategy is ``'none'``; this raises ValueError otherwise.
+    preserves the global non-causal baseline. ``finite_gaussian_support``
+    truncates Gaussian fills to the radius tracked by causal availability.
+    Callers must skip invocation when strategy is ``'none'``; this raises
+    ValueError otherwise.
     """
     label_weighting = {**DEFAULTS_LABEL_WEIGHTING, **weighting_config}
     strategy = label_weighting["strategy"]
@@ -2803,11 +2876,23 @@ def compute_label_weights(
         )
     elif fill_method == FILL_METHODS[2]:  # "gaussian"
         fill_weights = _compute_gaussian_bumps(
-            n_values, indices_array, valid_mask, weights, label_weighting, logger=logger
+            n_values,
+            indices_array,
+            valid_mask,
+            weights,
+            label_weighting,
+            finite_support=finite_gaussian_support,
+            logger=logger,
         )
     elif fill_method == FILL_METHODS[3]:  # "epsilon_gaussian"
         fill_weights = _compute_gaussian_bumps(
-            n_values, indices_array, valid_mask, weights, label_weighting, logger=logger
+            n_values,
+            indices_array,
+            valid_mask,
+            weights,
+            label_weighting,
+            finite_support=finite_gaussian_support,
+            logger=logger,
         )
         np.add(
             fill_weights,
@@ -2837,21 +2922,14 @@ def compute_label_weights(
     )
 
 
-# k for the weight-fill availability band: rows within k*fill_sigma_candles of a
-# pivot receive its Gaussian bump. Material radius (bump > 1% of pivot peak) is
-# sqrt(2*ln(100))=3.0349 sigma; k=4 -> tail exp(-8)=3.4e-4, ~30x below 1% with a
-# margin robust to sigma rounding (k=3 under-covers for larger sigma).
-_WEIGHT_FILL_RADIUS_SIGMA_MULTIPLIER: Final[float] = 4.0
-
-
 def weight_fill_radius(weighting_config: dict[str, Any]) -> int:
     """Row radius over which a pivot's Gaussian-fill weight is causally shared.
 
     Zero unless the off-pivot fill spreads a pivot's weight into neighbors
     (``gaussian``/``epsilon_gaussian``). ``fill_sigma_candles`` upper-bounds the
     per-pivot sigma (including ``knn``, which clips below it), so
-    ``ceil(k*fill_sigma_candles)`` covers every pivot's material Gaussian
-    support. The additive epsilon floor uses its separate per-row causal
+    ``ceil(k*fill_sigma_candles)`` is the finite support used by Gaussian
+    generation. The additive epsilon floor uses its separate per-row causal
     baseline, so it needs no local fill radius.
     """
     label_weighting = {**DEFAULTS_LABEL_WEIGHTING, **weighting_config}
@@ -2866,40 +2944,207 @@ def weight_fill_radius(weighting_config: dict[str, Any]) -> int:
     )
 
 
+def _compute_knn_pivot_sigma_availability(
+    pivot_indices: NDArray[np.integer],
+    known_at_positions: NDArray[np.integer],
+    neighbors: int,
+    alpha: float,
+    sigma_min_candles: float,
+    sigma_candles: float,
+    n: int,
+) -> NDArray[np.int64]:
+    """Absolute availability position of each k-NN pivot bandwidth.
+
+    At each atomically complete confirmation group, the possible effective
+    k-th-neighbor distances include every confirmable finite suffix of the
+    remaining frame. Successive Zigzag pivots need at least five slope
+    observations and are therefore at least six candles apart. The bandwidth
+    settles once every possible clipped sigma equals its final value. Otherwise
+    it is knowable only at the frame boundary.
+    """
+    pivot_count = pivot_indices.size
+    availability = np.full(pivot_count, n, dtype=np.int64)
+    if pivot_count == 0:
+        return availability
+
+    pivot_positions = pivot_indices.astype(np.int64, copy=False)
+    pivot_confirmations = known_at_positions[pivot_positions]
+    confirmation_group_ends = np.flatnonzero(
+        np.r_[pivot_confirmations[:-1] != pivot_confirmations[1:], True]
+    )
+    pivot_spacing = _ZIGZAG_MIN_CONFIRMATION_SLOPES + 1
+    last_future_pivot_position = n - pivot_spacing
+    kth_distances = (
+        np.full(1, np.inf)
+        if pivot_count == 1
+        else _compute_pivot_kth_neighbor_distances(
+            pivot_positions.astype(float),
+            neighbors,
+        )
+    )
+    sigma_min = min(float(sigma_min_candles), float(sigma_candles))
+    sigma_max = float(sigma_candles)
+    alpha_value = float(alpha)
+    for i, (pivot_position, kth_distance) in enumerate(
+        zip(pivot_positions, kth_distances)
+    ):
+        raw_sigma = alpha_value * kth_distance
+        if raw_sigma >= sigma_max:
+            lower, upper = 0, n
+            while lower < upper:
+                middle = (lower + upper) // 2
+                if alpha_value * middle < sigma_max:
+                    lower = middle + 1
+                else:
+                    upper = middle
+            closer_radius = lower - 1
+            within_radius = closer_radius
+        elif raw_sigma <= sigma_min:
+            lower, upper = 0, n
+            while lower < upper:
+                middle = (lower + upper) // 2
+                if alpha_value * middle <= sigma_min:
+                    lower = middle + 1
+                else:
+                    upper = middle
+            within_radius = lower - 1
+            closer_radius = within_radius
+        else:
+            within_radius = int(kth_distance)
+            closer_radius = within_radius - 1
+        closer_left = int(
+            np.searchsorted(
+                pivot_positions,
+                pivot_position - closer_radius,
+                side="left",
+            )
+        )
+        closer_right = int(
+            np.searchsorted(
+                pivot_positions,
+                pivot_position + closer_radius,
+                side="right",
+            )
+        )
+        within_left = int(
+            np.searchsorted(
+                pivot_positions,
+                pivot_position - within_radius,
+                side="left",
+            )
+        )
+        within_right = int(
+            np.searchsorted(
+                pivot_positions,
+                pivot_position + within_radius,
+                side="right",
+            )
+        )
+        future_closer_end = pivot_position + closer_radius
+        future_within_end = pivot_position + within_radius
+
+        # Prefixes are evaluated only after complete confirmation groups. The
+        # stability predicate is monotone because each group removes suffixes
+        # from the same finite set of possible continuations.
+        left = int(np.searchsorted(confirmation_group_ends, i, side="left"))
+        right = confirmation_group_ends.size
+        while left < right:
+            middle = (left + right) // 2
+            bound = int(confirmation_group_ends[middle])
+            prefix_end = bound + 1
+            confirmed_neighbors = bound
+            if pivot_confirmations[bound] == pivot_confirmations[0]:
+                # Initial-orientation replay may confirm several pivots
+                # atomically. The last replayed pivot's internal confirmation
+                # is hidden by that watermark, but its successor must still be
+                # at least ``pivot_spacing`` candles later.
+                first_future_pivot_position = (
+                    int(pivot_positions[bound]) + pivot_spacing
+                )
+            else:
+                first_future_pivot_position = int(pivot_confirmations[bound]) + 1
+            has_future = first_future_pivot_position <= last_future_pivot_position
+            confirmed_rank = min(neighbors, confirmed_neighbors)
+            confirmed_closer = max(0, min(prefix_end, closer_right) - closer_left - 1)
+            confirmed_within = max(0, min(prefix_end, within_right) - within_left - 1)
+            last_future_closer_position = min(
+                last_future_pivot_position, future_closer_end
+            )
+            future_closer = (
+                0
+                if last_future_closer_position < first_future_pivot_position
+                else (last_future_closer_position - first_future_pivot_position)
+                // pivot_spacing
+                + 1
+            )
+            all_future_within = (
+                not has_future or last_future_pivot_position <= future_within_end
+            )
+
+            if raw_sigma >= sigma_max:
+                prefix_matches = (
+                    confirmed_neighbors == 0 or confirmed_closer < confirmed_rank
+                )
+                suffix_matches = (
+                    future_closer == 0
+                    if confirmed_neighbors == 0
+                    else confirmed_closer + future_closer < neighbors
+                )
+                stable = prefix_matches and suffix_matches
+            elif raw_sigma <= sigma_min:
+                stable = (
+                    confirmed_neighbors > 0
+                    and confirmed_within >= confirmed_rank
+                    and (confirmed_neighbors >= neighbors or all_future_within)
+                )
+            else:
+                stable = (
+                    confirmed_neighbors > 0
+                    and confirmed_closer < confirmed_rank <= confirmed_within
+                    and confirmed_closer + future_closer < neighbors
+                    and (confirmed_neighbors >= neighbors or all_future_within)
+                )
+            if stable:
+                right = middle
+            else:
+                left = middle + 1
+        if left < confirmation_group_ends.size:
+            bound = int(confirmation_group_ends[left])
+            availability[i] = pivot_confirmations[bound]
+    return availability
+
+
 def compute_label_weight_known_at_lookahead(
     known_at_lookahead: pd.Series,
     indices: Sequence[int] | NDArray[np.integer],
     fill_radius: int = 0,
+    *,
+    weighting_config: dict[str, Any] | None = None,
 ) -> pd.Series:
     """Per-row causal availability (in candles) of the label WEIGHT column.
 
-    A pivot's swing metric (its weight source) is backfilled from the adjacent
-    closing pivot, so it only becomes computable at the next pivot's
-    confirmation ``i_{k+1} == known_at_positions[indices[k+1]]``, one pivot
-    later than the pivot's own label availability ``i_k``. The trailing pivot has
-    no closing swing (weight 0 via ``_impute_weights``, so it never resolves
-    in-frame -> ``n``). Pivot rows are bumped to that lag; off-pivot rows keep
-    their label availability, EXCEPT that a Gaussian fill spreads each pivot's
-    weight over a local band ``[idx-fill_radius, idx+fill_radius]``
-    (0 disables), whose availability is bounded too. The band is LOCAL per pivot,
-    never a global max (a global bound forces ``n`` on all rows -> total train
-    purge). Folded via ``max(label, weight)`` by the causal purge.
-
-    The ``i_{k+1}`` availability is exact for ``fill_bandwidth='fixed'`` and for
-    ``knn`` with ``fill_bandwidth_neighbors=1`` (a pivot sigma then depends only
-    on already-confirmed adjacent pivots), up to each pivot's material Gaussian
-    support (the tail beyond ``fill_radius`` is immaterial by design, see
-    ``weight_fill_radius``). Under ``knn`` with ``fill_bandwidth_neighbors>=2`` a
-    pivot sigma can depend on a k-th nearest neighbor confirmed after ``i_{k+1}``,
-    making ``i_{k+1}`` a lower bound. This understates the band availability only
-    when that neighbor also lies outside ``fill_radius``, which requires
-    ``fill_bandwidth_alpha < 0.25`` (otherwise ``fill_radius = ceil(4 *
-    fill_sigma_candles)`` covers it and the neighbor's own later availability
-    dominates the ``max`` fold): the understatement is nil at the default
-    ``fill_bandwidth_alpha=0.5`` and can otherwise reach a large fraction of the
-    peak weight on the affected rows. Prefer ``fill_bandwidth='fixed'``,
-    ``fill_bandwidth_neighbors=1``, or ``fill_bandwidth_alpha>=0.25`` for
-    exactness.
+    A metric-based pivot's weight is backfilled from the adjacent closing pivot,
+    so it becomes computable at the next pivot's confirmation
+    ``i_{k+1} == known_at_positions[indices[k+1]]``; the trailing pivot has no
+    closing swing (weight 0 via ``_impute_weights``) and never resolves in-frame
+    -> ``n``. A uniform pivot instead has a unit weight at its own label
+    availability. Off-pivot rows keep their label availability, except that a
+    Gaussian fill spreads each pivot's weight over a LOCAL band
+    ``[idx-fill_radius, idx+fill_radius]`` (0 disables) -- never a global max,
+    which would force ``n`` on all rows (total train purge). Folded via
+    ``max(label, weight)`` by the causal purge.
+
+    For adaptive k-NN bandwidths a pivot's band additionally waits until every
+    confirmable finite suffix of the frame yields the same clipped sigma --
+    jointly over geometry and the effective rank ``min(k, pivot_count - 1)`` --
+    evaluated only at atomically complete confirmation groups; this tracks when
+    the final-frame sigma is knowable, not merely computable from a prefix, and
+    is ``n`` absent such proof. Under pure-Gaussian ``uniform`` weighting every
+    bump is bounded by the unit pivot weight, so pivot centers keep their own
+    label availability (fixed, constant-clipped, adaptive), while adaptive
+    off-center bands still wait for the pivot confirmation and sigma. Other
+    strategies and additive fills keep their existing competing-band
+    dependencies.
     """
     n = len(known_at_lookahead)
     positions, known_at_lookahead_values = _sanitize_known_at_lookahead(
@@ -2912,20 +3157,58 @@ def compute_label_weight_known_at_lookahead(
     idx = np.sort(idx[(idx >= 0) & (idx < n)])
     base = known_at_positions.copy()
     if idx.size:
-        avail_pivot = np.empty(idx.size, dtype=np.int64)
-        avail_pivot[:-1] = known_at_positions[idx[1:]]
-        avail_pivot[-1] = n
+        weight_availability = np.empty(idx.size, dtype=np.int64)
+        weight_availability[:-1] = known_at_positions[idx[1:]]
+        weight_availability[-1] = n
+        label_weighting = {
+            **DEFAULTS_LABEL_WEIGHTING,
+            **(weighting_config or {}),
+        }
+        adaptive_knn = (
+            fill_radius > 0
+            and label_weighting["fill_bandwidth"] == FILL_BANDWIDTHS[1]  # "knn"
+            and float(label_weighting["fill_sigma_min_candles"])
+            < float(label_weighting["fill_sigma_candles"])
+        )
+        uniform_gaussian = (
+            label_weighting["strategy"] == WEIGHT_STRATEGIES[1]
+            and label_weighting["fill_method"] == FILL_METHODS[2]  # "gaussian"
+        )
+        band_weight_availability = (
+            known_at_positions[idx].copy() if uniform_gaussian else weight_availability
+        )
+        avail_pivot = band_weight_availability.copy()
+        if adaptive_knn:
+            sigma_availability = _compute_knn_pivot_sigma_availability(
+                idx,
+                known_at_positions,
+                label_weighting["fill_bandwidth_neighbors"],
+                label_weighting["fill_bandwidth_alpha"],
+                label_weighting["fill_sigma_min_candles"],
+                label_weighting["fill_sigma_candles"],
+                n,
+            )
+            np.maximum(avail_pivot, sigma_availability, out=avail_pivot)
         base[idx] = np.maximum(base[idx], avail_pivot)
         if fill_radius > 0:
-            for pivot_pos, pivot_avail in zip(idx.tolist(), avail_pivot.tolist()):
-                # Skip any pivot whose weight never resolves in-frame (sentinel
-                # availability == n): its Gaussian bump is 0, so it contributes
-                # to no row (on real _zigzag output only the trailing pivot).
-                if pivot_avail >= n:
+            for pivot_pos, pivot_avail, weight_avail in zip(
+                idx.tolist(),
+                avail_pivot.tolist(),
+                band_weight_availability.tolist(),
+            ):
+                # A metric-based trailing pivot never resolves in-frame and its
+                # Gaussian bump is 0. Pure-Gaussian uniform pivots use their own
+                # confirmation, so only that path retains the trailing band.
+                if weight_avail >= n:
                     continue
                 lo = max(0, pivot_pos - fill_radius)
                 hi = min(n, pivot_pos + fill_radius + 1)
                 np.maximum(base[lo:hi], pivot_avail, out=base[lo:hi])
+        if uniform_gaussian:
+            base[idx] = np.maximum(
+                known_at_positions[idx],
+                band_weight_availability,
+            )
     base = np.clip(base, positions, n)
     return pd.Series(base - positions, index=known_at_lookahead.index, dtype=np.int64)
 
@@ -3981,7 +4264,7 @@ def _zigzag(
         candidate_pivot_pos: int,
         direction: TrendDirection,
         min_slope: float = np.finfo(float).eps,
-        alpha: float = 0.05,
+        alpha: float = _ZIGZAG_CONFIRMATION_ALPHA,
     ) -> bool:
         start_pos = min(candidate_pivot_pos + 1, n)
         end_pos = min(pos + 1, n)
diff --git a/quickadapter/user_data/strategies/tests/test_causal_weight_availability.py b/quickadapter/user_data/strategies/tests/test_causal_weight_availability.py
new file mode 100644 (file)
index 0000000..e35fd25
--- /dev/null
@@ -0,0 +1,98 @@
+"""Regression tests for the causal label-weight availability invariants (#131).
+
+`_compute_knn_pivot_sigma_availability` proves leak-free causal availability by
+assuming the Zigzag confirmation geometry: successive pivots are at least
+`_ZIGZAG_MIN_CONFIRMATION_SLOPES + 1` candles apart, and the earliest possible
+successor position (`first_future`) never exceeds the actual next pivot. These
+tests lock both against the real `_zigzag`, so a future Zigzag change that
+weakens either assumption fails here instead of silently leaking future info.
+
+Requires the freqtrade runtime stack (talib). Run in the container:
+    python -m pytest user_data/strategies/tests/ -q
+or directly:
+    python user_data/strategies/tests/test_causal_weight_availability.py
+"""
+
+import os
+import sys
+
+import numpy as np
+import pandas as pd
+
+sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
+
+import Utils  # noqa: E402  (needs the strategies dir on sys.path)
+
+_PIVOT_SPACING = Utils._ZIGZAG_MIN_CONFIRMATION_SLOPES + 1
+
+
+def _make_ohlc(rng: np.random.Generator, n: int) -> pd.DataFrame:
+    log_return = rng.normal(0.0, 0.01, n).cumsum()
+    close = 100.0 * np.exp(log_return)
+    high = close * (1.0 + np.abs(rng.normal(0.0, 0.004, n)))
+    low = close * (1.0 - np.abs(rng.normal(0.0, 0.004, n)))
+    volume = rng.uniform(1e3, 1e4, n)
+    return pd.DataFrame({"close": close, "high": high, "low": low, "volume": volume})
+
+
+def _real_pivot_series(seed: int, series: int = 120):
+    rng = np.random.default_rng(seed)
+    out = []
+    for _ in range(series):
+        n = int(rng.integers(250, 1000))
+        result = Utils._zigzag(
+            _make_ohlc(rng, n),
+            natr_period=int(rng.integers(10, 20)),
+            natr_multiplier=float(rng.uniform(6.0, 12.0)),
+        )
+        idx = np.asarray(result.indices, dtype=np.int64)
+        if idx.size >= 2:
+            out.append((idx, result.known_at_positions, n))
+    return out
+
+
+def test_min_pivot_spacing_covers_confirmation_bound() -> None:
+    """Real consecutive pivots are never closer than `_PIVOT_SPACING`."""
+    worst = min(
+        int(np.diff(idx).min()) for idx, _known_at, _n in _real_pivot_series(20260729)
+    )
+    assert worst >= _PIVOT_SPACING, worst
+
+
+def test_first_future_bound_is_never_understated() -> None:
+    """`first_future` <= actual next pivot, so possible futures are over-counted.
+
+    Ordinary group: `confirmation + 1`. Initial-orientation replay group (shared
+    confirmation watermark): `position + _PIVOT_SPACING`. Either bound must not
+    exceed the real next pivot, or the availability predicate would understate
+    availability and leak future information.
+    """
+    for idx, known_at_positions, _n in _real_pivot_series(20260729):
+        confirmation_0 = known_at_positions[idx[0]]
+        for k in range(idx.size - 1):
+            confirmation_k = int(known_at_positions[idx[k]])
+            first_future = (
+                idx[k] + _PIVOT_SPACING
+                if confirmation_k == confirmation_0
+                else confirmation_k + 1
+            )
+            assert first_future <= idx[k + 1], (k, first_future, int(idx[k + 1]))
+
+
+def test_knn_sigma_availability_within_bounds() -> None:
+    """Availability stays in `[pivot confirmation, n]` on real pivots."""
+    for idx, known_at_positions, n in _real_pivot_series(1, series=30):
+        availability = Utils._compute_knn_pivot_sigma_availability(
+            idx, known_at_positions, 4, 0.2, 0.5, 2.0, n
+        )
+        assert availability.shape == idx.shape
+        assert np.all(availability >= known_at_positions[idx])
+        assert np.all(availability <= n)
+
+
+if __name__ == "__main__":
+    for name, test in sorted(globals().items()):
+        if name.startswith("test_") and callable(test):
+            test()
+            print(f"PASS {name}")
+    print("ALL PASS")