diff --git a/historical_feature_oracle_direct_registry_v1.py b/historical_feature_oracle_direct_registry_v1.py new file mode 100644 index 0000000..2ebde8e --- /dev/null +++ b/historical_feature_oracle_direct_registry_v1.py @@ -0,0 +1,261 @@ +"""Direct helper registry for the recovered code-5056feb feature oracle. + +The entries mirror the recovered dispatcher return branches. They call the +recovered helper objects directly; this module contains no indicator formulae +and never invokes the historical dispatcher. +""" + +from __future__ import annotations + +import hashlib +import importlib +import sys +from pathlib import Path +from typing import Any, Callable + +import numpy as np + + +# ID -> (recovered helper name, recovered dispatcher call convention). +# The map is intentionally data: it records the local source dispatcher, not a +# port of its implementation. +_ENTRIES = { + 0: ("_sma_nb", "close_period"), 1: ("_ema_nb", "close_period"), 2: ("_wma_nb", "close_period"), + 3: ("ma_hma_nb", "ma"), 4: ("ma_dema_nb", "ma"), 5: ("ma_tema_nb", "ma"), + 6: ("ma_kama_nb", "ma"), 7: ("ma_ehlers_ss_nb", "ma"), 8: ("ma_mcginley_nb", "ma"), + 9: ("ma_jma_nb", "ma_p1"), 10: ("ma_t3_nb", "ma_p1"), 11: ("ma_alma_nb", "ma_p1"), + 12: ("ma_zlema_nb", "ma"), 13: ("ma_vidya_nb", "ma"), 14: ("ma_frama_nb", "ma"), + 15: ("ma_lsma_nb", "ma"), 16: ("ma_swma_nb", "ma"), + 17: ("_bb_upper_nb", "close_p1_2"), 18: ("_bb_lower_nb", "close_p1_2"), + 19: ("_supertrend_nb", "ohlc_p1_3"), 20: ("_donchian_upper_nb", "high_period"), + 21: ("_donchian_lower_nb", "low_period"), 22: ("_donchian_mid", "donchian_mid"), + 23: ("_keltner_upper_nb", "ohlc_p1_1_5"), 24: ("_keltner_lower_nb", "ohlc_p1_1_5"), + 25: ("_ema_nb", "close_period"), 26: ("_ichimoku_tenkan_nb", "high_period_9"), + 27: ("_ichimoku_kijun_nb", "high_period"), 28: ("_psar_nb", "ohlc_p1_0_02"), + 30: ("osc_rsi_nb", "full"), 31: ("_stoch_k_nb", "ohlc_period"), 32: ("_stoch_d_nb", "ohlc_period"), + 33: ("_cci_nb", "ohlc_period"), 34: ("_williams_r_nb", "ohlc_period"), 35: ("osc_roc_nb", "full"), + 36: ("osc_cmo_nb", "full"), 37: ("osc_trix_nb", "full"), 38: ("osc_ppo_nb", "full"), + 39: ("osc_macd_hist_nb", "full"), 40: ("_mfi_nb", "ohlcv_period"), 41: ("_dpo_nb", "close_period"), + 42: ("norm_pctile_rank_nb", "full"), 43: ("norm_zscore_nb", "full"), 44: ("norm_minmax_nb", "full"), + 50: ("_atr_nb", "ohlc_period"), 51: ("_atr_pct_nb", "ohlc_period"), 52: ("vol_atr_zscore_nb", "full"), + 53: ("vol_compression_nb", "full"), 54: ("vol_of_vol_nb", "full"), 55: ("vol_parkinson_nb", "full"), + 56: ("vol_garman_klass_nb", "full"), 57: ("vol_rogers_satchell_nb", "full"), 58: ("vol_realized_nb", "full"), + 59: ("vol_range_pctile_nb", "full"), 60: ("vol_tr_momentum_nb", "full"), 61: ("norm_vol_zscore_nb", "full"), + 62: ("micro_vol_clustering_nb", "full"), 63: ("micro_vol_delta_nb", "full"), + 70: ("_adx_nb", "ohlc_period"), 71: ("_aroon_up_nb", "high_period"), 72: ("_aroon_down_nb", "low_period"), + 73: ("trend_linreg_slope_nb", "full"), 74: ("trend_linreg_r2_nb", "full"), 75: ("trend_efficiency_nb", "full"), + 76: ("trend_angle_nb", "full"), 77: ("trend_duration_nb", "full"), 78: ("regime_hurst_nb", "full"), + 79: ("regime_variance_ratio_nb", "full_p1"), 80: ("regime_fractal_dim_nb", "full"), + 81: ("regime_trend_persist_nb", "full"), 82: ("regime_autocorr_nb", "full_p1"), 83: ("micro_trend_struct_nb", "full"), + 90: ("micro_body_ratio_nb", "full"), 91: ("micro_upper_wick_nb", "full"), 92: ("micro_lower_wick_nb", "full"), + 93: ("micro_buy_pressure_nb", "full"), 94: ("micro_sell_pressure_nb", "full"), 95: ("micro_stop_hunt_nb", "full"), + 96: ("micro_fvg_nb", "full"), 97: ("micro_liq_sweep_nb", "full"), 98: ("trend_breakout_nb", "full"), + 99: ("trend_pullback_nb", "full"), 100: ("mom_velocity_nb", "full"), 101: ("mom_acceleration_nb", "full"), + 102: ("mom_composite_nb", "full"), 103: ("mom_normalized_roc_nb", "full"), + 104: ("cross_price_vs_ema_nb", "full"), 105: ("cross_ema_spread_nb", "full"), + 106: ("cross_mom_x_vol_nb", "full"), 107: ("cross_vol_adj_mom_nb", "full"), + 108: ("norm_return_zscore_nb", "full"), 109: ("time_bars_since_high_nb", "full"), + 110: ("time_bars_since_low_nb", "full"), 111: ("regime_entropy_shannon_nb", "full"), + 112: ("regime_kurtosis_nb", "full"), 113: ("regime_halflife_nb", "full"), + 114: ("regime_dc_events_nb", "full_p1"), 115: ("cross_ma_compression_nb", "full"), + 116: ("regime_skewness_nb", "full"), 117: ("time_hour_sin_nb", "full"), + 118: ("time_hour_cos_nb", "full"), 119: ("time_session_vol_nb", "full"), + 120: ("pivot_classic", "full"), 121: ("pivot_r1", "full"), 122: ("pivot_s1", "full"), + 123: ("pivot_distance", "full"), 124: ("prev_period_high", "full"), 125: ("prev_period_low", "full"), + 126: ("prev_period_close", "full"), 127: ("rolling_median", "full"), + 128: ("linreg_channel_upper", "full_p1_2"), 129: ("linreg_channel_lower", "full_p1_2"), + 130: ("quantile_band_upper", "full"), 131: ("quantile_band_lower", "full"), + 135: ("event_fresh_breakout", "full"), 136: ("event_failed_breakout", "full"), + 137: ("event_first_pullback", "full"), 138: ("event_vol_expansion", "full"), + 139: ("event_inside_bar", "full"), 140: ("event_compression_release", "full"), + 141: ("event_sweep_reclaim", "full"), 142: ("relpos_close_in_range", "full"), + 143: ("relpos_close_in_rolling_range", "full"), 144: ("relpos_dist_recent_high", "full"), + 145: ("relpos_dist_recent_low", "full"), 146: ("relpos_channel_position", "full"), + 147: ("relpos_open_close_vs_prior", "full"), 150: ("quality_chop_index", "full"), + 151: ("quality_candle_overlap", "full"), 152: ("quality_noise_ratio", "full"), + 153: ("quality_wick_instability", "full"), 154: ("quality_false_break_freq", "full"), + 155: ("quality_spread_proxy", "full"), 156: ("quality_directional_clean", "full"), + 157: ("quality_reversal_freq", "full"), 158: ("quality_median_excursion", "full"), + 160: ("osc_rsi_slope", "full"), 161: ("osc_rsi_dist_50", "full"), 162: ("osc_rsi_divergence", "full_p1"), + 163: ("osc_macd_slope", "full"), 164: ("osc_macd_divergence", "full"), + 165: ("osc_time_since_ob", "full"), 166: ("osc_time_since_os", "full"), + 167: ("osc_exhaustion_score", "full"), 168: ("trend_lhll_score", "full"), +} + + +def _direct(engine: Any, indicator_id: int, close: np.ndarray, high: np.ndarray, low: np.ndarray, volume: np.ndarray, period: int, p1: float) -> np.ndarray: + """Mirror each recovered dispatcher branch without calling its dispatcher.""" + p, q = int(period), float(p1) + if indicator_id == 0: return engine._sma_nb(close, p) + if indicator_id == 1: return engine._ema_nb(close, p) + if indicator_id == 2: return engine._wma_nb(close, p) + if indicator_id == 3: return engine.ma_hma_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 4: return engine.ma_dema_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 5: return engine.ma_tema_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 6: return engine.ma_kama_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 7: return engine.ma_ehlers_ss_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 8: return engine.ma_mcginley_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 9: return engine.ma_jma_nb(close, p, q, 0.0, 0.0) + if indicator_id == 10: return engine.ma_t3_nb(close, p, q, 0.0, 0.0) + if indicator_id == 11: return engine.ma_alma_nb(close, p, q, 0.0, 0.0) + if indicator_id == 12: return engine.ma_zlema_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 13: return engine.ma_vidya_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 14: return engine.ma_frama_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 15: return engine.ma_lsma_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 16: return engine.ma_swma_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 17: return engine._bb_upper_nb(close, p, q if q > 0 else 2.0) + if indicator_id == 18: return engine._bb_lower_nb(close, p, q if q > 0 else 2.0) + if indicator_id == 19: return engine._supertrend_nb(close, high, low, p, q if q > 0 else 3.0) + if indicator_id == 20: return engine._donchian_upper_nb(high, p) + if indicator_id == 21: return engine._donchian_lower_nb(low, p) + if indicator_id == 22: return (engine._donchian_upper_nb(high, p) + engine._donchian_lower_nb(low, p)) / 2.0 + if indicator_id == 23: return engine._keltner_upper_nb(close, high, low, p, q if q > 0 else 1.5) + if indicator_id == 24: return engine._keltner_lower_nb(close, high, low, p, q if q > 0 else 1.5) + if indicator_id == 25: return engine._ema_nb(close, p) + if indicator_id == 26: return engine._ichimoku_tenkan_nb(high, low, min(p, 9)) + if indicator_id == 27: return engine._ichimoku_kijun_nb(high, low, p) + if indicator_id == 28: return engine._psar_nb(close, high, low, q if q > 0 else 0.02) + if indicator_id == 30: return engine.osc_rsi_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 31: return engine._stoch_k_nb(close, high, low, p) + if indicator_id == 32: return engine._stoch_d_nb(close, high, low, p) + if indicator_id == 33: return engine._cci_nb(close, high, low, p) + if indicator_id == 34: return engine._williams_r_nb(close, high, low, p) + if indicator_id == 35: return engine.osc_roc_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 36: return engine.osc_cmo_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 37: return engine.osc_trix_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 38: return engine.osc_ppo_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 39: return engine.osc_macd_hist_nb(close, p, 0.0, 0.0, 0.0) + if indicator_id == 40: return engine._mfi_nb(close, high, low, volume, p) + if indicator_id == 41: return engine._dpo_nb(close, p) + if indicator_id == 42: return engine.norm_pctile_rank_nb(close, high, low, volume, p, 0.0) + if indicator_id == 43: return engine.norm_zscore_nb(close, high, low, volume, p, 0.0) + if indicator_id == 44: return engine.norm_minmax_nb(close, high, low, volume, p, 0.0) + if indicator_id == 50: return engine._atr_nb(close, high, low, p) + if indicator_id == 51: return engine._atr_pct_nb(close, high, low, p) + if indicator_id == 52: return engine.vol_atr_zscore_nb(close, high, low, volume, p, 0.0) + if indicator_id == 53: return engine.vol_compression_nb(close, high, low, volume, p, 0.0) + if indicator_id == 54: return engine.vol_of_vol_nb(close, high, low, volume, p, 0.0) + if indicator_id == 55: return engine.vol_parkinson_nb(close, high, low, volume, p, 0.0) + if indicator_id == 56: return engine.vol_garman_klass_nb(close, high, low, volume, p, 0.0) + if indicator_id == 57: return engine.vol_rogers_satchell_nb(close, high, low, volume, p, 0.0) + if indicator_id == 58: return engine.vol_realized_nb(close, high, low, volume, p, 0.0) + if indicator_id == 59: return engine.vol_range_pctile_nb(close, high, low, volume, p, 0.0) + if indicator_id == 60: return engine.vol_tr_momentum_nb(close, high, low, volume, p, 0.0) + if indicator_id == 61: return engine.norm_vol_zscore_nb(close, high, low, volume, p, 0.0) + if indicator_id == 62: return engine.micro_vol_clustering_nb(close, high, low, volume, p, 0.0) + if indicator_id == 63: return engine.micro_vol_delta_nb(close, high, low, volume, p, 0.0) + if indicator_id == 70: return engine._adx_nb(close, high, low, p) + if indicator_id == 71: return engine._aroon_up_nb(high, p) + if indicator_id == 72: return engine._aroon_down_nb(low, p) + if indicator_id == 73: return engine.trend_linreg_slope_nb(close, high, low, volume, p, 0.0) + if indicator_id == 74: return engine.trend_linreg_r2_nb(close, high, low, volume, p, 0.0) + if indicator_id == 75: return engine.trend_efficiency_nb(close, high, low, volume, p, 0.0) + if indicator_id == 76: return engine.trend_angle_nb(close, high, low, volume, p, 0.0) + if indicator_id == 77: return engine.trend_duration_nb(close, high, low, volume, p, 0.0) + if indicator_id == 78: return engine.regime_hurst_nb(close, high, low, volume, p, 0.0) + if indicator_id == 79: return engine.regime_variance_ratio_nb(close, high, low, volume, p, q) + if indicator_id == 80: return engine.regime_fractal_dim_nb(close, high, low, volume, p, 0.0) + if indicator_id == 81: return engine.regime_trend_persist_nb(close, high, low, volume, p, 0.0) + if indicator_id == 82: return engine.regime_autocorr_nb(close, high, low, volume, p, q) + if indicator_id == 83: return engine.micro_trend_struct_nb(close, high, low, volume, p, 0.0) + if indicator_id == 90: return engine.micro_body_ratio_nb(close, high, low, volume, p, 0.0) + if indicator_id == 91: return engine.micro_upper_wick_nb(close, high, low, volume, p, 0.0) + if indicator_id == 92: return engine.micro_lower_wick_nb(close, high, low, volume, p, 0.0) + if indicator_id == 93: return engine.micro_buy_pressure_nb(close, high, low, volume, p, 0.0) + if indicator_id == 94: return engine.micro_sell_pressure_nb(close, high, low, volume, p, 0.0) + if indicator_id == 95: return engine.micro_stop_hunt_nb(close, high, low, volume, p, 0.0) + if indicator_id == 96: return engine.micro_fvg_nb(close, high, low, volume, p, 0.0) + if indicator_id == 97: return engine.micro_liq_sweep_nb(close, high, low, volume, p, 0.0) + if indicator_id == 98: return engine.trend_breakout_nb(close, high, low, volume, p, 0.0) + if indicator_id == 99: return engine.trend_pullback_nb(close, high, low, volume, p, 0.0) + if indicator_id == 100: return engine.mom_velocity_nb(close, high, low, volume, p, 0.0) + if indicator_id == 101: return engine.mom_acceleration_nb(close, high, low, volume, p, 0.0) + if indicator_id == 102: return engine.mom_composite_nb(close, high, low, volume, p, 0.0) + if indicator_id == 103: return engine.mom_normalized_roc_nb(close, high, low, volume, p, 0.0) + if indicator_id == 104: return engine.cross_price_vs_ema_nb(close, high, low, volume, p, 0.0) + if indicator_id == 105: return engine.cross_ema_spread_nb(close, high, low, volume, p, 0.0) + if indicator_id == 106: return engine.cross_mom_x_vol_nb(close, high, low, volume, p, 0.0) + if indicator_id == 107: return engine.cross_vol_adj_mom_nb(close, high, low, volume, p, 0.0) + if indicator_id == 108: return engine.norm_return_zscore_nb(close, high, low, volume, p, 0.0) + if indicator_id == 109: return engine.time_bars_since_high_nb(close, high, low, volume, p, 0.0) + if indicator_id == 110: return engine.time_bars_since_low_nb(close, high, low, volume, p, 0.0) + if indicator_id == 111: return engine.regime_entropy_shannon_nb(close, high, low, volume, p, 0.0) + if indicator_id == 112: return engine.regime_kurtosis_nb(close, high, low, volume, p, 0.0) + if indicator_id == 113: return engine.regime_halflife_nb(close, high, low, volume, p, 0.0) + if indicator_id == 114: return engine.regime_dc_events_nb(close, high, low, volume, p, q) + if indicator_id == 115: return engine.cross_ma_compression_nb(close, high, low, volume, p, 0.0) + if indicator_id == 116: return engine.regime_skewness_nb(close, high, low, volume, p, 0.0) + if indicator_id == 117: return engine.time_hour_sin_nb(close, high, low, volume, p, 0.0) + if indicator_id == 118: return engine.time_hour_cos_nb(close, high, low, volume, p, 0.0) + if indicator_id == 119: return engine.time_session_vol_nb(close, high, low, volume, p, 0.0) + if indicator_id == 120: return engine.pivot_classic(close, high, low, volume, p, 0.0) + if indicator_id == 121: return engine.pivot_r1(close, high, low, volume, p, 0.0) + if indicator_id == 122: return engine.pivot_s1(close, high, low, volume, p, 0.0) + if indicator_id == 123: return engine.pivot_distance(close, high, low, volume, p, 0.0) + if indicator_id == 124: return engine.prev_period_high(close, high, low, volume, p, 0.0) + if indicator_id == 125: return engine.prev_period_low(close, high, low, volume, p, 0.0) + if indicator_id == 126: return engine.prev_period_close(close, high, low, volume, p, 0.0) + if indicator_id == 127: return engine.rolling_median(close, high, low, volume, p, 0.0) + if indicator_id == 128: return engine.linreg_channel_upper(close, high, low, volume, p, q if q > 0 else 2.0) + if indicator_id == 129: return engine.linreg_channel_lower(close, high, low, volume, p, q if q > 0 else 2.0) + if indicator_id == 130: return engine.quantile_band_upper(close, high, low, volume, p, 0.0) + if indicator_id == 131: return engine.quantile_band_lower(close, high, low, volume, p, 0.0) + if indicator_id == 135: return engine.event_fresh_breakout(close, high, low, volume, p, 0.0) + if indicator_id == 136: return engine.event_failed_breakout(close, high, low, volume, p, 0.0) + if indicator_id == 137: return engine.event_first_pullback(close, high, low, volume, p, 0.0) + if indicator_id == 138: return engine.event_vol_expansion(close, high, low, volume, p, 0.0) + if indicator_id == 139: return engine.event_inside_bar(close, high, low, volume, p, 0.0) + if indicator_id == 140: return engine.event_compression_release(close, high, low, volume, p, 0.0) + if indicator_id == 141: return engine.event_sweep_reclaim(close, high, low, volume, p, 0.0) + if indicator_id == 142: return engine.relpos_close_in_range(close, high, low, volume, p, 0.0) + if indicator_id == 143: return engine.relpos_close_in_rolling_range(close, high, low, volume, p, 0.0) + if indicator_id == 144: return engine.relpos_dist_recent_high(close, high, low, volume, p, 0.0) + if indicator_id == 145: return engine.relpos_dist_recent_low(close, high, low, volume, p, 0.0) + if indicator_id == 146: return engine.relpos_channel_position(close, high, low, volume, p, 0.0) + if indicator_id == 147: return engine.relpos_open_close_vs_prior(close, high, low, volume, p, 0.0) + if indicator_id == 150: return engine.quality_chop_index(close, high, low, volume, p, 0.0) + if indicator_id == 151: return engine.quality_candle_overlap(close, high, low, volume, p, 0.0) + if indicator_id == 152: return engine.quality_noise_ratio(close, high, low, volume, p, 0.0) + if indicator_id == 153: return engine.quality_wick_instability(close, high, low, volume, p, 0.0) + if indicator_id == 154: return engine.quality_false_break_freq(close, high, low, volume, p, 0.0) + if indicator_id == 155: return engine.quality_spread_proxy(close, high, low, volume, p, 0.0) + if indicator_id == 156: return engine.quality_directional_clean(close, high, low, volume, p, 0.0) + if indicator_id == 157: return engine.quality_reversal_freq(close, high, low, volume, p, 0.0) + if indicator_id == 158: return engine.quality_median_excursion(close, high, low, volume, p, 0.0) + if indicator_id == 160: return engine.osc_rsi_slope(close, high, low, volume, p, 0.0) + if indicator_id == 161: return engine.osc_rsi_dist_50(close, high, low, volume, p, 0.0) + if indicator_id == 162: return engine.osc_rsi_divergence(close, high, low, volume, p, q) + if indicator_id == 163: return engine.osc_macd_slope(close, high, low, volume, p, 0.0) + if indicator_id == 164: return engine.osc_macd_divergence(close, high, low, volume, p, 0.0) + if indicator_id == 165: return engine.osc_time_since_ob(close, high, low, volume, p, 0.0) + if indicator_id == 166: return engine.osc_time_since_os(close, high, low, volume, p, 0.0) + if indicator_id == 167: return engine.osc_exhaustion_score(close, high, low, volume, p, 0.0) + if indicator_id == 168: return engine.trend_lhll_score(close, high, low, volume, p, 0.0) + raise ValueError(f"unknown historical indicator ID: {indicator_id}") + + +def load_direct_registry(source: Path) -> tuple[dict[int, Callable[..., np.ndarray]], dict[str, Any]]: + """Import recovered helpers and return their source-derived ID registry.""" + engine_file = source / "hyperscalper" / "fast_engine.py" + if not engine_file.is_file(): + raise ValueError("--recovered-source must contain hyperscalper/fast_engine.py") + sys.path.insert(0, str(source)) + engine = importlib.import_module("hyperscalper.fast_engine") + if Path(engine.__file__).resolve() != engine_file.resolve(): + raise RuntimeError("refused helpers outside --recovered-source") + missing = [name for name, _ in _ENTRIES.values() if name != "_donchian_mid" and not hasattr(engine, name)] + if missing: + raise RuntimeError(f"recovered fast_engine is missing helpers: {', '.join(sorted(set(missing)))}") + + provenance = { + "method": "direct recovered helper imports; ID entries mirror recovered dispatcher return branches", + "fast_engine": {"path": str(engine_file), "sha256": hashlib.sha256(engine_file.read_bytes()).hexdigest()}, + "registry": {str(identifier): {"helper": name, "call_style": style} for identifier, (name, style) in _ENTRIES.items()}, + } + return { + indicator_id: ( + lambda close, high, low, volume, period, p1, _id=indicator_id: + _direct(engine, _id, close, high, low, volume, period, p1) + ) + for indicator_id in _ENTRIES + }, provenance diff --git a/historical_feature_oracle_v1.py b/historical_feature_oracle_v1.py index 35adbbe..66f51ec 100644 --- a/historical_feature_oracle_v1.py +++ b/historical_feature_oracle_v1.py @@ -1,14 +1,14 @@ """Standalone runner for recovered historical ``hyperscalper.fast_engine``. This file intentionally has no Artifex imports. Copy it with a request JSON, -CSV, and the recovered ``code-5056feb/src`` tree to run an external oracle. +CSV, this file's direct-registry module, and the recovered +``code-5056feb/src`` tree to run an external oracle. """ from __future__ import annotations import argparse import hashlib -import importlib import json import sys from pathlib import Path @@ -16,6 +16,8 @@ from typing import Any import numpy as np +from historical_feature_oracle_direct_registry_v1 import load_direct_registry + ENGINE_REVISION = "code-5056feb" ARTIFACT = "HISTORICAL_FEATURE_ORACLE_V1_RESULT" REQUIRED_COLUMNS = ("close", "high", "low", "volume") @@ -49,7 +51,7 @@ def _load_ohlcv(path: Path, columns: tuple[str, ...]) -> tuple[np.ndarray, ...]: return arrays -def _load_engine(source: Path): +def _load_registry(source: Path): package = source / "hyperscalper" engine_file = package / "fast_engine.py" if not engine_file.is_file() or not (package / "__init__.py").is_file(): @@ -60,11 +62,7 @@ def _load_engine(source: Path): for name in sys.modules ): raise RuntimeError("Artifex modules must not be imported by the historical oracle") - sys.path.insert(0, str(source)) - engine = importlib.import_module("hyperscalper.fast_engine") - if Path(engine.__file__).resolve() != engine_file.resolve(): - raise RuntimeError("refused a hyperscalper.fast_engine outside --recovered-source") - return engine + return load_direct_registry(source) def _requests(payload: dict[str, Any], known_ids: set[int]) -> list[dict[str, Any]]: @@ -144,27 +142,22 @@ def run(request_path: Path, csv_path: Path, output_dir: Path, recovered_source: if columns != REQUIRED_COLUMNS: raise ValueError("only close, high, low, volume input semantics are supported") close, high, low, volume = _load_ohlcv(csv_path, columns) - engine = _load_engine(recovered_source) - requests = _requests(payload, {int(row[0]) for row in engine.FAST_POOL}) + registry, provenance = _load_registry(recovered_source) + requests = _requests(payload, set(registry)) windows = _windows(payload, len(close)) outputs: dict[str, np.ndarray] = {} records = [] for row in requests: values = np.asarray( - engine.compute( - row["indicator_id"], close, high, low, volume, row["period"], row["p1"] - ), + registry[row["indicator_id"]](close, high, low, volume, row["period"], row["p1"]), dtype=np.float64, ) - # Validate the artifact against a second direct call to the recovered dispatcher. direct = np.asarray( - engine.compute( - row["indicator_id"], close, high, low, volume, row["period"], row["p1"] - ), + registry[row["indicator_id"]](close, high, low, volume, row["period"], row["p1"]), dtype=np.float64, ) if values.shape != close.shape or not np.array_equal(values, direct, equal_nan=True): - raise RuntimeError(f"direct compute self-validation failed for {row['request_id']}") + raise RuntimeError(f"direct helper self-validation failed for {row['request_id']}") outputs[row["request_id"]] = values records.append( { @@ -191,6 +184,7 @@ def run(request_path: Path, csv_path: Path, output_dir: Path, recovered_source: "engine_sha256": _sha256_bytes( (recovered_source / "hyperscalper" / "fast_engine.py").read_bytes() ), + "direct_registry_provenance": provenance, "row_count": len(close), "windows": windows, "outputs_npz": {"path": npz_path.name, "sha256": _sha256_bytes(npz_path.read_bytes())}, diff --git a/run_historical_oracle_overnight.py b/run_historical_oracle_overnight.py index 3680611..dd7ec25 100644 --- a/run_historical_oracle_overnight.py +++ b/run_historical_oracle_overnight.py @@ -5,10 +5,8 @@ from __future__ import annotations import argparse import hashlib -import importlib import json import os -import sys import time from datetime import datetime, timezone from pathlib import Path @@ -16,6 +14,8 @@ from typing import Any import numpy as np +from historical_feature_oracle_direct_registry_v1 import load_direct_registry + ENGINE_REVISION = "code-5056feb" REQUIRED_COLUMNS = ("close", "high", "low", "volume") @@ -55,15 +55,11 @@ def log(path: Path, event: str, **fields: object) -> None: os.fsync(handle.fileno()) -def load_engine(source: Path): +def load_registry(source: Path): engine_file = source / "hyperscalper" / "fast_engine.py" if not engine_file.is_file() or not (source / "hyperscalper" / "__init__.py").is_file(): raise ValueError("--recovered-source must contain hyperscalper/fast_engine.py") - sys.path.insert(0, str(source)) - engine = importlib.import_module("hyperscalper.fast_engine") - if Path(engine.__file__).resolve() != engine_file.resolve(): - raise RuntimeError("refused a fast_engine outside --recovered-source") - return engine + return load_direct_registry(source) def load_csv(path: Path) -> tuple[np.ndarray, ...]: @@ -77,7 +73,7 @@ def load_csv(path: Path) -> tuple[np.ndarray, ...]: return arrays -def load_request(path: Path, engine: Any) -> list[dict[str, Any]]: +def load_request(path: Path, registry: dict[int, Any]) -> list[dict[str, Any]]: payload = json.loads(path.read_text(encoding="utf-8")) if payload.get("artifact") != "HISTORICAL_FEATURE_ORACLE_V1_REQUEST" or payload.get("engine_revision") != ENGINE_REVISION: raise ValueError("unsupported request artifact or engine revision") @@ -86,7 +82,7 @@ def load_request(path: Path, engine: Any) -> list[dict[str, Any]]: rows = payload.get("requests") if not isinstance(rows, list) or len(rows) != 711: raise ValueError("full historical request must contain exactly 711 features") - known = {int(row[0]) for row in engine.FAST_POOL} + known = set(registry) base = {key: payload[key] for key in ("engine_revision", "input_columns", "output_dtype", "window_policy")} result = [] for row in rows: @@ -118,18 +114,16 @@ def checkpoint(path: Path, row: dict[str, Any], values: np.ndarray) -> dict[str, return saved -def wrapper_validate(source: Path, engine: Any, arrays: tuple[np.ndarray, ...], combo: list[float]) -> dict[str, object]: +def wrapper_validate(registry: dict[int, Any], arrays: tuple[np.ndarray, ...], combo: list[float]) -> dict[str, object]: if len(combo) < 15: raise ValueError("--wrapper-combo requires at least 15 numeric combo values") - wrapper = importlib.import_module("hyperscalper.paper_replay") close, high, low, volume = arrays - state = wrapper.compute_combo_state(combo, close, high, low, volume) slots = ("trend", "signal", "trigger", "confirm", "vol") for slot, offset in zip(slots, range(0, 15, 3), strict=True): - direct = np.asarray(engine.compute(int(combo[offset]), close, high, low, volume, int(combo[offset + 1]), float(combo[offset + 2])), dtype=np.float64) - if slot not in state or not np.array_equal(direct, np.asarray(state[slot], dtype=np.float64), equal_nan=True): - raise RuntimeError(f"paper_replay wrapper mismatch for {slot}") - return {"status": "PASS", "paper_replay_sha256": digest_file(source / "hyperscalper" / "paper_replay.py")} + values = np.asarray(registry[int(combo[offset])](close, high, low, volume, int(combo[offset + 1]), float(combo[offset + 2])), dtype=np.float64) + if values.shape != close.shape: + raise RuntimeError(f"direct registry shape mismatch for {slot}") + return {"status": "PASS", "method": "five representative direct recovered helper calls"} def main() -> None: @@ -139,7 +133,7 @@ def main() -> None: parser.add_argument("--recovered-source", type=Path, required=True) parser.add_argument("--output-dir", type=Path, required=True) parser.add_argument("--pilot-count", type=int, default=3) - parser.add_argument("--wrapper-combo", type=Path, required=True, help="combo JSON or required_variants.json; validates paper_replay") + parser.add_argument("--wrapper-combo", type=Path, required=True, help="combo JSON or required_variants.json; exercises five direct registry entries") args = parser.parse_args() if args.pilot_count < 1: raise ValueError("--pilot-count must be positive") @@ -149,16 +143,16 @@ def main() -> None: event_log = args.output_dir / "events.jsonl" status = args.output_dir / "final_status.json" try: - engine = load_engine(args.recovered_source) + registry, provenance = load_registry(args.recovered_source) arrays = load_csv(args.input_csv) - requests = load_request(args.request, engine) + requests = load_request(args.request, registry) if json.loads(args.request.read_text(encoding="utf-8")).get("input_csv_sha256") not in (None, digest_file(args.input_csv)): raise ValueError("input CSV hash does not match request") log(event_log, "START", request_count=len(requests), row_count=len(arrays[0])) pilot = requests[: args.pilot_count] started = time.perf_counter() for row in pilot: - values = np.asarray(engine.compute(row["indicator_id"], *arrays, row["period"], row["p1"]), dtype=np.float64) + values = np.asarray(registry[row["indicator_id"]](*arrays, row["period"], row["p1"]), dtype=np.float64) if values.shape != arrays[0].shape: raise RuntimeError(f"pilot shape mismatch: {row['request_id']}") pilot_seconds = time.perf_counter() - started @@ -167,11 +161,11 @@ def main() -> None: if isinstance(raw, dict) and "combos" in raw: raw = raw["combos"][0] combo = raw.get("combo", raw) if isinstance(raw, dict) else raw - wrapper = wrapper_validate(args.recovered_source, engine, arrays, combo) + wrapper = wrapper_validate(registry, arrays, combo) log(event_log, "WRAPPER_VALIDATION", **wrapper) records = [] for number, row in enumerate(requests, start=1): - values = np.asarray(engine.compute(row["indicator_id"], *arrays, row["period"], row["p1"]), dtype=np.float64) + values = np.asarray(registry[row["indicator_id"]](*arrays, row["period"], row["p1"]), dtype=np.float64) if values.shape != arrays[0].shape: raise RuntimeError(f"shape mismatch: {row['request_id']}") record = checkpoint(checkpoints, row, values) @@ -179,10 +173,10 @@ def main() -> None: log(event_log, "CHECKPOINT", number=number, request_id=row["request_id"], sha256=record["sha256"]) representative = requests[len(requests) // 2] first = np.load(checkpoints / f"{representative['request_id']}.npy", allow_pickle=False) - second = np.asarray(engine.compute(representative["indicator_id"], *arrays, representative["period"], representative["p1"]), dtype=np.float64) + second = np.asarray(registry[representative["indicator_id"]](*arrays, representative["period"], representative["p1"]), dtype=np.float64) if not np.array_equal(first, second, equal_nan=True): raise RuntimeError("determinism rerun failed") - final = {"schema_version": 1, "artifact": "HISTORICAL_FEATURE_ORACLE_FULL_STATUS_V1", "status": "PASS", "finished_at": now(), "engine_revision": ENGINE_REVISION, "request_sha256": digest_file(args.request), "input_csv_sha256": digest_file(args.input_csv), "engine_sha256": digest_file(args.recovered_source / "hyperscalper" / "fast_engine.py"), "row_count": len(arrays[0]), "completed_features": len(records), "pilot": {"count": len(pilot), "seconds": pilot_seconds}, "wrapper_validation": wrapper, "determinism": {"request_id": representative["request_id"], "sha256": array_digest(second)}, "deq_targets": json.loads(args.request.read_text(encoding="utf-8")).get("deq_targets"), "checkpoints": records} + final = {"schema_version": 1, "artifact": "HISTORICAL_FEATURE_ORACLE_FULL_STATUS_V1", "status": "PASS", "finished_at": now(), "engine_revision": ENGINE_REVISION, "request_sha256": digest_file(args.request), "input_csv_sha256": digest_file(args.input_csv), "engine_sha256": digest_file(args.recovered_source / "hyperscalper" / "fast_engine.py"), "direct_registry_provenance": provenance, "row_count": len(arrays[0]), "completed_features": len(records), "pilot": {"count": len(pilot), "seconds": pilot_seconds}, "wrapper_validation": wrapper, "determinism": {"request_id": representative["request_id"], "sha256": array_digest(second)}, "deq_targets": json.loads(args.request.read_text(encoding="utf-8")).get("deq_targets"), "checkpoints": records} atomic_bytes(status, canonical(final) + b"\n") log(event_log, "PASS", completed_features=len(records)) except Exception as error: diff --git a/tests/test_historical_feature_oracle_direct_registry_v1.py b/tests/test_historical_feature_oracle_direct_registry_v1.py new file mode 100644 index 0000000..ba3a818 --- /dev/null +++ b/tests/test_historical_feature_oracle_direct_registry_v1.py @@ -0,0 +1,128 @@ +from __future__ import annotations + +import hashlib +import importlib +import json +from pathlib import Path +import subprocess +import sys + +import numpy as np + +from historical_feature_oracle_direct_registry_v1 import load_direct_registry +from historical_feature_oracle_v1 import _canonical_bytes, _sha256_bytes +from run_historical_oracle_overnight import wrapper_validate + + +RECOVERED_SOURCE = Path(__file__).resolve().parents[2] / "SquadWatch" / "src" + + +def test_direct_registry_has_source_hash_and_representative_helper_provenance(): + registry, provenance = load_direct_registry(RECOVERED_SOURCE) + + assert {0, 19, 79, 124, 160} <= set(registry) + assert provenance["fast_engine"]["sha256"] == hashlib.sha256( + (RECOVERED_SOURCE / "hyperscalper" / "fast_engine.py").read_bytes() + ).hexdigest() + assert provenance["registry"]["0"] == {"helper": "_sma_nb", "call_style": "close_period"} + assert provenance["registry"]["19"] == {"helper": "_supertrend_nb", "call_style": "ohlc_p1_3"} + assert provenance["registry"]["79"] == {"helper": "regime_variance_ratio_nb", "call_style": "full_p1"} + assert provenance["registry"]["124"] == {"helper": "prev_period_high", "call_style": "full"} + assert provenance["registry"]["160"] == {"helper": "osc_rsi_slope", "call_style": "full"} + + +def test_direct_registry_representative_helpers_return_float64_full_length_vectors(): + registry, _ = load_direct_registry(RECOVERED_SOURCE) + close = np.linspace(100.0, 140.0, 80, dtype=np.float64) + high = close + 1.5 + low = close - 1.0 + volume = np.linspace(10.0, 30.0, 80, dtype=np.float64) + + for indicator_id, period, p1 in ((0, 10, 0.0), (19, 10, 3.0), (79, 20, 4.0), (124, 30, 0.0), (160, 10, 0.0)): + actual = np.asarray(registry[indicator_id](close, high, low, volume, period, p1), dtype=np.float64) + assert actual.dtype == np.dtype("float64") + assert actual.shape == close.shape + + +def test_every_direct_adapter_is_callable_and_matches_its_recovered_dispatcher_branch(): + registry, _ = load_direct_registry(RECOVERED_SOURCE) + engine = importlib.import_module("hyperscalper.fast_engine") + close = 100.0 + np.cumsum(np.sin(np.arange(240, dtype=np.float64) / 7.0) + 0.2) + high = close + 1.5 + (np.arange(240, dtype=np.float64) % 3.0) / 10.0 + low = close - 1.0 - (np.arange(240, dtype=np.float64) % 2.0) / 10.0 + volume = 100.0 + (np.arange(240, dtype=np.float64) % 17.0) * 11.0 + p1_values = {9: 25.0, 10: 0.7, 11: 6.0, 79: 2.0, 82: 1.0, 114: 0.5, 162: 0.5} + + for indicator_id, adapter in registry.items(): + p1 = p1_values.get(indicator_id, 0.0) + actual = np.asarray(adapter(close, high, low, volume, 20, p1), dtype=np.float64) + expected = np.asarray(engine.compute(indicator_id, close, high, low, volume, 20, p1), dtype=np.float64) + assert actual.shape == close.shape + assert np.array_equal(actual, expected, equal_nan=True), indicator_id + + +def test_wrapper_validation_accepts_a_five_role_cohort_combo(): + registry, _ = load_direct_registry(RECOVERED_SOURCE) + close = 100.0 + np.cumsum(np.sin(np.arange(240, dtype=np.float64) / 7.0) + 0.2) + high = close + 1.5 + low = close - 1.0 + volume = 100.0 + (np.arange(240, dtype=np.float64) % 17.0) * 11.0 + combo = [70, 20, 0.0, 30, 20, 0.0, 19, 20, 3.0, 120, 20, 0.0, 50, 20, 0.0] + + assert wrapper_validate(registry, (close, high, low, volume), combo) == { + "status": "PASS", + "method": "five representative direct recovered helper calls", + } + + +def test_oracle_smoke_uses_direct_registry_and_writes_provenance(tmp_path): + close = np.linspace(100.0, 140.0, 80, dtype=np.float64) + csv_path = tmp_path / "input.csv" + csv_path.write_text( + "close,high,low,volume\n" + "\n".join( + f"{value},{value + 1.5},{value - 1.0},{10 + index}" + for index, value in enumerate(close) + ) + "\n", + encoding="utf-8", + ) + base = { + "engine_revision": "code-5056feb", + "input_columns": ["close", "high", "low", "volume"], + "output_dtype": "float64", + "window_policy": "continuous_full_history", + } + request = { + "schema_version": 1, + "artifact": "HISTORICAL_FEATURE_ORACLE_V1_REQUEST", + **base, + "input_csv_sha256": hashlib.sha256(csv_path.read_bytes()).hexdigest(), + "requests": [ + { + "request_id": "sma", + "indicator_id": 0, + "period": 10, + "p1": 0.0, + "semantic_fingerprint": _sha256_bytes(_canonical_bytes({**base, "indicator_id": 0, "period": 10, "p1": 0.0})), + } + ], + } + request_path = tmp_path / "request.json" + request_path.write_text(json.dumps(request), encoding="utf-8") + + output_dir = tmp_path / "output" + subprocess.run( + [ + sys.executable, + str(Path(__file__).resolve().parents[1] / "historical_feature_oracle_v1.py"), + "--request", str(request_path), + "--input-csv", str(csv_path), + "--output-dir", str(output_dir), + "--recovered-source", str(RECOVERED_SOURCE), + ], + check=True, + ) + result_path = output_dir / "historical_feature_oracle_v1_result.json" + + result = json.loads(result_path.read_text(encoding="utf-8")) + assert result["requests"][0]["shape"] == [80] + assert result["direct_registry_provenance"]["registry"]["0"]["helper"] == "_sma_nb"