docs: clarify legacy ML target semantics

This commit is contained in:
Hermes Agent
2026-07-26 23:00:46 +00:00
parent 81654b5743
commit a54dec357f
+7 -12
View File
@@ -161,8 +161,8 @@ def compute_features(df, config):
def create_accumulation_target(df, config):
"""Create accumulation score target based on forward returns.
For each candle, compute actual forward returns at multiple horizons,
rank them, and create a weighted accumulation score (0-100).
For each candle, compute actual forward returns at multiple horizons and
map them through fixed, configured return scales to a weighted 0-100 score.
Times when buying led to the best long-term returns get highest scores.
"""
tgt = config.get("target", {})
@@ -931,10 +931,8 @@ def compile_results(predictions, per_window_cost_improvement,
else:
avg_actual_strong = 0.0
# We need forward return info. Since actual_score is a rank-based measure (0-100),
# and we want to report real forward returns, we approximate:
# actual_score > 80 means the buy was in the top 20% of quality.
# For actual forward return stats, we use actual score as a proxy.
# The target is a bounded return-quality score, not a realized return.
# Report it explicitly as quality rather than approximating a return.
# Profitable signals: those where actual score is also above median (50)
if strong_buy_count > 0:
@@ -968,14 +966,14 @@ def compile_results(predictions, per_window_cost_improvement,
signal_frequency = strong_buy_count / total_candles * 100 if total_candles > 0 else 0
# --- Score at actual extremes ---
# "Actual bottoms" = candles with actual score > 85 (top 15% buy opportunities)
# "Actual bottoms" = candles with a high realized return-quality score.
actual_bottom_mask = actual_scores > 85
if np.any(actual_bottom_mask):
avg_score_at_bottoms = float(np.mean(pred_scores[actual_bottom_mask]))
else:
avg_score_at_bottoms = 0.0
# "Actual tops" = candles with actual score < 15 (worst 15% buy times)
# "Actual tops" = candles with a low realized return-quality score.
actual_top_mask = actual_scores < 15
if np.any(actual_top_mask):
avg_score_at_tops = float(np.mean(pred_scores[actual_top_mask]))
@@ -1000,10 +998,7 @@ def compile_results(predictions, per_window_cost_improvement,
count = int(np.sum((pred_scores >= lo) & (pred_scores < (hi if hi < 100 else 101))))
score_distribution[key] = count
# --- Forward return approximation from actual scores ---
# Map actual score to approximate return quality
# Score 90+ = historically best 10% buys, score 10- = worst 10%
# Use actual score as proxy for "quality rank"
# --- Realized return-quality summary ---
if strong_buy_count > 0:
# Average actual quality score for strong buy signals
avg_quality_strong = float(np.mean(actual_scores[strong_buy_mask]))