DockerSpace / tests /test_multi_factor_finetune_report.py
DennisChan0909's picture
Backup current stock predictor strategies
ee37d63
Raw
History Blame Contribute Delete
14.5 kB
import json
import pandas as pd
from scripts.run_multi_factor_finetune_report import (
_hold_biases,
_score_candidate,
evaluate_configs,
generate_beam_refinement_configs,
generate_finetune_configs,
load_seed_config,
)
def test_protected_metrics_ranking_preserves_signal_quality_before_accuracy():
high_accuracy_with_precision_loss = _score_candidate(
{
"accuracy_delta_pp": 4.0,
"direction_accuracy_delta_pp": -0.1,
"buy_precision_delta_pp": -0.1,
"sell_precision_delta_pp": -0.1,
},
{"accuracy": 44.0},
1.0,
ranking_strategy="protected_metrics_first",
min_accuracy_delta_pp=1.0,
min_buy_precision_delta_pp=0.0,
min_direction_accuracy_delta_pp=0.0,
max_sell_precision_drop_pp=0.0,
)
protected_signal_quality = _score_candidate(
{
"accuracy_delta_pp": 0.5,
"direction_accuracy_delta_pp": 0.1,
"buy_precision_delta_pp": 0.1,
"sell_precision_delta_pp": 0.1,
},
{"accuracy": 40.5},
1.0,
ranking_strategy="protected_metrics_first",
min_accuracy_delta_pp=1.0,
min_buy_precision_delta_pp=0.0,
min_direction_accuracy_delta_pp=0.0,
max_sell_precision_drop_pp=0.0,
)
assert protected_signal_quality > high_accuracy_with_precision_loss
def test_protected_metrics_ranking_improves_weakest_gate_before_pass_count():
imbalanced = _score_candidate(
{
"accuracy_delta_pp": 3.0,
"direction_accuracy_delta_pp": 0.5,
"buy_precision_delta_pp": 1.1,
"sell_precision_delta_pp": -1.5,
},
{"accuracy": 43.0},
1.0,
ranking_strategy="protected_metrics_first",
min_accuracy_delta_pp=1.0,
min_buy_precision_delta_pp=1.0,
min_direction_accuracy_delta_pp=0.0,
max_sell_precision_drop_pp=0.0,
)
balanced = _score_candidate(
{
"accuracy_delta_pp": 2.0,
"direction_accuracy_delta_pp": 0.1,
"buy_precision_delta_pp": 0.4,
"sell_precision_delta_pp": -0.2,
},
{"accuracy": 42.0},
1.0,
ranking_strategy="protected_metrics_first",
min_accuracy_delta_pp=1.0,
min_buy_precision_delta_pp=1.0,
min_direction_accuracy_delta_pp=0.0,
max_sell_precision_drop_pp=0.0,
)
assert balanced > imbalanced
def test_protected_metrics_ranking_rejects_precision_gain_from_low_coverage():
common = {
"accuracy_delta_pp": 1.2,
"direction_accuracy_delta_pp": 0.2,
"buy_precision_delta_pp": 1.2,
"sell_precision_delta_pp": 0.2,
}
low_coverage = _score_candidate(
common,
{"accuracy": 41.2},
1.0,
signal_ratio=0.5,
ranking_strategy="protected_metrics_first",
min_accuracy_delta_pp=1.0,
min_buy_precision_delta_pp=1.0,
min_direction_accuracy_delta_pp=0.0,
max_sell_precision_drop_pp=0.0,
min_signal_ratio=0.7,
)
healthy_coverage = _score_candidate(
common,
{"accuracy": 41.2},
1.0,
signal_ratio=0.8,
ranking_strategy="protected_metrics_first",
min_accuracy_delta_pp=1.0,
min_buy_precision_delta_pp=1.0,
min_direction_accuracy_delta_pp=0.0,
max_sell_precision_drop_pp=0.0,
min_signal_ratio=0.7,
)
assert healthy_coverage > low_coverage
def test_load_seed_config_reads_golden_weights(tmp_path):
seed_json = tmp_path / "seed.json"
seed_json.write_text(
json.dumps(
{
"generated_at": "2026-05-23T00:00:00+00:00",
"result": {
"golden_config": {
"config": {
"weights": {
"oldwang_trend": 0.01,
"volume_spike": 0.08,
}
}
}
},
}
)
)
weights, payload = load_seed_config(seed_json)
assert weights == {"oldwang_trend": 0.01, "volume_spike": 0.08}
assert payload["generated_at"] == "2026-05-23T00:00:00+00:00"
def test_load_seed_config_can_use_top_overall_as_explicit_research_seed(tmp_path):
seed_json = tmp_path / "failed_search.json"
seed_json.write_text(
json.dumps(
{
"result": {
"golden_config": None,
"top_overall": [{"config": {"weights": {"trend": 0.02}}}],
}
}
)
)
weights, payload = load_seed_config(seed_json, selection="top_overall")
assert weights == {"trend": 0.02}
assert payload["_seed_selection"] == {
"requested": "top_overall",
"selected_from": "top_overall[0]",
"research_seed_only": True,
}
def test_generate_finetune_configs_keeps_every_selected_weight_nonzero():
configs = generate_finetune_configs(
{"oldwang_trend": 0.01, "volume_spike": 0.08, "trust_flow": 0.02},
delta_values=[-0.01, 0.0, 0.01],
scale_values=[0.9, 1.0, 1.1],
random_configs=20,
max_configs=30,
weight_step=0.005,
min_weight=0.005,
max_weight=0.10,
random_seed=7,
)
assert configs[0].weights == {"oldwang_trend": 0.01, "volume_spike": 0.08, "trust_flow": 0.02}
assert 1 < len(configs) <= 30
assert all(set(config.weights) == {"oldwang_trend", "volume_spike", "trust_flow"} for config in configs)
assert all(all(weight > 0 for weight in config.weights.values()) for config in configs)
assert all(all(0.005 <= weight <= 0.10 for weight in config.weights.values()) for config in configs)
def test_generate_finetune_configs_respects_zero_weight_limit():
configs = generate_finetune_configs(
{"trend": 0.0, "volume": 0.25, "chip": 0.0, "risk": 0.5},
delta_values=[-0.01, 0.0, 0.01],
scale_values=[1.0],
random_configs=20,
max_configs=30,
weight_step=0.01,
min_weight=0.0,
max_weight=1.0,
random_seed=7,
max_zero_factor_weights=1,
)
assert configs
assert all(
sum(1 for weight in cfg.weights.values() if abs(weight) <= 1e-12) <= 1
for cfg in configs
)
def test_generate_finetune_configs_lifts_excess_seed_zeros_before_random_search():
configs = generate_finetune_configs(
{"a": 0.0, "b": 0.0, "c": 0.0, "d": 0.0, "e": 0.5},
delta_values=[-0.01, 0.0, 0.01],
scale_values=[1.0],
random_configs=20,
max_configs=30,
weight_step=0.01,
min_weight=0.0,
max_weight=1.0,
random_seed=7,
max_zero_factor_weights=1,
)
assert len(configs) > 5
assert all(sum(1 for weight in cfg.weights.values() if weight == 0.0) <= 1 for cfg in configs)
assert configs[0].weights == {"a": 0.01, "b": 0.01, "c": 0.01, "d": 0.0, "e": 0.5}
def test_generate_finetune_configs_uses_local_mutations_for_many_low_weights():
seed = {f"factor_{idx}": 0.01 for idx in range(30)}
configs = generate_finetune_configs(
seed,
delta_values=[-0.01, 0.0, 0.01],
scale_values=[1.0],
random_configs=100,
max_configs=120,
weight_step=0.01,
min_weight=0.0,
max_weight=1.0,
random_seed=7,
max_zero_factor_weights=3,
)
assert len(configs) >= 100
assert all(sum(1 for weight in cfg.weights.values() if weight == 0.0) <= 3 for cfg in configs)
def test_generate_beam_refinement_configs_uses_cent_step_and_zero_limit():
rows = []
for idx in range(12):
bullish = idx % 2 == 0
rows.append(
{
"stock": "2330",
"date": f"2025-04-{idx + 1:02d}",
"y_true": 1 if bullish else -1,
"base_pred": 0,
"p_buy": 0.49,
"p_hold": 0.52,
"p_sell": 0.49,
"factor__trend": 2.0 if bullish else -2.0,
"factor__volume": 0.0,
}
)
configs = generate_beam_refinement_configs(
{"trend": 0.0, "volume": 0.25},
events=pd.DataFrame(rows),
optimize_ratio=0.5,
weight_step=0.01,
min_weight=0.0,
max_weight=1.0,
beam_width=3,
beam_passes=1,
max_zero_factor_weights=1,
)
assert configs
assert all(
all(abs((weight * 100) - round(weight * 100)) < 1e-9 for weight in cfg.weights.values())
for cfg in configs
)
assert all(
sum(1 for weight in cfg.weights.values() if abs(weight) <= 1e-12) <= 1
for cfg in configs
)
def test_generate_beam_refinement_configs_searches_hold_bias_threshold():
events = pd.DataFrame(
[
{
"stock": "2330",
"date": f"2025-05-{idx + 1:02d}",
"y_true": 0,
"base_pred": 1,
"p_buy": 0.51,
"p_hold": 0.50,
"p_sell": 0.0,
"factor__trend": 0.0,
}
for idx in range(6)
]
)
configs = generate_beam_refinement_configs(
{"trend": 0.01},
events=events,
optimize_ratio=0.5,
weight_step=0.01,
min_weight=0.0,
max_weight=0.01,
beam_width=2,
beam_passes=1,
max_zero_factor_weights=1,
hold_bias_values=[0.0, 0.02],
)
assert {cfg.hold_bias for cfg in configs} == {0.0, 0.02}
def test_hold_bias_values_must_be_inside_probability_range():
assert _hold_biases("0.0,0.05") == [0.0, 0.05]
try:
_hold_biases("1.01")
except ValueError as exc:
assert "inside [0, 1]" in str(exc)
else:
raise AssertionError("Expected invalid hold bias to fail")
def test_evaluate_configs_reports_current_seed_and_best_separately():
rows = []
for idx in range(12):
bullish = idx % 2 == 0
rows.append(
{
"stock": "2330" if idx < 6 else "2317",
"date": f"2025-04-{idx + 1:02d}",
"y_true": 1 if bullish else -1,
"base_pred": 0,
"p_buy": 0.49,
"p_hold": 0.52,
"p_sell": 0.49,
"factor__trend": 2.0 if bullish else -2.0,
"factor__volume": 0.0,
}
)
configs = generate_finetune_configs(
{"trend": 0.01, "volume": 0.01},
delta_values=[0.01],
scale_values=[1.0],
random_configs=0,
max_configs=10,
weight_step=0.005,
min_weight=0.005,
max_weight=0.10,
random_seed=1,
)
result = evaluate_configs(
pd.DataFrame(rows),
configs,
optimize_ratio=0.5,
min_accuracy_delta_pp=0.0,
min_buy_precision_delta_pp=0.0,
min_signal_ratio=0.5,
max_buy_count_ratio=None,
max_sell_precision_drop_pp=None,
max_zero_factor_weights=None,
top_n=5,
)
assert result["baseline"]["validation"]["accuracy"] == 0.0
assert result["ranking_source"] == "tune"
assert result["seed_config"]["name"] == "seed"
assert result["seed_config"]["validation"]["metrics"]["accuracy"] == 0.0
assert result["golden_config"]["validation"]["metrics"]["accuracy"] == 100.0
assert result["golden_config"]["config"]["weights"]["trend"] == 0.02
assert result["golden_config"]["gate"]["checks"]["all_factor_weights_nonzero"] is True
assert result["best_vs_seed_deltas"]["accuracy_delta_pp"] == 100.0
def test_evaluate_configs_reports_per_stock_for_top_candidate_without_golden():
rows = []
for idx in range(12):
bullish = idx % 2 == 0
rows.append(
{
"stock": "2330" if idx < 6 else "2317",
"date": f"2025-05-{idx + 1:02d}",
"y_true": 1 if bullish else -1,
"base_pred": 0,
"p_buy": 0.49,
"p_hold": 0.52,
"p_sell": 0.49,
"factor__trend": 2.0 if bullish else -2.0,
"factor__volume": 0.0,
}
)
configs = generate_finetune_configs(
{"trend": 0.01, "volume": 0.01},
delta_values=[0.01],
scale_values=[1.0],
random_configs=0,
max_configs=10,
weight_step=0.005,
min_weight=0.005,
max_weight=0.10,
random_seed=1,
)
result = evaluate_configs(
pd.DataFrame(rows),
configs,
optimize_ratio=0.5,
min_accuracy_delta_pp=101.0,
min_buy_precision_delta_pp=0.0,
min_signal_ratio=0.5,
max_buy_count_ratio=None,
max_sell_precision_drop_pp=None,
max_zero_factor_weights=None,
top_n=5,
)
assert result["golden_config"] is None
assert result["per_stock_validation_config_name"] == "candidate_00001"
assert result["per_stock_validation_config_passed"] is False
assert set(result["per_stock_validation"]) == {"2317"}
def test_evaluate_configs_can_retain_all_ranked_combinations():
rows = [
{
"stock": "2330",
"date": f"2025-06-{idx + 1:02d}",
"y_true": 1 if idx % 2 == 0 else -1,
"base_pred": 0,
"p_buy": 0.49,
"p_hold": 0.52,
"p_sell": 0.49,
"factor__trend": 2.0 if idx % 2 == 0 else -2.0,
}
for idx in range(12)
]
configs = generate_finetune_configs(
{"trend": 0.01},
delta_values=[0.01, 0.02],
scale_values=[1.0],
random_configs=0,
max_configs=10,
weight_step=0.01,
min_weight=0.01,
max_weight=0.10,
random_seed=1,
)
result = evaluate_configs(
pd.DataFrame(rows),
configs,
optimize_ratio=0.5,
min_accuracy_delta_pp=0.0,
min_buy_precision_delta_pp=0.0,
min_signal_ratio=0.5,
max_buy_count_ratio=None,
max_sell_precision_drop_pp=None,
max_zero_factor_weights=0,
top_n=1,
retain_all_configs=True,
)
assert len(result["all_configs"]) == len(configs)
assert len(result["top_overall"]) == 1