Files

172 lines
7.1 KiB
Python

# tests/factor/test_exam_gate.py
"""考场一键化(#70③): 候选+基准同批评测+建议报告(promotion 永不自动化)."""
import pytest
from sanguo_factor.exam_gate import judge_exam, render_report, main
def _row(icir_by_h, factor="x"):
return {"factor": factor, "metrics": {
h: {"t_stat": 3.0, "ic_mean": 0.05, "icir": v, "count": 200}
for h, v in icir_by_h.items()}}
def test_judge_all_win_suggest_promote():
j = judge_exam(_row({"1": 0.30, "5": 0.40, "10": 0.35}),
_row({"1": 0.28, "5": 0.38, "10": 0.35})) # 10d 打平
assert j["verdict"] == "SUGGEST_PROMOTE"
assert (j["wins"], j["losses"], j["compared"]) == (2, 0, 3)
def test_judge_any_loss_no_promote():
j = judge_exam(_row({"1": 0.30, "5": 0.36, "10": 0.35}),
_row({"1": 0.28, "5": 0.38, "10": 0.35}))
assert j["verdict"] == "NO_PROMOTE" and j["losses"] == 1
# (P3-16 后缺周期不再跳过不计,原 test_judge_missing_horizon_skipped 的
# SUGGEST_PROMOTE 语义由 test_judge_missing_any_horizon_insufficient_periods 取代)
def test_judge_no_overlap_data_insufficient():
j = judge_exam({"factor": "c", "metrics": {}},
_row({"1": 0.28}))
assert j["verdict"] == "DATA_INSUFFICIENT" and j["compared"] == 0
def test_render_report_contains_table_and_constitution_note():
j = judge_exam(_row({"1": 0.30}), _row({"1": 0.28}))
md = render_report("cand_x", "composite_quant12_v2a",
"2021-07-01", "2026-10-03", j)
assert "cand_x" in md and "composite_quant12_v2a" in md
assert "| 1d |" in md and "永不自动化" in md
def test_main_unresolvable_candidate_exits_2(tmp_path, capsys, monkeypatch):
monkeypatch.setattr("sanguo_factor.exam_gate.get_factor", lambda n: None)
rc = main(["--candidate", "nope_factor",
"--eval-db", str(tmp_path / "e.db")])
assert rc == 2 and "不可解析" in capsys.readouterr().err
def test_main_runs_candidate_and_baseline_same_run(tmp_path, capsys, monkeypatch):
calls = {}
def fake_run(names, start, end, eval_db, label, **kw):
calls["names"], calls["label"] = list(names), label
return {"run_id": "r1", "factors_total": 2, "factors_done": 2}
def fake_rows(db, run_id):
return [_row({"1": 0.30, "5": 0.40, "10": 0.35}, factor="cand_x"),
_row({"1": 0.28, "5": 0.38, "10": 0.30}, factor="base_y")]
import sanguo_factor.exam_gate as eg
monkeypatch.setattr(eg, "get_factor", lambda n: {"expr": "x"})
rc = main(["--candidate", "cand_x", "--baseline", "base_y",
"--end", "2026-10-03",
"--eval-db", str(tmp_path / "e.db"),
"--report-dir", str(tmp_path)],
run=fake_run, fetch_rows=fake_rows)
assert rc == 0
assert sorted(calls["names"]) == ["base_y", "cand_x"] # 同批铁律
assert calls["label"] == "exam_cand_x_2026-10-03" # eval run label 不变(P3-17 只动报告件名)
assert len(list(tmp_path.glob("exam_cand_x_2026-10-03_??????.md"))) == 1
assert "建议晋级" in capsys.readouterr().out
# —— 审计 P3-16: 缺任一周期→INSUFFICIENT_PERIODS fail-closed(单周期对比不再晋级) ——
def test_judge_missing_any_horizon_insufficient_periods():
j = judge_exam(_row({"1": 0.30, "5": 0.40}), # 候选缺 10d
_row({"1": 0.28, "5": 0.38, "10": 0.35}))
assert j["verdict"] == "INSUFFICIENT_PERIODS"
assert j["missing_periods"] == ["10"]
assert (j["wins"], j["compared"]) == (2, 2) # 有对比分但判定 fail-closed,不作晋级依据
def test_judge_missing_on_base_side_also_fails_closed():
j = judge_exam(_row({"1": 0.30, "5": 0.40, "10": 0.35}),
_row({"1": 0.28, "5": 0.38})) # 基准缺 10d
assert j["verdict"] == "INSUFFICIENT_PERIODS" and j["missing_periods"] == ["10"]
def test_render_report_names_missing_periods():
j = judge_exam(_row({"1": 0.30, "5": 0.40}),
_row({"1": 0.28, "5": 0.38, "10": 0.35}))
md = render_report("cand_x", "base_y", "2021-07-01", "2026-10-03", j)
assert "周期覆盖不足" in md and "缺周期" in md and "10d" in md # 明示缺哪期
# —— 审计 P3-17: 报告文件名带 HHMMSS,同日重跑不覆盖既有报告 ——
class _FakeStamp:
def __init__(self, n):
self._n = n
def strftime(self, fmt): # exam_gate 只用 %H%M%S
return f"1200{self._n:02d}" if fmt == "%H%M%S" else "000000"
def isoformat(self, timespec=None):
return f"2026-10-05T12:00:0{self._n}"
class _FakeDT:
seq = 0
@staticmethod
def now():
_FakeDT.seq += 1
return _FakeStamp(_FakeDT.seq)
def test_main_rerun_same_day_keeps_both_reports(tmp_path, capsys, monkeypatch):
import sanguo_factor.exam_gate as eg
monkeypatch.setattr(eg, "datetime", _FakeDT)
def fake_run(names, start, end, eval_db, label, **kw):
return {"run_id": "r1", "factors_total": 2, "factors_done": 2}
def fake_rows(db, run_id):
return [_row({"1": 0.30, "5": 0.40, "10": 0.35}, factor="cand_x"),
_row({"1": 0.28, "5": 0.38, "10": 0.30}, factor="base_y")]
monkeypatch.setattr(eg, "get_factor", lambda n: {"expr": "x"})
argv = ["--candidate", "cand_x", "--baseline", "base_y",
"--end", "2026-10-03", "--eval-db", str(tmp_path / "e.db"),
"--report-dir", str(tmp_path)]
assert main(argv, run=fake_run, fetch_rows=fake_rows) == 0
assert main(argv, run=fake_run, fetch_rows=fake_rows) == 0
reports = sorted(tmp_path.glob("exam_cand_x_2026-10-03_??????.md"))
assert len(reports) == 2 # 两跑两件,不覆盖
bodies = [p.read_text(encoding="utf-8") for p in reports]
assert len(set(bodies)) == 2 # 生成时刻不同,内容确不同件
# —— #91⑤: NaN ICIR fail-closed(原 NaN 无声计入 compared 不判胜负,可搭
# SUGGEST_PROMOTE 顺风车;现与 None 同判为缺失) ——
# (helper 名用 _icir_row 而非任务书的 _row:后者会遮蔽顶部带 factor 参数的
# 同名 helper,连崩存量 main 系测试,实测已证)
def _icir_row(icir_by_h):
return {"metrics": {h: {"icir": v} for h, v in icir_by_h.items()}}
def test_judge_nan_icir_counts_missing_not_compared():
judged = judge_exam(_icir_row({"1": 0.30, "5": float("nan"), "10": 0.20}),
_icir_row({"1": 0.10, "5": 0.10, "10": 0.10}))
assert judged["verdict"] == "INSUFFICIENT_PERIODS"
assert judged["missing_periods"] == ["5"]
assert judged["compared"] == 2 and judged["wins"] == 2
def test_judge_nan_on_base_side_also_fails_closed():
judged = judge_exam(_icir_row({"1": 0.3, "5": 0.2, "10": 0.1}),
_icir_row({"1": 0.1, "5": float("nan"), "10": 0.0}))
assert judged["verdict"] == "INSUFFICIENT_PERIODS"
assert judged["missing_periods"] == ["5"]
def test_judge_all_nan_data_insufficient():
nan = float("nan")
judged = judge_exam(_icir_row({"1": nan, "5": nan, "10": nan}),
_icir_row({"1": nan, "5": nan, "10": nan}))
assert judged["verdict"] == "DATA_INSUFFICIENT"
assert judged["compared"] == 0