172 lines
7.1 KiB
Python
172 lines
7.1 KiB
Python
# tests/factor/test_exam_gate.py
|
|
"""考场一键化(#70③): 候选+基准同批评测+建议报告(promotion 永不自动化)."""
|
|
import pytest
|
|
|
|
from sanguo_factor.exam_gate import judge_exam, render_report, main
|
|
|
|
|
|
def _row(icir_by_h, factor="x"):
|
|
return {"factor": factor, "metrics": {
|
|
h: {"t_stat": 3.0, "ic_mean": 0.05, "icir": v, "count": 200}
|
|
for h, v in icir_by_h.items()}}
|
|
|
|
|
|
def test_judge_all_win_suggest_promote():
|
|
j = judge_exam(_row({"1": 0.30, "5": 0.40, "10": 0.35}),
|
|
_row({"1": 0.28, "5": 0.38, "10": 0.35})) # 10d 打平
|
|
assert j["verdict"] == "SUGGEST_PROMOTE"
|
|
assert (j["wins"], j["losses"], j["compared"]) == (2, 0, 3)
|
|
|
|
|
|
def test_judge_any_loss_no_promote():
|
|
j = judge_exam(_row({"1": 0.30, "5": 0.36, "10": 0.35}),
|
|
_row({"1": 0.28, "5": 0.38, "10": 0.35}))
|
|
assert j["verdict"] == "NO_PROMOTE" and j["losses"] == 1
|
|
|
|
|
|
# (P3-16 后缺周期不再跳过不计,原 test_judge_missing_horizon_skipped 的
|
|
# SUGGEST_PROMOTE 语义由 test_judge_missing_any_horizon_insufficient_periods 取代)
|
|
|
|
|
|
def test_judge_no_overlap_data_insufficient():
|
|
j = judge_exam({"factor": "c", "metrics": {}},
|
|
_row({"1": 0.28}))
|
|
assert j["verdict"] == "DATA_INSUFFICIENT" and j["compared"] == 0
|
|
|
|
|
|
def test_render_report_contains_table_and_constitution_note():
|
|
j = judge_exam(_row({"1": 0.30}), _row({"1": 0.28}))
|
|
md = render_report("cand_x", "composite_quant12_v2a",
|
|
"2021-07-01", "2026-10-03", j)
|
|
assert "cand_x" in md and "composite_quant12_v2a" in md
|
|
assert "| 1d |" in md and "永不自动化" in md
|
|
|
|
|
|
def test_main_unresolvable_candidate_exits_2(tmp_path, capsys, monkeypatch):
|
|
monkeypatch.setattr("sanguo_factor.exam_gate.get_factor", lambda n: None)
|
|
rc = main(["--candidate", "nope_factor",
|
|
"--eval-db", str(tmp_path / "e.db")])
|
|
assert rc == 2 and "不可解析" in capsys.readouterr().err
|
|
|
|
|
|
def test_main_runs_candidate_and_baseline_same_run(tmp_path, capsys, monkeypatch):
|
|
calls = {}
|
|
|
|
def fake_run(names, start, end, eval_db, label, **kw):
|
|
calls["names"], calls["label"] = list(names), label
|
|
return {"run_id": "r1", "factors_total": 2, "factors_done": 2}
|
|
|
|
def fake_rows(db, run_id):
|
|
return [_row({"1": 0.30, "5": 0.40, "10": 0.35}, factor="cand_x"),
|
|
_row({"1": 0.28, "5": 0.38, "10": 0.30}, factor="base_y")]
|
|
|
|
import sanguo_factor.exam_gate as eg
|
|
monkeypatch.setattr(eg, "get_factor", lambda n: {"expr": "x"})
|
|
rc = main(["--candidate", "cand_x", "--baseline", "base_y",
|
|
"--end", "2026-10-03",
|
|
"--eval-db", str(tmp_path / "e.db"),
|
|
"--report-dir", str(tmp_path)],
|
|
run=fake_run, fetch_rows=fake_rows)
|
|
assert rc == 0
|
|
assert sorted(calls["names"]) == ["base_y", "cand_x"] # 同批铁律
|
|
assert calls["label"] == "exam_cand_x_2026-10-03" # eval run label 不变(P3-17 只动报告件名)
|
|
assert len(list(tmp_path.glob("exam_cand_x_2026-10-03_??????.md"))) == 1
|
|
assert "建议晋级" in capsys.readouterr().out
|
|
|
|
|
|
# —— 审计 P3-16: 缺任一周期→INSUFFICIENT_PERIODS fail-closed(单周期对比不再晋级) ——
|
|
def test_judge_missing_any_horizon_insufficient_periods():
|
|
j = judge_exam(_row({"1": 0.30, "5": 0.40}), # 候选缺 10d
|
|
_row({"1": 0.28, "5": 0.38, "10": 0.35}))
|
|
assert j["verdict"] == "INSUFFICIENT_PERIODS"
|
|
assert j["missing_periods"] == ["10"]
|
|
assert (j["wins"], j["compared"]) == (2, 2) # 有对比分但判定 fail-closed,不作晋级依据
|
|
|
|
|
|
def test_judge_missing_on_base_side_also_fails_closed():
|
|
j = judge_exam(_row({"1": 0.30, "5": 0.40, "10": 0.35}),
|
|
_row({"1": 0.28, "5": 0.38})) # 基准缺 10d
|
|
assert j["verdict"] == "INSUFFICIENT_PERIODS" and j["missing_periods"] == ["10"]
|
|
|
|
|
|
def test_render_report_names_missing_periods():
|
|
j = judge_exam(_row({"1": 0.30, "5": 0.40}),
|
|
_row({"1": 0.28, "5": 0.38, "10": 0.35}))
|
|
md = render_report("cand_x", "base_y", "2021-07-01", "2026-10-03", j)
|
|
assert "周期覆盖不足" in md and "缺周期" in md and "10d" in md # 明示缺哪期
|
|
|
|
|
|
# —— 审计 P3-17: 报告文件名带 HHMMSS,同日重跑不覆盖既有报告 ——
|
|
class _FakeStamp:
|
|
def __init__(self, n):
|
|
self._n = n
|
|
|
|
def strftime(self, fmt): # exam_gate 只用 %H%M%S
|
|
return f"1200{self._n:02d}" if fmt == "%H%M%S" else "000000"
|
|
|
|
def isoformat(self, timespec=None):
|
|
return f"2026-10-05T12:00:0{self._n}"
|
|
|
|
|
|
class _FakeDT:
|
|
seq = 0
|
|
|
|
@staticmethod
|
|
def now():
|
|
_FakeDT.seq += 1
|
|
return _FakeStamp(_FakeDT.seq)
|
|
|
|
|
|
def test_main_rerun_same_day_keeps_both_reports(tmp_path, capsys, monkeypatch):
|
|
import sanguo_factor.exam_gate as eg
|
|
monkeypatch.setattr(eg, "datetime", _FakeDT)
|
|
|
|
def fake_run(names, start, end, eval_db, label, **kw):
|
|
return {"run_id": "r1", "factors_total": 2, "factors_done": 2}
|
|
|
|
def fake_rows(db, run_id):
|
|
return [_row({"1": 0.30, "5": 0.40, "10": 0.35}, factor="cand_x"),
|
|
_row({"1": 0.28, "5": 0.38, "10": 0.30}, factor="base_y")]
|
|
|
|
monkeypatch.setattr(eg, "get_factor", lambda n: {"expr": "x"})
|
|
argv = ["--candidate", "cand_x", "--baseline", "base_y",
|
|
"--end", "2026-10-03", "--eval-db", str(tmp_path / "e.db"),
|
|
"--report-dir", str(tmp_path)]
|
|
assert main(argv, run=fake_run, fetch_rows=fake_rows) == 0
|
|
assert main(argv, run=fake_run, fetch_rows=fake_rows) == 0
|
|
reports = sorted(tmp_path.glob("exam_cand_x_2026-10-03_??????.md"))
|
|
assert len(reports) == 2 # 两跑两件,不覆盖
|
|
bodies = [p.read_text(encoding="utf-8") for p in reports]
|
|
assert len(set(bodies)) == 2 # 生成时刻不同,内容确不同件
|
|
|
|
|
|
# —— #91⑤: NaN ICIR fail-closed(原 NaN 无声计入 compared 不判胜负,可搭
|
|
# SUGGEST_PROMOTE 顺风车;现与 None 同判为缺失) ——
|
|
# (helper 名用 _icir_row 而非任务书的 _row:后者会遮蔽顶部带 factor 参数的
|
|
# 同名 helper,连崩存量 main 系测试,实测已证)
|
|
def _icir_row(icir_by_h):
|
|
return {"metrics": {h: {"icir": v} for h, v in icir_by_h.items()}}
|
|
|
|
|
|
def test_judge_nan_icir_counts_missing_not_compared():
|
|
judged = judge_exam(_icir_row({"1": 0.30, "5": float("nan"), "10": 0.20}),
|
|
_icir_row({"1": 0.10, "5": 0.10, "10": 0.10}))
|
|
assert judged["verdict"] == "INSUFFICIENT_PERIODS"
|
|
assert judged["missing_periods"] == ["5"]
|
|
assert judged["compared"] == 2 and judged["wins"] == 2
|
|
|
|
|
|
def test_judge_nan_on_base_side_also_fails_closed():
|
|
judged = judge_exam(_icir_row({"1": 0.3, "5": 0.2, "10": 0.1}),
|
|
_icir_row({"1": 0.1, "5": float("nan"), "10": 0.0}))
|
|
assert judged["verdict"] == "INSUFFICIENT_PERIODS"
|
|
assert judged["missing_periods"] == ["5"]
|
|
|
|
|
|
def test_judge_all_nan_data_insufficient():
|
|
nan = float("nan")
|
|
judged = judge_exam(_icir_row({"1": nan, "5": nan, "10": nan}),
|
|
_icir_row({"1": nan, "5": nan, "10": nan}))
|
|
assert judged["verdict"] == "DATA_INSUFFICIENT"
|
|
assert judged["compared"] == 0
|