feat(factor): IC 正交闸(D4)——逐日截面 Pearson 对在库 max≥0.99 置疑(报告件不自动转移)+monthly_batch values 导出接线;origin=decomposer 为新因子判定 #84 [vps] [no-doc]
Co-Authored-By: Claude Code <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,79 @@
|
||||
# sanguo_factor/ic_gate.py
|
||||
"""IC 正交闸(D4,spec §4.8 ⑧)——新因子与在库逐列 IC≥0.99=值级换皮置疑.
|
||||
|
||||
搬运: RD-Agent factor_runner.py:46-61 deduplicate_new_factors(逐日截面
|
||||
Pearson→mean→对在库 max→<0.99 保留).数据契约=run_batch_eval
|
||||
factor_values_out 宽表 parquet({name}.parquet,datetime 索引×vt_symbol 列).
|
||||
定位=报告件:不自动转移(裁决权在判定五问/人;graveyard 走官方 CLI+D5 回写).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
import pandas as pd
|
||||
|
||||
IC_THRESHOLD = 0.99
|
||||
_MIN_DAYS = 20 # 共同日下限:少于此无统计意义
|
||||
_MIN_SYMS = 10 # 单日截面标的下限
|
||||
|
||||
|
||||
def _load_wide(path: str) -> pd.DataFrame | None:
|
||||
if not os.path.exists(path):
|
||||
return None
|
||||
df = pd.read_parquet(path)
|
||||
return df if not df.empty else None
|
||||
|
||||
|
||||
def _cross_sectional_ic(a: pd.DataFrame, b: pd.DataFrame) -> float | None:
|
||||
"""逐交易日截面 Pearson→均值;无共同日/截面太小→None."""
|
||||
days = a.index.intersection(b.index)
|
||||
if len(days) < _MIN_DAYS:
|
||||
return None
|
||||
syms = a.columns.intersection(b.columns)
|
||||
if len(syms) < _MIN_SYMS:
|
||||
return None
|
||||
a2, b2 = a.loc[days, syms], b.loc[days, syms]
|
||||
corrs = []
|
||||
for day in days:
|
||||
c = a2.loc[day].corr(b2.loc[day])
|
||||
if c == c: # NaN 过滤(常数截面)
|
||||
corrs.append(c)
|
||||
return sum(corrs) / len(corrs) if corrs else None
|
||||
|
||||
|
||||
def orthogonality_report(values_dir: str, new_names: list[str],
|
||||
library_names: list[str],
|
||||
*, threshold: float = IC_THRESHOLD) -> dict:
|
||||
"""逐新因子→对在库 max |IC|;json 形状 {threshold, checked, flagged}."""
|
||||
cache: dict[str, pd.DataFrame | None] = {}
|
||||
|
||||
def wide(name: str) -> pd.DataFrame | None:
|
||||
if name not in cache:
|
||||
cache[name] = _load_wide(
|
||||
os.path.join(values_dir, f"{name}.parquet"))
|
||||
return cache[name]
|
||||
|
||||
flagged, checked = [], 0
|
||||
for new in new_names:
|
||||
nf = wide(new)
|
||||
if nf is None:
|
||||
continue
|
||||
checked += 1
|
||||
best_name, best_ic = "", None
|
||||
for lib in library_names:
|
||||
if lib == new:
|
||||
continue
|
||||
ic = _cross_sectional_ic(nf, wide(lib)) if wide(lib) is not None else None
|
||||
if ic is not None and (best_ic is None or abs(ic) > abs(best_ic)):
|
||||
best_name, best_ic = lib, ic
|
||||
if best_ic is not None and abs(best_ic) >= threshold:
|
||||
flagged.append({"factor": new, "vs": best_name,
|
||||
"ic": round(best_ic, 4)})
|
||||
return {"threshold": threshold, "checked": checked, "flagged": flagged}
|
||||
|
||||
|
||||
def write_report(out_path: str, report: dict) -> None:
|
||||
os.makedirs(os.path.dirname(out_path), exist_ok=True)
|
||||
with open(out_path, "w", encoding="utf-8") as f:
|
||||
json.dump(report, f, ensure_ascii=False, indent=1)
|
||||
@@ -91,6 +91,8 @@ def main(argv: list[str] | None = None) -> int:
|
||||
"config/factor_registry.yaml"))
|
||||
ap.add_argument("--eval-db", default=None, help="缺省 eval_store.default_eval_db_path()")
|
||||
ap.add_argument("--window-months", type=int, default=12)
|
||||
ap.add_argument("--report-dir", default="reports/factor_monthly",
|
||||
help="values 导出与 IC 正交闸报告目录")
|
||||
args = ap.parse_args(argv)
|
||||
registry = load_registry(args.registry)
|
||||
plan = plan_batch(registry, args.as_of, args.window_months,
|
||||
@@ -104,9 +106,24 @@ def main(argv: list[str] | None = None) -> int:
|
||||
f"{plan['skipped_unregistered']}", file=sys.stderr)
|
||||
print(f"[monthly_batch] {len(plan['factor_names'])} 因子 "
|
||||
f"{plan['start']}~{plan['end']} → {db} (label={plan['label']})")
|
||||
values_dir = os.path.join(args.report_dir, "values", plan["label"])
|
||||
summary = run_batch_eval(plan["factor_names"], plan["start"], plan["end"],
|
||||
db, plan["label"])
|
||||
db, plan["label"], factor_values_out=values_dir)
|
||||
print(f"[monthly_batch] 完成: {summary}")
|
||||
# IC 正交闸(D4): 新因子=origin decomposer;报告件不自动转移
|
||||
new_names = [n for n, e in registry["factors"].items()
|
||||
if ((e.get("versions") or [{}])[-1].get("params") or {}
|
||||
).get("origin") == "decomposer"
|
||||
and n in plan["factor_names"]]
|
||||
if new_names:
|
||||
from sanguo_factor.ic_gate import orthogonality_report, write_report
|
||||
rep = orthogonality_report(values_dir, new_names, plan["factor_names"])
|
||||
write_report(os.path.join(args.report_dir,
|
||||
f"ic_gate_{plan['label']}.json"), rep)
|
||||
if rep["flagged"]:
|
||||
print(f"[monthly_batch] ⚠️ IC 正交闸: "
|
||||
+ ", ".join(f"{f['factor']}(vs {f['vs']} ic={f['ic']})"
|
||||
for f in rep["flagged"]))
|
||||
return 0
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
# tests/factor/test_ic_gate.py
|
||||
"""IC 正交闸(D4):逐日截面 Pearson→mean→对在库 max→≥0.99 置疑(报告件)."""
|
||||
import json
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from sanguo_factor.ic_gate import orthogonality_report, write_report
|
||||
|
||||
|
||||
def _wide(rng, n_days=25, n_syms=12, base=None):
|
||||
idx = pd.date_range("2026-09-01", periods=n_days, freq="D")
|
||||
cols = [f"{i}.SSE" for i in range(n_syms)]
|
||||
if base is None:
|
||||
vals = rng.normal(size=(n_days, n_syms))
|
||||
else:
|
||||
vals = base + rng.normal(scale=1e-6, size=(n_days, n_syms))
|
||||
return pd.DataFrame(vals, index=idx, columns=cols)
|
||||
|
||||
|
||||
def test_flagged_when_identical(tmp_path):
|
||||
rng = np.random.default_rng(7)
|
||||
dup = _wide(rng)
|
||||
for name, frame in [("llm_new", dup), ("f_lib", dup),
|
||||
("f_other", _wide(rng))]:
|
||||
frame.to_parquet(tmp_path / f"{name}.parquet")
|
||||
rep = orthogonality_report(str(tmp_path), ["llm_new"],
|
||||
["llm_new", "f_lib", "f_other"])
|
||||
assert rep["checked"] == 1
|
||||
assert len(rep["flagged"]) == 1
|
||||
assert rep["flagged"][0]["factor"] == "llm_new"
|
||||
assert rep["flagged"][0]["vs"] == "f_lib"
|
||||
assert rep["flagged"][0]["ic"] > 0.99
|
||||
|
||||
|
||||
def test_clean_when_uncorrelated(tmp_path):
|
||||
rng = np.random.default_rng(11)
|
||||
a = _wide(rng)
|
||||
b = rng.permutation(a.to_numpy()).reshape(a.shape) # 打乱≈去相关
|
||||
a.to_parquet(tmp_path / "llm_new.parquet")
|
||||
pd.DataFrame(b, index=a.index, columns=a.columns).to_parquet(
|
||||
tmp_path / "f_lib.parquet")
|
||||
rep = orthogonality_report(str(tmp_path), ["llm_new"], ["f_lib"])
|
||||
assert rep["flagged"] == []
|
||||
|
||||
|
||||
def test_missing_parquet_skipped_and_write_report(tmp_path):
|
||||
(tmp_path / "f_lib.parquet").unlink(missing_ok=True)
|
||||
rep = orthogonality_report(str(tmp_path), ["llm_new"], [])
|
||||
assert rep["checked"] == 0 and rep["flagged"] == []
|
||||
out = tmp_path / "sub" / "ic.json"
|
||||
write_report(str(out), rep)
|
||||
assert json.loads(out.read_text())["threshold"] == 0.99
|
||||
@@ -1,5 +1,7 @@
|
||||
# tests/factor/test_monthly_batch.py
|
||||
"""先批后评·第一段 TDD——决议 J:跑批触发从「人点」变「日历」."""
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from sanguo_factor import monthly_batch
|
||||
@@ -51,7 +53,7 @@ def test_plan_batch_excludes_dead_and_labels(registry):
|
||||
def test_main_invokes_run_batch_eval_with_plan(monkeypatch, registry, tmp_path):
|
||||
calls = {}
|
||||
|
||||
def fake_run(factor_names, start, end, eval_db, label):
|
||||
def fake_run(factor_names, start, end, eval_db, label, **kw):
|
||||
calls.update(names=factor_names, start=start, end=end, label=label)
|
||||
return {"run_id": "x"}
|
||||
|
||||
@@ -112,3 +114,50 @@ def test_plan_batch_picks_up_yaml_only_factor():
|
||||
plan = mb.plan_batch(reg, "2026-10-31", resolver=mb.resolve_with_yaml(reg))
|
||||
assert "llm_x" in plan["factor_names"]
|
||||
assert plan["skipped_unregistered"] == ["无表达式者"]
|
||||
|
||||
|
||||
def test_main_exports_values_and_writes_ic_gate(tmp_path, monkeypatch):
|
||||
import json
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import yaml
|
||||
import sanguo_factor # noqa: F401 内存库注册(供 fund 因子名解析)
|
||||
import sanguo_factor.monthly_batch as mb
|
||||
|
||||
reg_path = tmp_path / "reg.yaml"
|
||||
reg_path.write_text(yaml.safe_dump({"factors": {
|
||||
"llm_x": {"name": "llm_x", "hypothesis": "hyp-1", "status": "incubating",
|
||||
"versions": [{"v": 1, "commit": "decompose", "params": {
|
||||
"expression": "cs_rank(close)", "source": "bars_daily",
|
||||
"origin": "decomposer"}}]},
|
||||
# 手写库真名同批在库(闸对照腿;import sanguo_factor 即已注册可解析)
|
||||
"fund_gross_margin": {
|
||||
"name": "fund_gross_margin", "hypothesis": "hyp-0",
|
||||
"status": "incubating",
|
||||
"versions": [{"v": 1, "commit": "hand", "params": {}}]}}}),
|
||||
encoding="utf-8")
|
||||
calls = {}
|
||||
|
||||
def fake_run(factor_names, start, end, eval_db, label, **kw):
|
||||
calls.update(kw, names=factor_names)
|
||||
vdir = kw["factor_values_out"]
|
||||
os.makedirs(vdir, exist_ok=True)
|
||||
rng = np.random.default_rng(3)
|
||||
w = pd.DataFrame(rng.normal(size=(25, 12)),
|
||||
index=pd.date_range("2026-09-01", periods=25),
|
||||
columns=[f"{i}.SSE" for i in range(12)])
|
||||
for n in ("llm_x", "fund_gross_margin"):
|
||||
w.to_parquet(os.path.join(vdir, f"{n}.parquet"))
|
||||
return {"done": 2}
|
||||
|
||||
monkeypatch.setattr(mb, "run_batch_eval", fake_run)
|
||||
rc = mb.main(["--as-of", "2026-10-31", "--registry", str(reg_path),
|
||||
"--eval-db", str(tmp_path / "e.db"),
|
||||
"--report-dir", str(tmp_path / "rep")])
|
||||
assert rc == 0
|
||||
assert calls["names"] == ["fund_gross_margin", "llm_x"] # D2:yaml 因子进批
|
||||
rep = json.loads((tmp_path / "rep" / "ic_gate_monthly_2026-10.json"
|
||||
).read_text())
|
||||
assert rep["checked"] == 1 and rep["flagged"][0]["factor"] == "llm_x"
|
||||
assert rep["flagged"][0]["vs"] == "fund_gross_margin"
|
||||
|
||||
Reference in New Issue
Block a user