Files
sanguo_vnpy_v2/sanguo_factor/ic_gate.py
T

108 lines
4.1 KiB
Python

# sanguo_factor/ic_gate.py
"""IC 正交闸(D4,spec §4.8 ⑧)——新因子与在库逐列 IC≥0.99=值级换皮置疑.
搬运: RD-Agent factor_runner.py:46-61 deduplicate_new_factors(逐日截面
Pearson→mean→对在库 max→<0.99 保留).数据契约=run_batch_eval
factor_values_out 宽表 parquet({name}.parquet,datetime 索引×vt_symbol 列).
定位=报告件:不自动转移(裁决权在判定五问/人;graveyard 走官方 CLI+D5 回写).
"""
from __future__ import annotations
import json
import os
import pandas as pd
IC_THRESHOLD = 0.99
_MIN_DAYS = 20 # 共同日下限:少于此无统计意义
_MIN_SYMS = 10 # 单日截面标的下限
_MIN_CORR_DAYS = 20 # 有效 corr 日下限(P3-10):常数截面滤除后少数日均值不置疑
def _load_wide(path: str) -> pd.DataFrame | None:
if not os.path.exists(path):
return None
df = pd.read_parquet(path)
return df if not df.empty else None
def _cross_sectional_ic(a: pd.DataFrame, b: pd.DataFrame) -> tuple[float | None, int]:
"""逐交易日截面 Pearson→均值;返回 (均值 IC, 有效 corr 日数).
无共同日/截面太小→(None, 0);常数截面日 corr=NaN 滤除后计入有效日数
(P3-10:有效日数供报告层判 insufficient_days).
"""
days = a.index.intersection(b.index)
if len(days) < _MIN_DAYS:
return None, 0
syms = a.columns.intersection(b.columns)
if len(syms) < _MIN_SYMS:
return None, 0
a2, b2 = a.loc[days, syms], b.loc[days, syms]
corrs = []
for day in days:
c = a2.loc[day].corr(b2.loc[day])
if c == c: # NaN 过滤(常数截面)
corrs.append(c)
return (sum(corrs) / len(corrs) if corrs else None), len(corrs)
def orthogonality_report(values_dir: str, new_names: list[str],
library_names: list[str],
*, threshold: float = IC_THRESHOLD) -> dict:
"""逐新因子→对在库 max |IC|;json 形状 {threshold, checked, flagged,
skipped_missing_values, compared_pairs, insufficient_days}.
P2-9: checked < len(new_names)(缺件跳过)与 compared_pairs(实际比对
对数)如实计数——「查过且干净」与「根本没查成」可区分,系统性 eval
失败不再呈现绿灯(接线处 monthly_batch 据此 stderr 告警).
P3-10: 全部比对有效 corr 日数 <_MIN_CORR_DAYS 的新因子不进置疑清单,
入 insufficient_days 留痕.
"""
cache: dict[str, pd.DataFrame | None] = {}
def wide(name: str) -> pd.DataFrame | None:
if name not in cache:
cache[name] = _load_wide(
os.path.join(values_dir, f"{name}.parquet"))
return cache[name]
flagged, checked = [], 0
skipped = compared = 0
insufficient: list[str] = []
for new in new_names:
nf = wide(new)
if nf is None:
skipped += 1
continue
checked += 1
best_name, best_ic = "", None
judged = False # 至少一次比对达有效日数下限
for lib in library_names:
if lib == new:
continue
lf = wide(lib)
if lf is None:
continue
compared += 1
ic, n_valid = _cross_sectional_ic(nf, lf)
if n_valid < _MIN_CORR_DAYS:
continue
judged = True
if ic is not None and (best_ic is None or abs(ic) > abs(best_ic)):
best_name, best_ic = lib, ic
if best_ic is not None and abs(best_ic) >= threshold:
flagged.append({"factor": new, "vs": best_name,
"ic": round(best_ic, 4)})
elif not judged:
insufficient.append(new)
return {"threshold": threshold, "checked": checked, "flagged": flagged,
"skipped_missing_values": skipped, "compared_pairs": compared,
"insufficient_days": insufficient}
def write_report(out_path: str, report: dict) -> None:
os.makedirs(os.path.dirname(out_path), exist_ok=True)
with open(out_path, "w", encoding="utf-8") as f:
json.dump(report, f, ensure_ascii=False, indent=1)