108 lines
4.1 KiB
Python
108 lines
4.1 KiB
Python
# sanguo_factor/ic_gate.py
|
|
"""IC 正交闸(D4,spec §4.8 ⑧)——新因子与在库逐列 IC≥0.99=值级换皮置疑.
|
|
|
|
搬运: RD-Agent factor_runner.py:46-61 deduplicate_new_factors(逐日截面
|
|
Pearson→mean→对在库 max→<0.99 保留).数据契约=run_batch_eval
|
|
factor_values_out 宽表 parquet({name}.parquet,datetime 索引×vt_symbol 列).
|
|
定位=报告件:不自动转移(裁决权在判定五问/人;graveyard 走官方 CLI+D5 回写).
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
|
|
import pandas as pd
|
|
|
|
IC_THRESHOLD = 0.99
|
|
_MIN_DAYS = 20 # 共同日下限:少于此无统计意义
|
|
_MIN_SYMS = 10 # 单日截面标的下限
|
|
_MIN_CORR_DAYS = 20 # 有效 corr 日下限(P3-10):常数截面滤除后少数日均值不置疑
|
|
|
|
|
|
def _load_wide(path: str) -> pd.DataFrame | None:
|
|
if not os.path.exists(path):
|
|
return None
|
|
df = pd.read_parquet(path)
|
|
return df if not df.empty else None
|
|
|
|
|
|
def _cross_sectional_ic(a: pd.DataFrame, b: pd.DataFrame) -> tuple[float | None, int]:
|
|
"""逐交易日截面 Pearson→均值;返回 (均值 IC, 有效 corr 日数).
|
|
|
|
无共同日/截面太小→(None, 0);常数截面日 corr=NaN 滤除后计入有效日数
|
|
(P3-10:有效日数供报告层判 insufficient_days).
|
|
"""
|
|
days = a.index.intersection(b.index)
|
|
if len(days) < _MIN_DAYS:
|
|
return None, 0
|
|
syms = a.columns.intersection(b.columns)
|
|
if len(syms) < _MIN_SYMS:
|
|
return None, 0
|
|
a2, b2 = a.loc[days, syms], b.loc[days, syms]
|
|
corrs = []
|
|
for day in days:
|
|
c = a2.loc[day].corr(b2.loc[day])
|
|
if c == c: # NaN 过滤(常数截面)
|
|
corrs.append(c)
|
|
return (sum(corrs) / len(corrs) if corrs else None), len(corrs)
|
|
|
|
|
|
def orthogonality_report(values_dir: str, new_names: list[str],
|
|
library_names: list[str],
|
|
*, threshold: float = IC_THRESHOLD) -> dict:
|
|
"""逐新因子→对在库 max |IC|;json 形状 {threshold, checked, flagged,
|
|
skipped_missing_values, compared_pairs, insufficient_days}.
|
|
|
|
P2-9: checked < len(new_names)(缺件跳过)与 compared_pairs(实际比对
|
|
对数)如实计数——「查过且干净」与「根本没查成」可区分,系统性 eval
|
|
失败不再呈现绿灯(接线处 monthly_batch 据此 stderr 告警).
|
|
P3-10: 全部比对有效 corr 日数 <_MIN_CORR_DAYS 的新因子不进置疑清单,
|
|
入 insufficient_days 留痕.
|
|
"""
|
|
cache: dict[str, pd.DataFrame | None] = {}
|
|
|
|
def wide(name: str) -> pd.DataFrame | None:
|
|
if name not in cache:
|
|
cache[name] = _load_wide(
|
|
os.path.join(values_dir, f"{name}.parquet"))
|
|
return cache[name]
|
|
|
|
flagged, checked = [], 0
|
|
skipped = compared = 0
|
|
insufficient: list[str] = []
|
|
for new in new_names:
|
|
nf = wide(new)
|
|
if nf is None:
|
|
skipped += 1
|
|
continue
|
|
checked += 1
|
|
best_name, best_ic = "", None
|
|
judged = False # 至少一次比对达有效日数下限
|
|
for lib in library_names:
|
|
if lib == new:
|
|
continue
|
|
lf = wide(lib)
|
|
if lf is None:
|
|
continue
|
|
compared += 1
|
|
ic, n_valid = _cross_sectional_ic(nf, lf)
|
|
if n_valid < _MIN_CORR_DAYS:
|
|
continue
|
|
judged = True
|
|
if ic is not None and (best_ic is None or abs(ic) > abs(best_ic)):
|
|
best_name, best_ic = lib, ic
|
|
if best_ic is not None and abs(best_ic) >= threshold:
|
|
flagged.append({"factor": new, "vs": best_name,
|
|
"ic": round(best_ic, 4)})
|
|
elif not judged:
|
|
insufficient.append(new)
|
|
return {"threshold": threshold, "checked": checked, "flagged": flagged,
|
|
"skipped_missing_values": skipped, "compared_pairs": compared,
|
|
"insufficient_days": insufficient}
|
|
|
|
|
|
def write_report(out_path: str, report: dict) -> None:
|
|
os.makedirs(os.path.dirname(out_path), exist_ok=True)
|
|
with open(out_path, "w", encoding="utf-8") as f:
|
|
json.dump(report, f, ensure_ascii=False, indent=1)
|