feat(factor): P1-C 批研报共识 5+质押 4 因子 TDD(spec §20.13 三新域消费面)
CI/CD / test (push) Failing after 3s
CI/CD / nas-deploy (push) Has been skipped
CI/CD / nas-verify (push) Has been skipped

R 研报共识族(PIT=publishDate,180 天窗两端 cum 口径同 E11 手法):
- res_rating_pos 窗内买入+增持占比(emRatingName 正面集={买入,增持,推荐,强推,强买}),
  覆盖门窗内研报<3 条→NaN 稀样本失真宁缺毋假
- res_coverage 窗内研报条数(首研报前 NaN=未覆盖,曾覆盖后窗空=0 值两语义区分)
- res_fwd_ep 窗内 mean 预测 EPS/predictThisYearEps(空值不算分母;当年口径,
  跨年跳变由截面同年份对齐吸收)
- res_target_upside 最新带目标价条目+180 天新鲜度门——indvAimPriceT null 条目
  不进 target 流→不遮蔽上一份带价研报(join_asof 保留 eff 算 datetime−eff≤180)
- res_rating_chg 窗内净上调次数(ratingChange='上调'−'下调';EM 空值已统一转 None)

Q 质押族(无质押=0 真实零,fill 语义与缺域 NaN 严格区分):
- pledge_ratio 周五快照 asof backward(akshare 中文列,质押比例=总股本口径%)
- pledge_ctrl 实控人在押 max 占所持——pledge_detail 月桶区间重建:生效开始=
  max(PF_START_DATE,NOTICE_DATE) 公告日才可见宁晚毋早;结束=UNFREEZE_DATE
  (EM 无解押公告日列,假设解押日≈公告日);IS_CONTROL_SHAREHOLDER=='1' 白捡
  直标免 F10 actual_controller join;断点稀疏流=每股在押集合变化时刻聚合
  max(PF_HOLD_RATIO),断点间值恒定 join_asof 填格子
- pledge_margin_dist=−pledge_warn_line/close(在押最高预估平仓线占现价,
  低=安全垫厚;无在押 fill 0→0 值=最优档天然参与截面)
- pledge_net_180 窗内解押−新增笔数(生效开始=−1/解押=+1 两事件 cum 两端差);
  MXID 跨月桶去重取最新桶状态(文件名升序 unique keep last)
- 区间重建 1997 全史留存→全窗可回测,非前向积累(与 S 族对照)

工程: conftest 扩三新域合成树(周窗文件+周五快照+月桶断点/解押/PIT notice
晚 start 用例);计数列聚合即 cast Int64(u32 环回坑 P2 实锤先例延续);
FEATURE_COLUMNS+9 列/_DOMAIN_COLS 扩;既有计数断言 80→89;新增 16 用例,
全套 1458 绿;设计档 §3 表同步 P1-C 行 [nas]

Co-Authored-By: Claude Code <noreply@anthropic.com>
This commit is contained in:
2026-09-22 14:08:19 +08:00
parent 9741a69096
commit d92100fd1e
11 changed files with 711 additions and 11 deletions
@@ -53,6 +53,7 @@
| 第一批 | vnpy Alpha101(**实挂 82**,18 个 add_feature 是注释行,grep 假数)+ Alpha158 | 240 注册/227 可算 | ✅ |
| 财务批 | 聚宽财务 136 公式本地化(P0 32 个 TDD → P1 扩展) | 96 候选→32+37+6 | ✅ |
| P2 批(09-22) | 七项目指标调研增量:D16 PEG/D17 PS + E14-16 股息族(dividend 域除息日锚:div_12m 365 天滚动/派息率/连续分红年数断年重计);PEG 分母走守卫列 peg_denom(np_ttm≤0 或 g≤1% → NaN 宁缺毋假);S 族情绪 4 因子(sentiment_lexicon 词表短语优先+计数多数决,news_meta 打分 PIT=次日可见,覆盖门窗内<5 条 NaN;**前向积累型**——corpus 2026-09-06 起采集,历史窗全 NaN 诚实透出,IC 随语料攒厚度,首个有意义窗=10 月下旬 Q3 披露潮) | 9 | ✅ TDD |
| P1-C 批(09-22) | spec §20.13 三新域消费面:**R 研报共识 5**(res_rating_pos 买入+增持占比门窗≥3 / res_coverage 180 天窗研报数 / res_fwd_ep 窗内 mean 预测 EPS 空值不算分母 / res_target_upside 最新带价条目+180 天新鲜度门 null 目标价不进流不遮蔽 / res_rating_chg 窗内净上调;PIT=publishDate,180 天窗两端 cum 口径同 E11 手法)+ **Q 质押 4**(pledge_ratio 周五快照 asof / pledge_ctrl 实控人在押 max 占所持——detail 区间重建,生效=max(质押开始,公告日),IS_CONTROL_SHAREHOLDER=='1' 直标免 F10 join / pledge_margin_dist=−pledge_warn_line/close 在押最高预估平仓线 / pledge_net_180 窗内解押−新增笔数两事件 cum;**无质押=0 真实零** fill 语义与缺域 NaN 严格区分;1997 全史区间重建全窗可回测非前向积累) | 9 | ✅ TDD |
| GTJA191 | — | — | ❄️ 冷冻(用户 09-12 拍板) |
| Barra CNE5 | 风格因子 | — | 长线 |
+253 -1
View File
@@ -392,9 +392,250 @@ _EMPTY_CASH = {"cum": _EMPTY_EVENTS.with_columns(
pl.lit(None, dtype=pl.Float64).alias("div_streak"))}
# ==================== P1-C 三新域(spec §20.13,2026-09-22) ====================
RES_WINDOW_DAYS = 180 # 研报共识窗(半年: 年报季覆盖密度×评级新鲜度)
PLEDGE_NET_WINDOW_DAYS = 180 # 质押净变化窗(与研报窗对齐)
RES_MIN_REPORTS = 3 # rating_pos 覆盖门(窗内研报数下限,稀样本失真)
# emRatingName 正面评级集(买入/增持主流;推荐/强推/强买为券商变体兜底)
_RES_POS_RATINGS = {"买入", "增持", "推荐", "强推", "强买"}
def _empty_res() -> dict:
return {
"agg": pl.DataFrame(schema={
"vt_symbol": pl.Utf8, "eff": pl.Datetime("us"),
"_n_cum": pl.Int64, "_pos_cum": pl.Int64, "_up_cum": pl.Int64,
"_dn_cum": pl.Int64, "_eps_sum": pl.Float64, "_eps_n_cum": pl.Int64}),
"target": pl.DataFrame(schema={"vt_symbol": pl.Utf8,
"eff": pl.Datetime("us"),
"_tgt": pl.Float64}),
}
def _load_research_report_events(codes: list[str], data_dir: str) -> dict:
"""research_report 周窗文件 → 研报共识两事件流(PIT=条目 publishDate).
agg = (vt, eff, cum 列×6): _n/_pos/_up/_dn/_eps_n 计数与 _eps_sum 按股
事件日序 cum(计数即 cast Int64——Boolean.sum() 产 UInt32 与 cum/rolling
组合有 u32 环回坑,P2 实锤先例);pit 层 180 天两端差分出窗值(同 E11
手法)。周窗文件只做枚举容器,窗语义不参与(loader 全量 concat)。
target = (vt, eff, _tgt): 仅带 indvAimPriceT 的条目(null 目标价不进
流 → 不遮蔽上一份带价研报);新鲜度门在 pit 层(datetime−eff≤180)。
ratingChange 值域=「上调/下调/维持/首次」文字(EM 空值已统一转 None)。
"""
empty = _empty_res()
rr_dir = os.path.join(data_dir, "research_report")
if not os.path.isdir(rr_dir):
_warn_domain_missing("research_report", rr_dir)
return empty
code_set = set(codes)
frames = []
for fname in sorted(os.listdir(rr_dir)):
if not fname.endswith(".parquet"):
continue
try:
f = pl.read_parquet(os.path.join(rr_dir, fname))
except Exception:
continue
if not all(c in f.columns for c in ("stockCode", "publishDate")):
continue
eff = (pl.col("publishDate").cast(pl.Utf8).str.slice(0, 10)
.str.to_date("%Y-%m-%d", strict=False))
f = f.select(
_code6_to_vt(pl.col("stockCode").cast(pl.Utf8).str.strip_chars()
.str.zfill(6)).alias("vt_symbol"),
eff.alias("eff"),
pl.col("emRatingName").cast(pl.Utf8, strict=False).alias("_rating"),
pl.col("ratingChange").cast(pl.Utf8, strict=False).alias("_chg"),
pl.col("predictThisYearEps").cast(pl.Float64, strict=False).alias("_eps"),
pl.col("indvAimPriceT").cast(pl.Float64, strict=False).alias("_tgt"),
).filter(pl.col("vt_symbol").is_in(code_set) & pl.col("eff").is_not_null())
if f.height:
frames.append(f)
if not frames:
return empty
ev = pl.concat(frames).sort(["vt_symbol", "eff"])
agg = (ev.with_columns(
pl.when(pl.col("_rating").is_in(_RES_POS_RATINGS))
.then(1).otherwise(0).alias("_p"),
pl.when(pl.col("_chg") == "上调").then(1).otherwise(0).alias("_u"),
pl.when(pl.col("_chg") == "下调").then(1).otherwise(0).alias("_d"))
.group_by(["vt_symbol", "eff"]).agg(
pl.len().cast(pl.Int64).alias("_n"),
pl.col("_p").sum().cast(pl.Int64).alias("_pos"),
pl.col("_u").sum().cast(pl.Int64).alias("_up"),
pl.col("_d").sum().cast(pl.Int64).alias("_dn"),
# 组内全 null EPS → 0 和(有无 EPS 条目由 _eps_n 区分)
pl.col("_eps").fill_null(0.0).sum().alias("_eps_sum"),
pl.col("_eps").is_not_null().sum().cast(pl.Int64).alias("_eps_n"),
).sort(["vt_symbol", "eff"]))
agg = agg.with_columns(
*[pl.col(c).cum_sum().over(_SYM).alias(f"{c}_cum")
for c in ("_n", "_pos", "_up", "_dn", "_eps_n")],
pl.col("_eps_sum").cum_sum().over(_SYM).alias("_eps_sum"),
).select("vt_symbol", pl.col("eff").cast(pl.Datetime("us")),
"_n_cum", "_pos_cum", "_up_cum", "_dn_cum",
"_eps_sum", "_eps_n_cum")
target = (ev.filter(pl.col("_tgt").is_not_null())
.select("vt_symbol", pl.col("eff").cast(pl.Datetime("us")), "_tgt")
.sort("eff"))
return {"agg": agg, "target": target}
def _load_pledge_ratio_snapshots(codes: list[str], data_dir: str) -> pl.DataFrame:
"""pledge_ratio 周五快照 → (vt, eff=交易日期, pledge_ratio) 周频流.
列=akshare 中文列(质押比例=总股本口径 %);快照只含有质押的股票
→ 股无行=无质押=0(pit 层 fill 0,与缺域 NaN 语义区分)。
"""
schema = {"vt_symbol": pl.Utf8, "eff": pl.Datetime("us"),
"pledge_ratio": pl.Float64}
pr_dir = os.path.join(data_dir, "pledge_ratio")
if not os.path.isdir(pr_dir):
_warn_domain_missing("pledge_ratio", pr_dir)
return pl.DataFrame(schema=schema)
code_set = set(codes)
frames = []
for fname in sorted(os.listdir(pr_dir)):
if not fname.endswith(".parquet"):
continue
try:
f = pl.read_parquet(os.path.join(pr_dir, fname))
except Exception:
continue
if not all(c in f.columns for c in ("股票代码", "交易日期", "质押比例")):
continue
eff = (pl.col("交易日期").cast(pl.Utf8).str.slice(0, 10)
.str.to_date("%Y-%m-%d", strict=False))
f = f.select(
_code6_to_vt(pl.col("股票代码").cast(pl.Utf8).str.strip_chars()
.str.zfill(6)).alias("vt_symbol"),
eff.alias("eff"),
pl.col("质押比例").cast(pl.Float64, strict=False).alias("pledge_ratio"),
).filter(pl.col("vt_symbol").is_in(code_set) & pl.col("eff").is_not_null()
& pl.col("pledge_ratio").is_not_null())
if f.height:
frames.append(f)
if not frames:
return pl.DataFrame(schema=schema)
return (pl.concat(frames)
.unique(subset=["vt_symbol", "eff"], keep="last")
.sort("eff")
.with_columns(pl.col("eff").cast(pl.Datetime("us"))))
def _empty_pledge() -> dict:
return {
"ctrl": pl.DataFrame(schema={"vt_symbol": pl.Utf8, "eff": pl.Datetime("us"),
"pledge_ctrl": pl.Float64}),
"wl": pl.DataFrame(schema={"vt_symbol": pl.Utf8, "eff": pl.Datetime("us"),
"pledge_warn_line": pl.Float64}),
"net": pl.DataFrame(schema={"vt_symbol": pl.Utf8, "eff": pl.Datetime("us"),
"_net_cum": pl.Int64}),
}
def _load_pledge_detail_features(codes: list[str], data_dir: str) -> dict:
"""pledge_detail 月桶 EM raw → 区间重建三流(spec §20.13.2).
PIT 红线: 生效开始 = max(PF_START_DATE, NOTICE_DATE)(质押公告日才
可见,宁晚毋早);结束 = UNFREEZE_DATE(解押即视为可见——EM 无解押
公告日列,假设解押日≈公告日);NOTICE_DATE/PF_START_DATE 缺失行丢弃
(宁缺毋假)。月桶=全量快照切片,同笔质押跨桶按 MXID 取最新桶状态
(文件名升序=时间升序 → unique keep last)。
- ctrl/wl = 断点稀疏流: 每股在全部 生效开始/解押 时刻断点处聚合
「在押集合」(开始≤t<结束)的 max(实控人 PF_HOLD_RATIO)/
max(WARNING_LINE)——最激素质押/最迫近平仓线口径;断点间值恒定,
pit 层 join_asof backward 填格子。IS_CONTROL_SHAREHOLDER=='1'
=控股股东直标(白捡,免 F10 actual_controller join)。
- net = 两事件流(生效开始=−1 新增质押/解押=+1 缓解)cum 累计,
pit 层 180 天两端差(同 E11 手法);计数列 cast Int64。
"""
empty = _empty_pledge()
pdet_dir = os.path.join(data_dir, "pledge_detail")
if not os.path.isdir(pdet_dir):
_warn_domain_missing("pledge_detail", pdet_dir)
return empty
frames = []
for fname in sorted(os.listdir(pdet_dir)): # 升序=旧桶在前(keep last=最新)
if not fname.endswith(".parquet"):
continue
try:
f = pl.read_parquet(os.path.join(pdet_dir, fname))
except Exception:
continue
frames.append(f)
if not frames:
return empty
raw = pl.concat(frames, how="diagonal")
if "MXID" in raw.columns:
raw = raw.unique(subset=["MXID"], keep="last")
need = ("SECURITY_CODE", "PF_START_DATE", "NOTICE_DATE", "PF_HOLD_RATIO",
"WARNING_LINE", "IS_CONTROL_SHAREHOLDER")
if not all(c in raw.columns for c in need):
return empty
if "UNFREEZE_DATE" not in raw.columns:
raw = raw.with_columns(pl.lit(None, dtype=pl.Utf8).alias("UNFREEZE_DATE"))
code_set = set(codes)
def _d(col: str) -> pl.Expr:
return (pl.col(col).cast(pl.Utf8).str.slice(0, 10)
.str.to_date("%Y-%m-%d", strict=False))
rows = raw.select(
_code6_to_vt(pl.col("SECURITY_CODE").cast(pl.Utf8).str.strip_chars()
.str.zfill(6)).alias("vt_symbol"),
_d("PF_START_DATE").alias("_st"),
_d("NOTICE_DATE").alias("_nt"),
_d("UNFREEZE_DATE").alias("_e"),
pl.col("PF_HOLD_RATIO").cast(pl.Float64, strict=False).alias("_hr"),
pl.col("WARNING_LINE").cast(pl.Float64, strict=False).alias("_wl"),
(pl.col("IS_CONTROL_SHAREHOLDER").cast(pl.Utf8, strict=False)
== "1").alias("_ctrl"),
).filter(pl.col("vt_symbol").is_in(code_set)
& pl.col("_st").is_not_null() & pl.col("_nt").is_not_null())
if not rows.height:
return empty
# 生效开始 = max(质押开始, 公告日)(公告晚于开始 → 公告日才可见)
rows = rows.with_columns(
pl.max_horizontal("_st", "_nt").alias("_s")).sort(["vt_symbol", "_s"])
# ---- ctrl/wl 断点稀疏流(在押集合区间重建) ----
bps = pl.concat([
rows.select("vt_symbol", pl.col("_s").alias("_t")),
rows.filter(pl.col("_e").is_not_null())
.select("vt_symbol", pl.col("_e").alias("_t")),
]).unique(subset=["vt_symbol", "_t"])
active = bps.join(rows, on="vt_symbol").filter(
(pl.col("_s") <= pl.col("_t"))
& (pl.col("_e").is_null() | (pl.col("_e") > pl.col("_t"))))
ctrl = (active.filter(pl.col("_ctrl"))
.group_by(["vt_symbol", "_t"]).agg(pl.col("_hr").max())
.rename({"_hr": "pledge_ctrl", "_t": "eff"})
.select("vt_symbol", pl.col("eff").cast(pl.Datetime("us")),
"pledge_ctrl").sort("eff"))
wl = (active.group_by(["vt_symbol", "_t"]).agg(pl.col("_wl").max())
.rename({"_wl": "pledge_warn_line", "_t": "eff"})
.select("vt_symbol", pl.col("eff").cast(pl.Datetime("us")),
"pledge_warn_line").sort("eff"))
# ---- net 两事件流(新增=−1/解押=+1)cum ----
net_ev = pl.concat([
rows.select("vt_symbol", pl.col("_s").alias("eff"),
pl.lit(-1, dtype=pl.Int64).alias("_d")),
rows.filter(pl.col("_e").is_not_null())
.select("vt_symbol", pl.col("_e").alias("eff"),
pl.lit(1, dtype=pl.Int64).alias("_d")),
]).group_by(["vt_symbol", "eff"]).agg(pl.col("_d").sum()).sort(
["vt_symbol", "eff"])
net = net_ev.with_columns(
pl.col("_d").cum_sum().over(_SYM).alias("_net_cum")
).select("vt_symbol", pl.col("eff").cast(pl.Datetime("us")), "_net_cum")
return {"ctrl": ctrl, "wl": wl, "net": net}
def _load_extra_domains(codes: list[str], data_dir: str,
start: str, end: str, out_cols: list[str]) -> dict:
"""P1-B 四新域 + P2 股息现金族按引用列加载(引用瘦身: 未引用的域不读盘)."""
"""P1-B 四新域 + P2 股息现金族 + P1-C 三新域按引用列加载(引用瘦身)."""
cols = set(out_cols)
return {
"dividend": (_load_dividend_cum(codes, data_dir)
@@ -408,4 +649,15 @@ def _load_extra_domains(codes: list[str], data_dir: str,
if "topholder_chg" in cols else _EMPTY_EVENTS),
"vb": (_load_vb_daily(codes, data_dir, start, end)
if "ep_vb" in cols or "bp_vb" in cols else _EMPTY_EVENTS),
"research": (_load_research_report_events(codes, data_dir)
if {"res_coverage", "res_rating_pos", "res_fwd_eps",
"res_target_price", "res_rating_chg"} & cols
else _empty_res()),
"pledge_ratio": (_load_pledge_ratio_snapshots(codes, data_dir)
if "pledge_ratio" in cols
else _EMPTY_EVENTS.with_columns(
pl.lit(None, dtype=pl.Float64).alias("pledge_ratio"))),
"pledge_detail": (_load_pledge_detail_features(codes, data_dir)
if {"pledge_ctrl", "pledge_warn_line", "pledge_net_180"} & cols
else _empty_pledge()),
}
+27 -2
View File
@@ -1,4 +1,4 @@
"""财务因子批表达式库: P0 32 + P1 37 + P1-B 6 注册(category="fundamental").
"""财务因子批表达式库: P0 32 + P1 37 + P1-B 6 + P2 5 + P1-C 9 注册(category="fundamental").
选型来源 docs/fundamental_factor_survey_20260907.md(§3 候选池 96 个 + §4.1 去重),
首批 32 = P0 50 个按族去重后的主代表;全部「因子 = cs_rank(基础指标)」一层截面,
@@ -159,10 +159,35 @@ FUNDAMENTAL_P2_FACTORS: list[tuple[str, str, str, str]] = [
]
# P1-C 批(spec §20.13 三新域联合设计 2026-09-22: 研报共识 5 + 质押 4;
# 编号=R/Q 新族,设计档 §3 定义)。口径: 研报族 180 天窗两端 cum 口径
# (PIT=publishDate;rating_pos 门 n≥3;fwd_ep 门带 EPS 条目 ≥1 空值不算
# 分母;target=最新带价条目+180 天新鲜度门,null 目标价不进流不遮蔽)。
# 质押族: ratio=周五快照 asof(无质押 fill 0 真实零);ctrl/wl=detail 区间
# 重建断点流(生效=max(开始,公告日),IS_CONTROL_SHAREHOLDER=='1' 实控人
# 直标);net=180 天窗解押−新增笔数;Q03 安全垫走 −pledge_warn_line/close
# (平仓线占现价低=厚;无在押 fill 0 → 0 值=最优档天然参与截面)。
FUNDAMENTAL_P1C_FACTORS: list[tuple[str, str, str, str]] = [
# ---- R 研报共识(5)----
("res_rating_pos", "cs_rank(res_rating_pos)", "R01", "+"),
# 覆盖数方向不定(机构认可 vs 高关注拥挤)→ 正向注册待 IC 实证
("res_coverage", "cs_rank(res_coverage)", "R02", "不定"),
("res_fwd_ep", "cs_rank(res_fwd_eps / close)", "R03", "+"),
("res_target_upside", "cs_rank(res_target_price / close - 1)", "R04", "+"),
("res_rating_chg", "cs_rank(res_rating_chg)", "R05", "+"),
# ---- Q 质押风险(4)----
("pledge_ratio", "cs_rank(-pledge_ratio)", "Q01", "-"),
("pledge_ctrl", "cs_rank(-pledge_ctrl)", "Q02", "-"),
("pledge_margin_dist", "cs_rank(-pledge_warn_line / close)", "Q03", "+"),
("pledge_net_chg", "cs_rank(pledge_net_180)", "Q04", "+"),
]
def _register_all() -> None:
"""注册全部财务因子(已存在同名跳过,幂等;同 library.py 模式)."""
batches = [*FUNDAMENTAL_FACTORS, *FUNDAMENTAL_P1_FACTORS,
*FUNDAMENTAL_P1B_FACTORS, *FUNDAMENTAL_P2_FACTORS]
*FUNDAMENTAL_P1B_FACTORS, *FUNDAMENTAL_P2_FACTORS,
*FUNDAMENTAL_P1C_FACTORS]
for name, expression, _doc_id, _ic in batches:
if name not in _REGISTRY:
register_factor(name, expression, category="fundamental")
+95 -1
View File
@@ -12,7 +12,12 @@ from datetime import datetime
import polars as pl
from .fundamental_domains import _load_extra_domains
from .fundamental_domains import (
PLEDGE_NET_WINDOW_DAYS,
RES_MIN_REPORTS,
RES_WINDOW_DAYS,
_load_extra_domains,
)
from .fundamental_forecast import _load_forecast_events
from .fundamental_report_features import _compute_report_features, _safe_ratio
from .fundamental_schema import (
@@ -205,6 +210,95 @@ def iter_fundamental_feature_chunks(
out = out.sort("datetime").join_asof(
cash["streak"], left_on="datetime", right_on="eff",
by="vt_symbol", strategy="backward").drop("eff")
# ---- P1-C 研报共识(180 天窗两端 cum 口径,同 E11 手法)----
rr = extra["research"]
_rr_cols = {"res_coverage", "res_rating_pos", "res_fwd_eps",
"res_rating_chg"}
if _rr_cols & set(out_cols) and rr["agg"].height:
_cums = ["_n_cum", "_pos_cum", "_up_cum", "_dn_cum",
"_eps_sum", "_eps_n_cum"]
rr180 = rr["agg"].rename({c: f"{c}_180" for c in _cums})
out = out.sort("datetime").join_asof(
rr["agg"], left_on="datetime", right_on="eff",
by="vt_symbol", strategy="backward").drop("eff")
out = (
out.with_columns(
(pl.col("datetime")
- pl.duration(days=RES_WINDOW_DAYS)).alias("_dtw"))
.sort("_dtw")
.join_asof(rr180, left_on="_dtw", right_on="eff",
by="vt_symbol", strategy="backward")
.drop("eff", "_dtw"))
n_w = pl.col("_n_cum") - pl.col("_n_cum_180").fill_null(0)
pos_w = pl.col("_pos_cum") - pl.col("_pos_cum_180").fill_null(0)
eps_n_w = (pl.col("_eps_n_cum")
- pl.col("_eps_n_cum_180").fill_null(0))
eps_sum_w = (pl.col("_eps_sum")
- pl.col("_eps_sum_180").fill_null(0.0))
out = out.with_columns(
# 首研报前=cum null → NaN(未覆盖,与窗空 0 值区分)
pl.when(pl.col("_n_cum").is_null()).then(None)
.otherwise(n_w.cast(pl.Float64)).alias("res_coverage"),
pl.when(pl.col("_n_cum").is_null()
| (n_w < RES_MIN_REPORTS)).then(None)
.otherwise(pos_w / n_w).alias("res_rating_pos"),
pl.when(pl.col("_n_cum").is_null() | (eps_n_w < 1))
.then(None).otherwise(eps_sum_w / eps_n_w)
.alias("res_fwd_eps"),
pl.when(pl.col("_n_cum").is_null()).then(None)
.otherwise((pl.col("_up_cum") - pl.col("_up_cum_180").fill_null(0))
- (pl.col("_dn_cum")
- pl.col("_dn_cum_180").fill_null(0)))
.cast(pl.Float64).alias("res_rating_chg"),
)
if "res_target_price" in out_cols and rr["target"].height:
# join_asof 保留右键 eff 算新鲜度(datetime−eff≤180 天;
# null 目标价条目不进流,上一份带价研报不被遮蔽)
out = out.sort("datetime").join_asof(
rr["target"], left_on="datetime", right_on="eff",
by="vt_symbol", strategy="backward")
out = out.with_columns(
pl.when(pl.col("_tgt").is_not_null() & pl.col("eff").is_not_null()
& ((pl.col("datetime") - pl.col("eff"))
.dt.total_days() <= RES_WINDOW_DAYS))
.then(pl.col("_tgt")).otherwise(None)
.alias("res_target_price")).drop("eff", "_tgt")
# ---- P1-C 质押族(ratio 周五快照 asof/detail 区间重建断点流;
# 无质押=0 真实零,fill 语义与缺域 NaN 严格区分)----
if extra["pledge_ratio"].height:
out = out.sort("datetime").join_asof(
extra["pledge_ratio"], left_on="datetime", right_on="eff",
by="vt_symbol", strategy="backward").drop("eff")
out = out.with_columns(pl.col("pledge_ratio").fill_null(0.0))
pdet = extra["pledge_detail"]
if pdet["ctrl"].height:
out = out.sort("datetime").join_asof(
pdet["ctrl"], left_on="datetime", right_on="eff",
by="vt_symbol", strategy="backward").drop("eff")
out = out.with_columns(pl.col("pledge_ctrl").fill_null(0.0))
if pdet["wl"].height:
out = out.sort("datetime").join_asof(
pdet["wl"], left_on="datetime", right_on="eff",
by="vt_symbol", strategy="backward").drop("eff")
out = out.with_columns(pl.col("pledge_warn_line").fill_null(0.0))
if pdet["net"].height:
net = pdet["net"]
net180 = net.rename({"_net_cum": "_net_cum_180"})
out = out.sort("datetime").join_asof(
net, left_on="datetime", right_on="eff",
by="vt_symbol", strategy="backward").drop("eff")
out = (
out.with_columns(
(pl.col("datetime") - pl.duration(days=PLEDGE_NET_WINDOW_DAYS)).alias("_dtw"))
.sort("_dtw")
.join_asof(net180, left_on="_dtw", right_on="eff",
by="vt_symbol", strategy="backward")
.drop("eff", "_dtw")
.with_columns(
# 无质押史/窗内无事件=0(无变化真实零)
(pl.col("_net_cum").fill_null(0)
- pl.col("_net_cum_180").fill_null(0))
.cast(pl.Float64).alias("pledge_net_180")))
for key in ("gdhs", "topholder"):
if extra[key].height:
out = out.sort("datetime").join_asof(
+17 -2
View File
@@ -112,6 +112,17 @@ FEATURE_COLUMNS: list[str] = [
"div_12m", # E14 股息率分子: 近365天现金分红/股(除息日锚)
"div_payout", # E15 派息率(np_ttm>0 守卫,pit 层后算)
"div_streak", # E16 连续现金分红年数(断年重计)
# ---- P1-C 批新增(spec §20.13 三新域: 研报共识 5 + 质押 4;契约由
# test_fundamental_p1c_* 锁定)----
"res_coverage", # R01 原料: 180 天窗研报条数(publishDate 锚两端 cum)
"res_rating_pos", # R02 原料: 窗内买入+增持占比(门 n≥3)
"res_fwd_eps", # R03 原料: 窗内 mean 预测 EPS(空值不算分母)
"res_target_price", # R04 原料: 最新可见目标价(180 天新鲜度门)
"res_rating_chg", # R05 原料: 窗内净上调次数(上调−下调)
"pledge_ratio", # Q01 原料: 总股本质押比例%(周五快照 asof,无=0)
"pledge_ctrl", # Q02 原料: 实控人在押 max 占所持%(区间重建)
"pledge_warn_line", # Q03 原料: 在押最高预估平仓线价(无在押=0)
"pledge_net_180", # Q04 原料: 180 天窗解押−新增笔数(无=0)
]
# 金融股置 NaN 的特征(盈利质量 B 族 + 成长 C 族 + 费用类,§7 红线 5;
# survey A 族注意点: A14/A16 金融无三费结构 → rd/sale_expense_ratio 同剔)
@@ -133,6 +144,10 @@ _FORECAST_COLS = ("forecast_type_score", "forecast_change_pct",
# 预告兑现差独立事件流(F07 前视红线: 锚 = max(实际披露日, 预告公告日))
_BEAT_COLS = ("forecast_beat",)
# P1-B 四新域列(非报告期特征,由各域事件/日频流产出;报表加载须排除)
# P2 股息三列(dividend 域除息日锚事件流,pit 层产出)同属域列
# P2 股息三列(dividend 域除息日锚事件流,pit 层产出)同属域列;
# P1-C 研报共识 5 + 质押 4(三新域事件/快照流)同属域列
_DOMAIN_COLS = ("send_total_12m", "gdhs_chg", "topholder_chg", "ep_vb", "bp_vb",
"div_12m", "div_payout", "div_streak")
"div_12m", "div_payout", "div_streak",
"res_coverage", "res_rating_pos", "res_fwd_eps",
"res_target_price", "res_rating_chg",
"pledge_ratio", "pledge_ctrl", "pledge_warn_line", "pledge_net_180")
+66
View File
@@ -343,6 +343,72 @@ def build_synthetic_static(root: str) -> str:
{"数据日期": date(2026, 1, 5), "PE(TTM)": 25.0, "市净率": 5.0},
{"数据日期": date(2026, 8, 12), "PE(TTM)": 24.0, "市净率": 4.8},
]).write_parquet(os.path.join(static_dir, "valuation", "600000.SH_valuation.parquet"))
# ==================== P1-C 批三新域(spec §20.13,2026-09-22) ====================
# 契约=data 侧指针包: research_report 周窗文件 EM 原生列直存(空值已统一
# 转 None); pledge_ratio 周五快照 akshare 中文列; pledge_detail 月桶
# EM raw 列(SECURITY_CODE/IS_CONTROL_SHAREHOLDER/PF_*/WARNING_LINE)。
# ---- research_report 周窗文件(A 五份 2023-03..2024-02;C 单份中性)----
# A 事件: e1 03-10 买入 eps1.0 / e2 05-12 增持 tgt20 eps1.1 维持 /
# e3 07-03 中性 tgt18 下调(eps=None→不算分母) / e4 11-17 买入 tgt22
# eps1.2 上调 / e5 24-02-09 增持 tgt=None 维持(null 目标价不进 target
# 流 → 不遮蔽 e4;e4 距 24-06-30=226 天>180 → 新鲜度门用例)
def _rr(code, publish, rating, chg, eps, tgt):
return {"stockCode": code, "orgSName": f"券商{code}",
"emRatingName": rating, "sRatingName": None, "ratingChange": chg,
"publishDate": f"{publish} 00:00:00", "predictThisYearEps": eps,
"indvAimPriceL": None, "indvAimPriceT": tgt,
"title": f"研报{publish}", "infoCode": f"{code}-{publish}"}
rr_dir = os.path.join(static_dir, "research_report")
os.makedirs(rr_dir, exist_ok=True)
for fname, rows in {
"20230310_research_report.parquet": [_rr("600000", "2023-03-10", "买入", None, 1.0, None)],
"20230512_research_report.parquet": [_rr("600000", "2023-05-12", "增持", "维持", 1.1, 20.0)],
"20230707_research_report.parquet": [_rr("600000", "2023-07-03", "中性", "下调", None, 18.0)],
"20231117_research_report.parquet": [_rr("600000", "2023-11-17", "买入", "上调", 1.2, 22.0)],
"20240209_research_report.parquet": [_rr("600000", "2024-02-09", "增持", "维持", None, None)],
"20220701_research_report.parquet": [_rr("300001", "2022-06-30", "中性", None, None, None)],
}.items():
pl.DataFrame(rows).write_parquet(os.path.join(rr_dir, fname))
# ---- pledge_ratio 周五快照(A 两期 5.0→6.0;C 一期 20.0;B 无行=fill 0 用例)----
pr_dir = os.path.join(static_dir, "pledge_ratio")
os.makedirs(pr_dir, exist_ok=True)
def _pr(code, name, day, ratio):
return {"股票代码": code, "股票简称": name, "交易日期": day,
"质押比例": ratio, "质押股数": 1000.0, "质押市值": 1e5,
"质押笔数": 3, "无限售股质押数": 800.0, "限售股质押数": 200.0}
pl.DataFrame([_pr("600000", "浦发", date(2023, 1, 6), 5.0),
_pr("300001", "特锐", date(2023, 1, 6), 20.0)]
).write_parquet(os.path.join(pr_dir, "20230106_pledge_ratio.parquet"))
pl.DataFrame([_pr("600000", "浦发", date(2023, 7, 7), 6.0)]
).write_parquet(os.path.join(pr_dir, "20230707_pledge_ratio.parquet"))
# ---- pledge_detail 月桶 EM raw(A 三笔: 实控人两笔[一笔 09-10 解押]+
# 基金一笔在押;C 一笔 2021 全程在押;notice 晚 start 5 天=PIT 用例)----
def _pd(code, holder, ctrl, start, notice, end, hr, wl):
return {"SECURITY_CODE": code, "HOLDER_NAME": holder,
"IS_CONTROL_SHAREHOLDER": ctrl, "PF_NUM": 1000.0,
"PF_HOLD_RATIO": hr, "PF_TSR": hr / 10.0, "PF_ORG": "某银行",
"CLOSE_PRICE": 10.0, "WARNING_LINE": wl,
"PF_START_DATE": f"{start} 00:00:00",
"NOTICE_DATE": f"{notice} 00:00:00",
"UNFREEZE_DATE": (f"{end} 00:00:00" if end else None),
"UNFREEZE_STATE": ("已解押" if end else "未解押"),
"MXID": f"{code}-{holder}-{start}"}
pdet_dir = os.path.join(static_dir, "pledge_detail")
os.makedirs(pdet_dir, exist_ok=True)
pl.DataFrame([
_pd("600000", "控股集团", "1", "2023-02-10", "2023-02-15", None, 30.0, 15.0),
_pd("600000", "控股集团", "1", "2023-05-20", "2023-05-20", "2023-09-10", 50.0, 25.0),
_pd("600000", "某基金", "0", "2023-08-01", "2023-08-01", None, 10.0, 8.0),
_pd("300001", "普通股东", "0", "2021-06-01", "2021-06-01", None, 40.0, 12.0),
]).write_parquet(os.path.join(pdet_dir, "202307_pledge_detail.parquet"))
return static_dir
+3 -3
View File
@@ -33,8 +33,8 @@ def test_p0_32_factors_registered():
names = {f["name"] for f in facs}
# P0 32 + P1 37(35 因子 + 2 互评变体;P1 名单见 test_fundamental_p1_library)
# + v2b 中性化 6(fund_*_neu,注册链经 composite_weighting 可在场,故派生不钉死)
assert len(facs) == 80 + len(FUND_NEU_SOURCES), \
f"P0+P1+P1-B+P2+neu 应为 {80 + len(FUND_NEU_SOURCES)} 个,实际 {len(facs)}"
assert len(facs) == 89 + len(FUND_NEU_SOURCES), \
f"P0+P1+P1-B+P2+neu 应为 {89 + len(FUND_NEU_SOURCES)} 个,实际 {len(facs)}"
# 六族代表抽查(全部名单见 fundamental_library 注释)
expect = {
"fund_roe_ttm", "fund_gp_over_assets", # A
@@ -86,4 +86,4 @@ def test_registration_idempotent():
"""重复 import 不炸(注册表防重入,同 library.py 模式)."""
import importlib
importlib.reload(fundamental_library)
assert len(_fundamental_factors()) == 80 + len(FUND_NEU_SOURCES)
assert len(_fundamental_factors()) == 89 + len(FUND_NEU_SOURCES)
+1 -1
View File
@@ -102,4 +102,4 @@ def test_p1_registration_idempotent():
fundamental_library._register_all()
fundamental_library._register_all()
# 80 = P0 32 + P1 37 + P1-B 6 + P2 5;v2b 另挂 6 个 fund_*_neu(fundamental 类)
assert len(list_factors("fundamental")) == 80 + len(FUND_NEU_SOURCES)
assert len(list_factors("fundamental")) == 89 + len(FUND_NEU_SOURCES)
+1 -1
View File
@@ -70,4 +70,4 @@ def test_p1b_registration_idempotent():
fundamental_library._register_all()
fundamental_library._register_all()
# P0 32 + P1 37 + P1-B 6 + P2 5 = 80;v2b 另挂 6 个 fund_*_neu(fundamental 类)
assert len(list_factors("fundamental")) == 80 + len(FUND_NEU_SOURCES)
assert len(list_factors("fundamental")) == 89 + len(FUND_NEU_SOURCES)
@@ -0,0 +1,177 @@
# tests/factor/test_fundamental_p1c_adapter.py
"""P1-C 批财务因子适配层: 研报共识 5 + 质押 4(spec §20.13 联合设计,2026-09-22).
口径锚(P1-C 开工包):
- 研报族 PIT=条目 publishDate;180 天窗两端 cum 口径(同 E11 手法);
rating_pos 门=窗内研报 ≥3;fwd_ep 门=窗内带 EPS 条目 ≥1(空值不算
分母);target=最新带目标价条目+180 天新鲜度门(null 目标价不进流
→ 不遮蔽上一份带价研报)
- pledge_ratio PIT=周五快照(asof 取 ≤asof 最近期);股无行=0(无质押
=真实零,与缺域 NaN 语义区分)
- pledge_detail 区间重建: 生效开始=max(PF_START_DATE, NOTICE_DATE)
(公告日才可见,宁晚毋早);结束=UNFREEZE_DATE(解押即视为可见——EM 无
解押公告日列);asof∈[开始,结束) ⇒ 在押;IS_CONTROL_SHAREHOLDER=='1'
行=实控人质押直标(免股东名 join)
- pledge_net_180 = 180 天窗(解押笔数 − 新增质押笔数);计数列聚合即
cast Int64(u32 环回坑,P2 实锤先例)
"""
import sys, os
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..")))
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..", "vnpy_v4.4.0")))
import pytest
from sanguo_factor.fundamental_adapter import build_fundamental_features
A, B, C, BANK = "600000.SSE", "000001.SZSE", "300001.SZSE", "601398.SSE"
_RES_COLS = ["res_coverage", "res_rating_pos", "res_fwd_eps",
"res_target_price", "res_rating_chg"]
_PL_COLS = ["pledge_ratio", "pledge_ctrl", "pledge_warn_line", "pledge_net_180"]
def _dt(day: str):
import datetime as _d
return _d.datetime.strptime(day, "%Y-%m-%d")
def val(df, vt: str, day: str, col: str):
row = df.filter((df["vt_symbol"] == vt) & (df["datetime"] == _dt(day)))
assert row.height == 1, f"grid 缺行 {vt} {day}"
v = row[col][0]
return None if v is None else float(v)
@pytest.fixture(scope="module")
def feat(synthetic_static):
return build_fundamental_features(
[A, B, C, BANK], "2021-01-01", "2024-06-30", data_dir=synthetic_static,
columns=_RES_COLS + _PL_COLS)
# ---------- R 族: res_coverage(180 天窗研报数) ----------
def test_res_coverage_window(feat):
# A 事件(窗右开两端差): 07-03 时窗含 e1..e3 → 3;08-29(−180d=03-02,
# e1@03-10 未出窗) → 3;24-06-30(−180d=2024-01-02,e1..e4 出窗)仅 e5 → 1
assert val(feat, A, "2023-03-10", "res_coverage") == 1.0
assert val(feat, A, "2023-07-03", "res_coverage") == 3.0
assert val(feat, A, "2023-08-29", "res_coverage") == 3.0
assert val(feat, A, "2024-06-30", "res_coverage") == 1.0
def test_res_coverage_zero_after_slideout(feat):
# C 单份 2022-06-30: 2023-08-29 窗已滑出 → 0(曾覆盖后窗空=0 值,
# 与首研报前 NaN 区分)
assert val(feat, C, "2023-08-29", "res_coverage") == 0.0
def test_res_coverage_nan_before_first(feat):
assert val(feat, A, "2023-03-09", "res_coverage") is None
# B 无研报 → 全 NaN(关注缺失≠零关注,与质押族 fill 0 语义相反)
assert val(feat, B, "2023-12-31", "res_coverage") is None
# ---------- R 族: res_rating_pos(买入+增持占比,门 n≥3) ----------
def test_res_rating_pos_value_and_gate(feat):
assert val(feat, A, "2023-07-03", "res_rating_pos") == pytest.approx(2.0 / 3.0)
# 门: 窗内 <3 条 → NaN(06-15 窗 e1,e2 共 2 条;24-06-30 窗仅 e5;
# C 单条)
assert val(feat, A, "2023-06-15", "res_rating_pos") is None
assert val(feat, A, "2024-06-30", "res_rating_pos") is None
assert val(feat, C, "2022-08-29", "res_rating_pos") is None # 窗 1 条
# ---------- R 族: res_fwd_eps(窗内 mean EPS,空值不算分母) ----------
def test_res_fwd_eps(feat):
assert val(feat, A, "2023-03-10", "res_fwd_eps") == 1.0 # 仅 e1
assert val(feat, A, "2023-07-03", "res_fwd_eps") == pytest.approx(1.05) # (1.0+1.1)/2,e3 无 eps
assert val(feat, A, "2023-08-29", "res_fwd_eps") == 1.05 # e1 未出窗(−180d=03-02)
# 24-06-30 窗仅 e5(无 eps) → NaN
assert val(feat, A, "2024-06-30", "res_fwd_eps") is None
# ---------- R 族: res_target_price(最新可见+180 天新鲜度门) ----------
def test_res_target_price_freshness(feat):
assert val(feat, A, "2023-07-03", "res_target_price") == 18.0
assert val(feat, A, "2023-08-29", "res_target_price") == 18.0 # e3 距 57 天
# e4(22.0,2023-11-17)距 2024-06-30=226 天 >180 → 过期 NaN;
# e5 tgt=None 不进流不遮蔽 → 整列 NaN(门失效会得 22.0)
assert val(feat, A, "2024-06-30", "res_target_price") is None
# 首个带价条目(e2 05-12)前 → NaN
assert val(feat, A, "2023-05-11", "res_target_price") is None
# ---------- R 族: res_rating_chg(窗内净上调次数) ----------
def test_res_rating_chg(feat):
assert val(feat, A, "2023-07-03", "res_rating_chg") == -1.0 # e3 下调
assert val(feat, A, "2023-08-29", "res_rating_chg") == -1.0
# 24-06-30 窗仅 e5(维持) → 0
assert val(feat, A, "2024-06-30", "res_rating_chg") == 0.0
assert val(feat, A, "2023-11-17", "res_rating_chg") == 0.0 # 窗(e3,e4)下调+上调相抵
# ---------- Q 族: pledge_ratio(周五快照 asof;无行 fill 0) ----------
def test_pledge_ratio_snapshot_asof(feat):
assert val(feat, A, "2023-01-05", "pledge_ratio") == 0.0 # 首期前 fill 0
assert val(feat, A, "2023-01-06", "pledge_ratio") == 5.0
assert val(feat, A, "2023-07-06", "pledge_ratio") == 5.0 # ≤asof 最近期=0106
assert val(feat, A, "2023-07-07", "pledge_ratio") == 6.0
assert val(feat, C, "2023-07-07", "pledge_ratio") == 20.0
# B 快照无行(无质押=真实零)
assert val(feat, B, "2023-12-31", "pledge_ratio") == 0.0
# ---------- Q 族: pledge_ctrl(实控人在押 max 占所持;区间重建) ----------
def test_pledge_ctrl_interval_rebuild(feat):
# p1 生效=max(start 02-10, notice 02-15)=02-15(PIT: 公告日才可见)
assert val(feat, A, "2023-02-14", "pledge_ctrl") == 0.0
assert val(feat, A, "2023-02-15", "pledge_ctrl") == 30.0
# 05-20 起 p1+p2 在押 → 实控人 max(30,50)=50
assert val(feat, A, "2023-06-30", "pledge_ctrl") == 50.0
# p2 09-10 解押 → 实控人在押仅 p1 → 30
assert val(feat, A, "2023-10-01", "pledge_ctrl") == 30.0
# C 无实控人质押行 → 0
assert val(feat, C, "2023-12-31", "pledge_ctrl") == 0.0
assert val(feat, B, "2023-12-31", "pledge_ctrl") == 0.0
# ---------- Q 族: pledge_warn_line(在押最高预估平仓线) ----------
def test_pledge_warn_line(feat):
assert val(feat, A, "2023-02-14", "pledge_warn_line") == 0.0
assert val(feat, A, "2023-06-30", "pledge_warn_line") == 25.0 # max(15,25)
assert val(feat, A, "2023-10-01", "pledge_warn_line") == 15.0 # p2 解押 → max(15,8)
assert val(feat, C, "2023-01-01", "pledge_warn_line") == 12.0 # 2021 全程在押
# ---------- Q 族: pledge_net_180(180 天窗解押−新增笔数) ----------
def test_pledge_net_chg(feat):
# A 事件: 02-15 新增 / 05-20 新增 / 08-01 新增 / 09-10 解押
assert val(feat, A, "2023-02-14", "pledge_net_180") == 0.0 # 首事件前 fill 0
assert val(feat, A, "2023-02-15", "pledge_net_180") == -1.0
assert val(feat, A, "2023-06-30", "pledge_net_180") == -2.0 # 新增 p1+p2
# 10-01 窗 (2023-04-04,10-01]: 新增 p3 − 解押 p2 相抵,但 180 天前
# 端 cum=−1(至 02-15) → 两端差 = −2 −(−1) = −1
assert val(feat, A, "2023-10-01", "pledge_net_180") == -1.0
# C 2021 事件早已出 180 天窗 → 0;B 无质押史 → 0
assert val(feat, C, "2023-01-01", "pledge_net_180") == 0.0
assert val(feat, B, "2023-12-31", "pledge_net_180") == 0.0
# ---------- 引用瘦身路径(仅 P1-C 域列请求) ----------
def test_p1c_columns_subset(synthetic_static):
df = build_fundamental_features(
[A], "2023-06-01", "2023-12-31", data_dir=synthetic_static,
columns=["pledge_ctrl", "res_coverage"])
assert df.columns == ["vt_symbol", "datetime", "pledge_ctrl", "res_coverage"]
assert val(df, A, "2023-06-30", "pledge_ctrl") == 50.0
assert val(df, A, "2023-07-03", "res_coverage") == 3.0
@@ -0,0 +1,70 @@
# tests/factor/test_fundamental_p1c_library.py
"""P1-C 批财务因子表达式库: 研报共识 5 + 质押 4 注册与契约锁定.
契约(同 P2):
- 表达式裸标识符 ⊆ adapter 特征列 + close/cs_rank(漏加列 = 拼写错)
- 全部 cs_rank 一层截面;负 IC 方向因子表达式取负(高=好统一)
- Q03 安全垫走 −pledge_warn_line/close(平仓线占现价低=厚;无在押
fill 0 → 0 值=最优档天然参与截面)
"""
import re
import sys, os
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..")))
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..", "vnpy_v4.4.0")))
import pytest
from sanguo_factor import fundamental_library # noqa: F401 import 即注册
from sanguo_factor.fundamental_adapter import FEATURE_COLUMNS
from sanguo_factor.fundamental_library import FUNDAMENTAL_P1C_FACTORS
from sanguo_factor.registry import list_factors, get_factor
@pytest.fixture(autouse=True)
def _ensure_fundamental_registered():
"""其它测试模块清空 _REGISTRY 后只重挂 alpha/builtin(顺序依赖前科),
这里逐测试幂等重注册财务因子,保证本模块与顺序无关."""
fundamental_library._register_all()
_ALLOWED = set(FEATURE_COLUMNS) | {"close", "cs_rank", "log"}
def test_p1c_9_factors_registered():
facs = {f["name"]: f for f in list_factors("fundamental")}
names = {name for name, _e, _d, _ic in FUNDAMENTAL_P1C_FACTORS}
assert len(names) == 9, f"P1-C 应 9 个,实际 {len(names)}"
for n in names:
assert n in facs, f"{n} 未注册为 fundamental"
assert facs[n]["expression"].startswith("cs_rank(")
assert facs[n]["expression"].count("cs_rank(") == 1, "必须一层截面"
def test_p1c_expressions_reference_only_known_columns():
for name, expr, _doc, _ic in FUNDAMENTAL_P1C_FACTORS:
bare = re.findall(r"[A-Za-z_][A-Za-z0-9_]*", expr)
unknown = [t for t in bare if t not in _ALLOWED]
assert not unknown, f"{name} 引用未知列 {unknown}(漏加 FEATURE_COLUMNS?)"
def test_p1c_factor_directions():
by = {name: (expr, doc) for name, expr, doc, _ic in FUNDAMENTAL_P1C_FACTORS}
# 研报: rating_pos/fwd_ep/target_upside/rating_chg 高=好 → 正;
# coverage 方向不定(高关注=拥挤假说)→ 正向注册待 IC 实证
assert by["res_rating_pos"][0] == "cs_rank(res_rating_pos)"
assert by["res_coverage"][0] == "cs_rank(res_coverage)"
assert by["res_fwd_ep"][0] == "cs_rank(res_fwd_eps / close)"
assert by["res_target_upside"][0] == "cs_rank(res_target_price / close - 1)"
assert by["res_rating_chg"][0] == "cs_rank(res_rating_chg)"
# 质押: ratio/ctrl 高=风险 → 取负;margin_dist 安全垫=−平仓线/现价;
# net_chg 解押净流入=缓解 → 正
assert by["pledge_ratio"][0] == "cs_rank(-pledge_ratio)"
assert by["pledge_ctrl"][0] == "cs_rank(-pledge_ctrl)"
assert by["pledge_margin_dist"][0] == "cs_rank(-pledge_warn_line / close)"
assert by["pledge_net_chg"][0] == "cs_rank(pledge_net_180)"
def test_p1c_doc_ids_unique():
docs = [d for _n, _e, d, _ic in FUNDAMENTAL_P1C_FACTORS]
assert docs == ["R01", "R02", "R03", "R04", "R05",
"Q01", "Q02", "Q03", "Q04"]