170 lines
7.3 KiB
Python
170 lines
7.3 KiB
Python
# tests/factor/test_sentiment_adapter.py
|
||
"""S 族情绪因子适配层: 词表打分 + corpus news_meta → PIT 日频特征.
|
||
|
||
口径锚(P2 任务书=七项目调研报告 §5.2.3,2026-09-22):
|
||
- 词表: 分析师短语优先(命中即定方向),否则正/负词计数多数决,平局 0
|
||
- PIT=次日可见: eff = show_time 所在日 + 1(宁晚毋早)
|
||
- art_code 幂等去重(核内零去重契约)
|
||
- 覆盖门: 20 行窗内新闻量 <5 → NaN(宁缺毋假,不硬算)
|
||
- 前向积累型: 无新闻股全列 NaN(≠0)
|
||
"""
|
||
import math
|
||
import sys, os
|
||
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..")))
|
||
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..", "vnpy_v4.4.0")))
|
||
|
||
import pandas as pd
|
||
import polars as pl
|
||
import pytest
|
||
|
||
from sanguo_factor.sentiment_lexicon import score_text
|
||
from sanguo_factor.sentiment_adapter import build_sentiment_features
|
||
|
||
A, B, C = "600000.SSE", "000001.SZSE", "300001.SZSE"
|
||
|
||
|
||
# ---------- 词表单元 ----------
|
||
|
||
def test_phrase_priority_over_words():
|
||
# 短语优先: 「上调评级」+多个负面词 → 仍 +1
|
||
assert score_text("机构上调评级,尽管此前有处罚与诉讼纠纷") == 1
|
||
assert score_text("下调评级,虽然公司中标新项目") == -1
|
||
|
||
|
||
def test_count_majority_and_tie():
|
||
assert score_text("公司中标重大项目") == 1
|
||
assert score_text("业绩下滑且收到处罚") == -1
|
||
assert score_text("召开股东大会审议年度报告") == 0 # 无命中
|
||
assert score_text("回购遇上减持") == 0 # 1:1 平局
|
||
|
||
|
||
# ---------- 合成 corpus 树 ----------
|
||
|
||
def _news(art, code, show, title):
|
||
return {"art_code": art, "stock_code": code,
|
||
"show_time": f"{show} 10:30:00", "title": title, "summary": "",
|
||
"media_name": "测试社", "url": "", "first_seen_date": show}
|
||
|
||
|
||
_A_NEWS = [
|
||
# 12 月六连(A06 eff=12-24=PIT 边界用例: 12-23 不可见/12-24 可见)
|
||
_news("A01", "600000", "2025-12-04", "公司中标重大项目"),
|
||
_news("A02", "600000", "2025-12-08", "营收增长30%"),
|
||
_news("A03", "600000", "2025-12-12", "被立案调查"),
|
||
_news("A04", "600000", "2025-12-16", "公司签订合同"),
|
||
_news("A05", "600000", "2025-12-20", "业绩下滑"),
|
||
_news("A06", "600000", "2025-12-23", "年报扭亏"),
|
||
# 1 月九连 + 爬虫跨天重复行(同 art_code,须幂等)
|
||
_news("A07", "600000", "2026-01-04", "回购股份方案公布"),
|
||
_news("A08", "600000", "2026-01-07", "净利润增长20%"),
|
||
_news("A09", "600000", "2026-01-11", "收到监管处罚"),
|
||
_news("A10", "600000", "2026-01-14", "多家机构上调评级"),
|
||
_news("A11", "600000", "2026-01-19", "大股东减持公告"),
|
||
_news("A11", "600000", "2026-01-20", "大股东减持公告"), # dup(同 art_code)
|
||
_news("A12", "600000", "2026-01-21", "召开股东大会"),
|
||
_news("A13", "600000", "2026-01-24", "计提商誉减值"),
|
||
_news("A14", "600000", "2026-01-26", "四季度扭亏"),
|
||
_news("A15", "600000", "2026-01-28", "收到警示函"),
|
||
]
|
||
# B: 12 连 neutral + 01-20 爆发日 4 条(全 neutral → mom=0,burst 纯量效应)
|
||
_B_NEWS = ([_news(f"B{i:02d}", "000001", f"2026-01-{d:02d}", "召开股东大会")
|
||
for i, d in enumerate(range(1, 13), 1)]
|
||
+ [_news(f"BX{i}", "000001", "2026-01-19", "公司发布公告") for i in range(4)])
|
||
|
||
|
||
@pytest.fixture()
|
||
def corpus_root(tmp_path, monkeypatch):
|
||
root = tmp_path / "corpus"
|
||
dom = root / "news_meta"
|
||
# 同日多行必须先分组再写(逐行覆盖写 part-00.parquet 会把同日早行吃掉——
|
||
# B 爆发日 4 条/A 与 B 共享分区日均中过招)
|
||
by_day: dict[str, list] = {}
|
||
for row in _A_NEWS + _B_NEWS:
|
||
by_day.setdefault(row["show_time"][:10], []).append(row)
|
||
for day, rows in by_day.items():
|
||
pdir = dom / f"dt={day}"
|
||
pdir.mkdir(parents=True, exist_ok=True)
|
||
pd.DataFrame(rows).to_parquet(pdir / "part-00.parquet", index=False)
|
||
monkeypatch.setenv("SANGUO_CORPUS_ROOT", str(root))
|
||
return root
|
||
|
||
|
||
def _val(df, vt, day, col):
|
||
import datetime as _d
|
||
row = df.filter((df["vt_symbol"] == vt)
|
||
& (df["datetime"] == _d.datetime.strptime(day, "%Y-%m-%d")))
|
||
assert row.height == 1, f"grid 缺行 {vt} {day}"
|
||
v = row[col][0]
|
||
return None if v is None else float(v)
|
||
|
||
|
||
@pytest.fixture(scope="module")
|
||
def feat_build():
|
||
return build_sentiment_features
|
||
|
||
|
||
def test_pit_next_day_visibility(corpus_root):
|
||
df = build_sentiment_features([A], "2025-11-01", "2026-01-31")
|
||
# A06(show 12-23)eff=12-24: 12-23 窗(12-04..12-23)5 条 pos3 neg2 → 0.2;
|
||
# 12-24 窗含它(6 条 pos4 neg2 → 1/3)——次日可见边界(当日盘后新闻不泄入当日)
|
||
assert _val(df, A, "2025-12-23", "sent_mom_20") == pytest.approx(0.2)
|
||
assert _val(df, A, "2025-12-24", "sent_mom_20") == pytest.approx(1.0 / 3.0)
|
||
|
||
|
||
def test_mom_window_value_and_gate(corpus_root):
|
||
df = build_sentiment_features([A], "2025-11-01", "2026-01-31")
|
||
# 01-31 的 20 行窗(01-12..01-31): eff 7 条(pos2 neg4;A11 重复行已去重)
|
||
assert _val(df, A, "2026-01-31", "sent_mom_20") == pytest.approx(-2.0 / 7.0)
|
||
# 覆盖门: 01-10 窗(12-22..01-10)仅 3 条 → NaN
|
||
assert _val(df, A, "2026-01-10", "sent_mom_20") is None
|
||
|
||
|
||
def test_art_code_dedup_in_window(corpus_root):
|
||
df = build_sentiment_features([A], "2025-11-01", "2026-01-31")
|
||
# 若 dup 未去重: 01-21 多一条 neg → mom=(2-5)/8;去重后 -2/7
|
||
assert _val(df, A, "2026-01-31", "sent_mom_20") == pytest.approx(-2.0 / 7.0)
|
||
|
||
|
||
def test_sent_chg_two_window_difference(corpus_root):
|
||
df = build_sentiment_features([A], "2025-11-01", "2026-01-31")
|
||
# 01-31: net30(01-02..01-31)=9 条 pos4 neg4 → 0;net30p(12-03..01-01)
|
||
# =6 条 pos4 neg2 → 1/3;chg = −1/3
|
||
assert _val(df, A, "2026-01-31", "sent_chg") == pytest.approx(-1.0 / 3.0)
|
||
|
||
|
||
def test_news_vol_chg_ratio(corpus_root):
|
||
df = build_sentiment_features([A], "2025-11-01", "2026-01-31")
|
||
# 01-31: c20=7(01-12..01-31),c20p=3(12-23..01-11 窗: 12-24/01-05/01-08)
|
||
# → 7/3−1 = 4/3
|
||
assert _val(df, A, "2026-01-31", "news_vol_chg") == pytest.approx(4.0 / 3.0)
|
||
|
||
|
||
def test_burst_zscore_and_gates(corpus_root):
|
||
df = build_sentiment_features([A, B], "2025-11-01", "2026-01-31")
|
||
# B 01-20(eff): 窗 20 行 = 12×1 + 4 + 7×0 → mean=0.8, var(ddof=1)=0.8
|
||
# z = (4−0.8)/sqrt(0.8)
|
||
assert _val(df, B, "2026-01-20", "sent_burst") == pytest.approx(
|
||
3.2 / math.sqrt(0.8))
|
||
# 非新闻日 → NaN(门 n>0)
|
||
assert _val(df, B, "2026-01-15", "sent_burst") is None
|
||
# 窗均量不足(0.5 门)→ NaN
|
||
assert _val(df, A, "2026-01-31", "sent_burst") is None
|
||
|
||
|
||
def test_no_news_stock_all_nan(corpus_root):
|
||
df = build_sentiment_features([A, C], "2025-11-01", "2026-01-31")
|
||
for col in ("sent_mom_20", "sent_chg", "sent_burst", "news_vol_chg"):
|
||
assert _val(df, C, "2026-01-15", col) is None, f"{col} 无新闻股应 NaN 非 0"
|
||
|
||
|
||
def test_columns_subset(corpus_root):
|
||
df = build_sentiment_features([A], "2026-01-01", "2026-01-31",
|
||
columns=["sent_mom_20"])
|
||
assert df.columns == ["vt_symbol", "datetime", "sent_mom_20"]
|
||
|
||
|
||
def test_empty_corpus_returns_schema_frame(tmp_path, monkeypatch):
|
||
monkeypatch.setenv("SANGUO_CORPUS_ROOT", str(tmp_path / "nothing"))
|
||
df = build_sentiment_features([A], "2026-01-01", "2026-01-31")
|
||
assert df.height == 0 and "sent_mom_20" in df.columns
|