Files
sanguo_vnpy_v2/tests/factor/test_sentiment_adapter.py

170 lines
7.3 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# tests/factor/test_sentiment_adapter.py
"""S 族情绪因子适配层: 词表打分 + corpus news_meta → PIT 日频特征.
口径锚(P2 任务书=七项目调研报告 §5.2.3,2026-09-22):
- 词表: 分析师短语优先(命中即定方向),否则正/负词计数多数决,平局 0
- PIT=次日可见: eff = show_time 所在日 + 1(宁晚毋早)
- art_code 幂等去重(核内零去重契约)
- 覆盖门: 20 行窗内新闻量 <5 → NaN(宁缺毋假,不硬算)
- 前向积累型: 无新闻股全列 NaN(≠0)
"""
import math
import sys, os
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..")))
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..", "vnpy_v4.4.0")))
import pandas as pd
import polars as pl
import pytest
from sanguo_factor.sentiment_lexicon import score_text
from sanguo_factor.sentiment_adapter import build_sentiment_features
A, B, C = "600000.SSE", "000001.SZSE", "300001.SZSE"
# ---------- 词表单元 ----------
def test_phrase_priority_over_words():
# 短语优先: 「上调评级」+多个负面词 → 仍 +1
assert score_text("机构上调评级,尽管此前有处罚与诉讼纠纷") == 1
assert score_text("下调评级,虽然公司中标新项目") == -1
def test_count_majority_and_tie():
assert score_text("公司中标重大项目") == 1
assert score_text("业绩下滑且收到处罚") == -1
assert score_text("召开股东大会审议年度报告") == 0 # 无命中
assert score_text("回购遇上减持") == 0 # 1:1 平局
# ---------- 合成 corpus 树 ----------
def _news(art, code, show, title):
return {"art_code": art, "stock_code": code,
"show_time": f"{show} 10:30:00", "title": title, "summary": "",
"media_name": "测试社", "url": "", "first_seen_date": show}
_A_NEWS = [
# 12 月六连(A06 eff=12-24=PIT 边界用例: 12-23 不可见/12-24 可见)
_news("A01", "600000", "2025-12-04", "公司中标重大项目"),
_news("A02", "600000", "2025-12-08", "营收增长30%"),
_news("A03", "600000", "2025-12-12", "被立案调查"),
_news("A04", "600000", "2025-12-16", "公司签订合同"),
_news("A05", "600000", "2025-12-20", "业绩下滑"),
_news("A06", "600000", "2025-12-23", "年报扭亏"),
# 1 月九连 + 爬虫跨天重复行(同 art_code,须幂等)
_news("A07", "600000", "2026-01-04", "回购股份方案公布"),
_news("A08", "600000", "2026-01-07", "净利润增长20%"),
_news("A09", "600000", "2026-01-11", "收到监管处罚"),
_news("A10", "600000", "2026-01-14", "多家机构上调评级"),
_news("A11", "600000", "2026-01-19", "大股东减持公告"),
_news("A11", "600000", "2026-01-20", "大股东减持公告"), # dup(同 art_code)
_news("A12", "600000", "2026-01-21", "召开股东大会"),
_news("A13", "600000", "2026-01-24", "计提商誉减值"),
_news("A14", "600000", "2026-01-26", "四季度扭亏"),
_news("A15", "600000", "2026-01-28", "收到警示函"),
]
# B: 12 连 neutral + 01-20 爆发日 4 条(全 neutral → mom=0,burst 纯量效应)
_B_NEWS = ([_news(f"B{i:02d}", "000001", f"2026-01-{d:02d}", "召开股东大会")
for i, d in enumerate(range(1, 13), 1)]
+ [_news(f"BX{i}", "000001", "2026-01-19", "公司发布公告") for i in range(4)])
@pytest.fixture()
def corpus_root(tmp_path, monkeypatch):
root = tmp_path / "corpus"
dom = root / "news_meta"
# 同日多行必须先分组再写(逐行覆盖写 part-00.parquet 会把同日早行吃掉——
# B 爆发日 4 条/A 与 B 共享分区日均中过招)
by_day: dict[str, list] = {}
for row in _A_NEWS + _B_NEWS:
by_day.setdefault(row["show_time"][:10], []).append(row)
for day, rows in by_day.items():
pdir = dom / f"dt={day}"
pdir.mkdir(parents=True, exist_ok=True)
pd.DataFrame(rows).to_parquet(pdir / "part-00.parquet", index=False)
monkeypatch.setenv("SANGUO_CORPUS_ROOT", str(root))
return root
def _val(df, vt, day, col):
import datetime as _d
row = df.filter((df["vt_symbol"] == vt)
& (df["datetime"] == _d.datetime.strptime(day, "%Y-%m-%d")))
assert row.height == 1, f"grid 缺行 {vt} {day}"
v = row[col][0]
return None if v is None else float(v)
@pytest.fixture(scope="module")
def feat_build():
return build_sentiment_features
def test_pit_next_day_visibility(corpus_root):
df = build_sentiment_features([A], "2025-11-01", "2026-01-31")
# A06(show 12-23)eff=12-24: 12-23 窗(12-04..12-23)5 条 pos3 neg2 → 0.2;
# 12-24 窗含它(6 条 pos4 neg2 → 1/3)——次日可见边界(当日盘后新闻不泄入当日)
assert _val(df, A, "2025-12-23", "sent_mom_20") == pytest.approx(0.2)
assert _val(df, A, "2025-12-24", "sent_mom_20") == pytest.approx(1.0 / 3.0)
def test_mom_window_value_and_gate(corpus_root):
df = build_sentiment_features([A], "2025-11-01", "2026-01-31")
# 01-31 的 20 行窗(01-12..01-31): eff 7 条(pos2 neg4;A11 重复行已去重)
assert _val(df, A, "2026-01-31", "sent_mom_20") == pytest.approx(-2.0 / 7.0)
# 覆盖门: 01-10 窗(12-22..01-10)仅 3 条 → NaN
assert _val(df, A, "2026-01-10", "sent_mom_20") is None
def test_art_code_dedup_in_window(corpus_root):
df = build_sentiment_features([A], "2025-11-01", "2026-01-31")
# 若 dup 未去重: 01-21 多一条 neg → mom=(2-5)/8;去重后 -2/7
assert _val(df, A, "2026-01-31", "sent_mom_20") == pytest.approx(-2.0 / 7.0)
def test_sent_chg_two_window_difference(corpus_root):
df = build_sentiment_features([A], "2025-11-01", "2026-01-31")
# 01-31: net30(01-02..01-31)=9 条 pos4 neg4 → 0;net30p(12-03..01-01)
# =6 条 pos4 neg2 → 1/3;chg = −1/3
assert _val(df, A, "2026-01-31", "sent_chg") == pytest.approx(-1.0 / 3.0)
def test_news_vol_chg_ratio(corpus_root):
df = build_sentiment_features([A], "2025-11-01", "2026-01-31")
# 01-31: c20=7(01-12..01-31),c20p=3(12-23..01-11 窗: 12-24/01-05/01-08)
# → 7/3−1 = 4/3
assert _val(df, A, "2026-01-31", "news_vol_chg") == pytest.approx(4.0 / 3.0)
def test_burst_zscore_and_gates(corpus_root):
df = build_sentiment_features([A, B], "2025-11-01", "2026-01-31")
# B 01-20(eff): 窗 20 行 = 12×1 + 4 + 7×0 → mean=0.8, var(ddof=1)=0.8
# z = (4−0.8)/sqrt(0.8)
assert _val(df, B, "2026-01-20", "sent_burst") == pytest.approx(
3.2 / math.sqrt(0.8))
# 非新闻日 → NaN(门 n>0)
assert _val(df, B, "2026-01-15", "sent_burst") is None
# 窗均量不足(0.5 门)→ NaN
assert _val(df, A, "2026-01-31", "sent_burst") is None
def test_no_news_stock_all_nan(corpus_root):
df = build_sentiment_features([A, C], "2025-11-01", "2026-01-31")
for col in ("sent_mom_20", "sent_chg", "sent_burst", "news_vol_chg"):
assert _val(df, C, "2026-01-15", col) is None, f"{col} 无新闻股应 NaN 非 0"
def test_columns_subset(corpus_root):
df = build_sentiment_features([A], "2026-01-01", "2026-01-31",
columns=["sent_mom_20"])
assert df.columns == ["vt_symbol", "datetime", "sent_mom_20"]
def test_empty_corpus_returns_schema_frame(tmp_path, monkeypatch):
monkeypatch.setenv("SANGUO_CORPUS_ROOT", str(tmp_path / "nothing"))
df = build_sentiment_features([A], "2026-01-01", "2026-01-31")
assert df.height == 0 and "sent_mom_20" in df.columns