Files
sanguo_vnpy_v2/sanguo_api/hypothesis_decompose.py
T
claude_dev 877723ab4a
CI/CD / test (push) Successful in 40s
CI/CD / nas-deploy (push) Successful in 30s
CI/CD / nas-verify (push) Successful in 9s
feat(api+frontend): D7 落成边亮灯——假设→因子边可见性三处 [vps]
体验稿(8823)逐条过审拍板后动工,不造新机器只做可见性:

件① 三件套(spec §7):
- 1a 卡片已产因子清单:列表项内联 factors(registry hypothesis
  血统反查:名/表达式/origin/落成日);单因子名 factorId 前端退役
- 1b 分解批次历史:GET /pipeline/hypotheses/{id}/decompose-jobs
  台账新→旧(QA tasks/list 端点形状照抄),卡片折叠区按需拉取
- 1c 轮次进度:run_decompose 每轮 on_progress 回调→内存 job
  progress(currentRound/totalRounds/passed/regen;真轮数非 QA 摆设),
  台账仍只记起跑+终态;worker 协议改收 job_id

件② 工厂来源列:factors 端点带 origin/hypothesis 血统,
⚡decomposer/✍manual 徽标+回链假设池

件③ 琥珀待办:todos 聚合加 queued/data_check 卡(已确认未分解),
分解转 building 即消行——人工卡点②显性化

件④ 结果持久落卡:前端一次性弹层退役,终态摘要+清单+批次全在卡

测试:后端 TestD7Visibility 六件(历史端点/内联清单/进度中飞可见/
终态不带过程态/琥珀消行/工厂血统)+前端三件套三测;8 目录 2780 绿
2026-10-09 11:51:00 +08:00

164 lines
7.9 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""假设卡→候选因子分解器(P4-2,决议 M②)——LLM 只写表达式,数字全由评估管线算.
反馈循环(C3):违规结构化渲染回 prompt 只重生成失败者+max_rounds 上限
(上游 QuantaAlpha proposal.py:438 while True 无上限=必修缺陷,调研报告 §4).
factor_guard 延迟导入(保持 API 启动重量不变,routes_pipeline 惰性先例).
"""
from __future__ import annotations
from typing import Any
from sanguo_api.llm import LLMClient
MAX_CANDIDATES = 8
MIN_CANDIDATES = 1
MAX_ROUNDS = 3
_SYSTEM_TEMPLATE = """你是A股量化因子工程师。给定一张已确认的投资假设卡片,把它分解为候选因子清单。你只写因子表达式(用给定算子和列),不产出任何数字结论——数字一律由评估管线计算。
可用算子(只能用以下函数,禁止自造):
{operators}
可用数据列(表达式变量只能用以下列名,禁止自造;括号内为所属数据域):
{columns}
输出格式:一个 JSON 对象 {{"factors": [{{"name": ..., "expression": ..., "justification": ...}}]}},1~8 条。
- name: 小写字母开头的 snake_case 标识符(3~40 字符),不得与算子、列名、已有因子重名
已占用名字(不得重名):{taken}
- expression: 因子表达式,如 cs_rank(-tacc) / cs_rank(ts_mean(close, 20)) / cs_rank(np_ttm / (close * share_capital))
- justification: 一句话说明该因子如何承载卡片假设(不超过 200 字,单行)
要求:
1. 只输出一个 JSON 对象,禁止 markdown 代码块和任何解释文字。
2. 每条字段单行。
3. 表达式只用上面列出的算子和列;至多引用 6 个不同列;财务列与情绪列不得混用。
4. 每条因子必须与卡片假设直接相关。"""
_FEEDBACK_TEMPLATE = """上一轮候选中以下 {n} 条未通过确定性校验(这是硬约束,不是建议)。只重新生成这几条因子:返回同样 JSON 格式,factors 数组只含重新生成的这几条,已通过的不要重复输出。
{blocks}"""
_FEEDBACK_BLOCK = """- name: {name}
expression: {expression}
未通过原因: {reasons}"""
def build_decompose_messages(card: dict,
*, feedback: list[dict] | None = None,
existing_names: set[str] | None = None
) -> list[dict]:
"""prompt 与校验白名单同源渲染(tickflow 范式,P4-1 load_domains 同纪律)."""
from sanguo_factor.factor_guard import ALLOWED_OPERATORS, COLUMN_DOMAINS
operators = "\n".join(f"- {o}" for o in sorted(ALLOWED_OPERATORS))
columns = "\n".join(f"- {c}({d})" for c, d in sorted(COLUMN_DOMAINS.items()))
taken = (",".join(sorted(existing_names)[:120])
if existing_names else "(无)") # 截断防爆 prompt
system = _SYSTEM_TEMPLATE.format(operators=operators, columns=columns,
taken=taken)
card_block = (f"假设卡片:\n标题: {card['title']}\n逻辑: {card['logic']}\n"
f"预期方向: {card['expectedSign']}\n"
f"可证伪声明: {card['falsifiable']}\n"
f"关联数据域: {', '.join(card.get('dataNeeds') or []) or '(无)'}")
user = card_block
if feedback:
blocks = "\n".join(
_FEEDBACK_BLOCK.format(
name=f["name"], expression=f["expression"],
reasons=";".join(_v_detail(v) for v in f["violations"]))
for f in feedback)
user = (card_block + "\n\n"
+ _FEEDBACK_TEMPLATE.format(n=len(feedback), blocks=blocks))
return [{"role": "system", "content": system},
{"role": "user", "content": user}]
def _v_detail(v: Any) -> str:
return v["detail"] if isinstance(v, dict) else v.detail
def normalize_candidates(raw: Any) -> list[dict]:
"""LLM 草稿形状校验(raise ValueError,中文;robust_json_parse 保证顶层 dict)."""
if not isinstance(raw, dict) or not isinstance(raw.get("factors"), list):
raise ValueError('输出须为 {"factors": [...]}')
factors = raw["factors"]
if not MIN_CANDIDATES <= len(factors) <= MAX_CANDIDATES:
raise ValueError(f"候选条数 {len(factors)} 不在 {MIN_CANDIDATES}~{MAX_CANDIDATES}")
out: list[dict] = []
for f in factors:
if not isinstance(f, dict):
raise ValueError("候选须为对象")
name = str(f.get("name") or "").strip()
expr = str(f.get("expression") or "").strip()
just = " ".join(str(f.get("justification") or "").split())[:200]
if not name or not expr:
raise ValueError("候选缺 name/expression")
if not just:
raise ValueError(f"候选 {name} 缺 justification")
out.append({"name": name, "expression": expr, "justification": just})
names = [c["name"] for c in out]
if len(set(names)) != len(names):
raise ValueError("候选内名字重复")
return out
async def run_decompose(client: LLMClient, card: dict, *,
existing_names: set[str],
library_exprs: dict[str, str],
unwired_domains: tuple[str, ...] = (),
max_rounds: int = MAX_ROUNDS,
on_progress=None) -> dict:
"""反馈循环:初代→逐条硬校验→失败者渲染违规重生成→只重验新一轮.
名字批次内占有:通过者名字进 claimed 防同批重名(失败者名字释放可重用).
on_progress(D7 件①1c):每轮校验完以进程内真轮数回调(通过/重生成计数),
QA task.progress 字段结构同位但其 currentRound 是摆设(只初始化无更新)
——这里的数据源是循环本身.
"""
from sanguo_factor.factor_guard import check_candidate
def _report(msg: str, regen: int) -> None:
if on_progress is not None:
on_progress(round_no=rounds, total_rounds=max_rounds,
passed=len(passed), regen=regen, message=msg)
unwired = tuple(d for d in unwired_domains if d)
passed: list[dict] = []
failed: list[dict] = []
rounds = 0
pending = normalize_candidates(
await client.chat_json(build_decompose_messages(card, existing_names=existing_names),
max_tokens=4000))
rounds += 1
while True:
newly_failed = []
for cand in pending:
prev = next((p for p in passed if p["name"] == cand["name"]), None)
if prev is not None and prev["expression"] == cand["expression"]:
continue # 原样复述已通过因子=同一因子跨轮保有,不重复收非 name_taken
claimed = existing_names | {p["name"] for p in passed}
# P2-7 批内防换皮: 对照集随本批 passed 的 expression 增量并入
# (曾只收名字——同构换皮异名候选可双双过门一次成型)
batch_libs = {**library_exprs,
**{p["name"]: p["expression"] for p in passed}}
violations = check_candidate(
cand["name"], cand["expression"], existing_names=claimed,
library_exprs=batch_libs, unwired_domains=unwired)
(newly_failed.append({**cand, "violations": violations})
if violations else passed.append(cand))
failed = newly_failed
_report(f"第 {rounds}/{max_rounds} 轮校验完成", len(failed))
if not failed or rounds >= max_rounds:
break
_report(f"第 {rounds}/{max_rounds} 轮失败者反馈重生成中", len(failed))
pending = normalize_candidates(await client.chat_json(
build_decompose_messages(card, feedback=failed,
existing_names=existing_names),
max_tokens=4000))
rounds += 1
return {"passed": passed,
"failed": [{**f, "violations": [{"code": v.code, "detail": v.detail}
for v in f["violations"]]}
for f in failed],
"rounds": rounds}