feat(api): 分解器输出 schema 四字段——description 必填清洗,缺失走反馈循环 [vps]

This commit is contained in:
2026-10-09 19:53:09 +08:00
parent f3bfbc9dad
commit d43ddd851d
2 changed files with 28 additions and 6 deletions
+7 -2
View File
@@ -22,10 +22,11 @@ _SYSTEM_TEMPLATE = """你是A股量化因子工程师。给定一张已确认的
可用数据列(表达式变量只能用以下列名,禁止自造;括号内为所属数据域):
{columns}
输出格式:一个 JSON 对象 {{"factors": [{{"name": ..., "expression": ..., "justification": ...}}]}},1~8 条。
输出格式:一个 JSON 对象 {{"factors": [{{"name": ..., "expression": ..., "description": ..., "justification": ...}}]}},1~8 条。
- name: 小写字母开头的 snake_case 标识符(3~40 字符),不得与算子、列名、已有因子重名
已占用名字(不得重名):{taken}
- expression: 因子表达式,如 cs_rank(-tacc) / cs_rank(ts_mean(close, 20)) / cs_rank(np_ttm / (close * share_capital))
- description: 一句话说明该因子本身是什么、怎么算的(不超过 120 字,单行,给人工看板看的自述)
- justification: 一句话说明该因子如何承载卡片假设(不超过 200 字,单行)
要求:
@@ -91,11 +92,15 @@ def normalize_candidates(raw: Any) -> list[dict]:
name = str(f.get("name") or "").strip()
expr = str(f.get("expression") or "").strip()
just = " ".join(str(f.get("justification") or "").split())[:200]
desc = " ".join(str(f.get("description") or "").split())[:120]
if not name or not expr:
raise ValueError("候选缺 name/expression")
if not just:
raise ValueError(f"候选 {name} 缺 justification")
out.append({"name": name, "expression": expr, "justification": just})
if not desc:
raise ValueError(f"候选 {name} 缺 description")
out.append({"name": name, "expression": expr,
"description": desc, "justification": just})
names = [c["name"] for c in out]
if len(set(names)) != len(names):
raise ValueError("候选内名字重复")
+21 -4
View File
@@ -11,9 +11,10 @@ CARD = {"title": "高管增持后动量延续", "logic": "If 高管增持,则信
"dataNeeds": ["bars_daily"], "id": "hyp-1", "state": "queued"}
GOOD = {"name": "llm_mom20", "expression": "cs_rank(ts_mean(close, 20))",
"description": "20 日收盘均价动量,量度股价短期延续性",
"justification": "20 日均值动量承载延续假设"}
BAD_COL = {"name": "llm_bad", "expression": "cs_rank(pledge_pct)",
"justification": "j"}
"description": "质押比例因子", "justification": "j"}
class FakeClient:
@@ -80,9 +81,9 @@ def test_loop_name_claimed_by_passed():
# 轮1: GOOD 过门+BAD 违规;轮2 重生成者抢了 GOOD 已占有的名字→name_taken;
# 轮3 达上限停,passed 只剩轮1 的 GOOD
bad = {"name": "llm_bad", "expression": "cs_rank(pledge_pct)",
"justification": "j"}
"description": "质押比例因子", "justification": "j"}
stealer = {"name": "llm_mom20", "expression": "cs_rank(ts_std(close, 20))",
"justification": "j"}
"description": "抢名因子", "justification": "j"}
c = FakeClient([{"factors": [GOOD, bad]}, {"factors": [stealer]},
{"factors": [stealer]}])
r = asyncio.run(hd.run_decompose(c, CARD, existing_names=set(),
@@ -97,7 +98,7 @@ def test_loop_batch_internal_same_expression_different_name_fails():
"""P2-7 批内防换皮: 同批「同构换皮、异名」候选不得双双过门——
对照集须随本批 passed 的 expression 增量并入(曾只收名字)."""
twin = {"name": "llm_twin", "expression": GOOD["expression"],
"justification": "j"}
"description": "换皮因子", "justification": "j"}
c = FakeClient([{"factors": [GOOD, twin]}, {"factors": [twin]},
{"factors": [twin]}])
r = asyncio.run(hd.run_decompose(c, CARD, existing_names=set(),
@@ -118,3 +119,19 @@ def test_messages_carry_taken_names():
"violations": [Violation("unknown_column", "列 pledge_pct 不在底座可用清单")]}],
existing_names={"fund_roe_ttm"})
assert "fund_roe_ttm" in fb[0]["content"]
def test_normalize_candidates_requires_description():
"""缺 description=同缺 justification 纪律:raise 走反馈循环重生成."""
import copy
no_desc = copy.deepcopy(GOOD)
del no_desc["description"]
with pytest.raises(ValueError):
hd.normalize_candidates({"factors": [no_desc]})
def test_normalize_candidates_cleans_description():
"""description 压空白+截 120 字(justification 同款清洗纪律)."""
raw = dict(GOOD, description=" 多空格\n与换行 " + "长" * 200)
out = hd.normalize_candidates({"factors": [raw]})
assert out[0]["description"] == "多空格 与换行 " + "长" * 112