# tests/api/test_routes_pipeline_hypotheses.py """向导三端点:draft 不落库/保存再验/列表.全部 mock LLM,零真调用. 分解=job 契约(10-09):POST 202 立返+后台任务,卡片列表项 decomposeJob 轮询到终态——fixture 须 with TestClient 保活事件循环(裸用=每请求独立 portal,后台 asyncio 任务响应后即被弃);decompose_jobs 内存字典逐测清空. """ import pytest from fastapi.testclient import TestClient from sanguo_api import hypothesis_card as hc from sanguo_api.app import create_app from sanguo_api.auth import create_token, set_jwt_config DRAFT = { "title": "高管增持公告后 60 日超额收益为正", "logic": "If 高管真金白银增持,则内部人信息优势预示基本面改善", "expected_sign": "positive", "falsifiable": "若增持公告后 60 日超额收益均值<=0 则证伪", "data_needs": ["corpus_sentiment"], } @pytest.fixture() def client(tmp_path, monkeypatch): # fixture 逐字对齐 test_routes_pipeline_monthly.py 既有范式, # 外包 with 保活后台任务(job 化前提,见模块 docstring) monkeypatch.setenv("SANGUO_PIPELINE_DB", str(tmp_path / "pipeline.db")) monkeypatch.setenv("SANGUO_LLM_API_KEY", "sk-test") set_jwt_config(secret="test", expire_minutes=60) from sanguo_api import decompose_jobs decompose_jobs._JOBS.clear() decompose_jobs._LATEST.clear() app = create_app(db_path=str(tmp_path / "t.db"), file_dir=None) with TestClient(app) as c: c.headers.update({"Authorization": f"Bearer {create_token('admin')}"}) yield c @pytest.fixture() def fake_llm(client, monkeypatch): """把 LLMClient.chat_json 整个替成内存假实现(记录调用).""" import asyncio calls, queue = [], [] class Fake: def __init__(self, *a, **k): pass async def chat_json(self, messages, **k): calls.append(messages) if queue: out = queue.pop(0) if isinstance(out, Exception): raise out return out return dict(DRAFT) import sanguo_api.routes_pipeline as rp monkeypatch.setattr(rp, "LLMClient", Fake) monkeypatch.setattr(hc, "load_domains", lambda: ["corpus_sentiment", "bars_daily"]) monkeypatch.setattr(rp, "_wizard_domains", hc.load_domains) return calls, queue class TestDraft: def test_draft_returns_five_fields_and_does_not_persist(self, client, fake_llm): calls, _ = fake_llm r = client.post("/api/v1/pipeline/hypotheses/draft", json={"sentence": "高管增持之后股价会涨"}) assert r.status_code == 200 d = r.json()["draft"] assert d["expectedSign"] == "positive" assert d["dataNeeds"] == ["corpus_sentiment"] # prompt 带白名单(同源纪律) assert "corpus_sentiment" in calls[0][0]["content"] # 不落库 assert client.get("/api/v1/pipeline/hypotheses").json()["items"] == [] def test_draft_sentence_too_long_400(self, client, fake_llm): r = client.post("/api/v1/pipeline/hypotheses/draft", json={"sentence": "长" * 501}) assert r.status_code == 400 def test_draft_llm_failure_502_no_retry_at_route(self, client, fake_llm): _, queue = fake_llm from sanguo_api.llm import LLMError queue.append(LLMError("LLM 返回非 JSON(两轮): 上游体片段 sk-secret123")) r = client.post("/api/v1/pipeline/hypotheses/draft", json={"sentence": "x"}) assert r.status_code == 502 # P3-8: detail 固定文案,不透传上游响应体任何片段(端点/账号上下文) assert r.json()["detail"] == "LLM 上游返回异常" assert "sk-secret123" not in r.json()["detail"] def test_draft_llm_bad_fields_502(self, client, fake_llm): _, queue = fake_llm queue.append({**DRAFT, "expected_sign": "横盘"}) r = client.post("/api/v1/pipeline/hypotheses/draft", json={"sentence": "x"}) assert r.status_code == 502 def test_draft_unconfigured_503(self, client, monkeypatch): monkeypatch.delenv("SANGUO_LLM_API_KEY") r = client.post("/api/v1/pipeline/hypotheses/draft", json={"sentence": "x"}) assert r.status_code == 503 and "SANGUO_LLM_API_KEY" in r.json()["detail"] class TestSave: def test_save_confirmed_card_201_and_lists(self, client, fake_llm): body = {"title": DRAFT["title"], "logic": DRAFT["logic"], "expectedSign": "positive", "falsifiable": DRAFT["falsifiable"], "dataNeeds": ["corpus_sentiment"], "sentence": "高管增持之后股价会涨"} r = client.post("/api/v1/pipeline/hypotheses", json=body) assert r.status_code == 201 item = r.json()["item"] assert item["state"] == "queued" and item["id"].startswith("hyp-") items = client.get("/api/v1/pipeline/hypotheses").json()["items"] assert len(items) == 1 and items[0]["falsifiable"] == DRAFT["falsifiable"] def test_save_fabricated_domain_dropped(self, client, fake_llm): body = {"title": "t", "logic": "If a,则 b", "expectedSign": "positive", "falsifiable": "若 a<=0 则证伪", "dataNeeds": ["corpus_sentiment", "编造域"]} r = client.post("/api/v1/pipeline/hypotheses", json=body) assert r.status_code == 201 assert r.json()["item"]["dataNeeds"] == ["corpus_sentiment"] def test_save_invalid_sign_400(self, client, fake_llm): body = {"title": "t", "logic": "l", "expectedSign": "横盘", "falsifiable": "f", "dataNeeds": []} assert client.post("/api/v1/pipeline/hypotheses", json=body).status_code == 400 def test_save_sentence_too_long_400(self, client, fake_llm): body = {"title": "t", "logic": "l", "expectedSign": "positive", "falsifiable": "f", "dataNeeds": [], "sentence": "长" * 501} assert client.post("/api/v1/pipeline/hypotheses", json=body).status_code == 400 class TestAuth: def test_no_token_401(self, tmp_path): set_jwt_config(secret="test", expire_minutes=60) c = TestClient(create_app(db_path=str(tmp_path / "t.db"), file_dir=None)) assert c.get("/api/v1/pipeline/hypotheses").status_code == 401 DECOMP_GOOD = {"factors": [ {"name": "llm_mom20", "expression": "cs_rank(ts_mean(close, 20))", "justification": "20 日动量承载假设"}]} def _seed_card(client): r = client.post("/api/v1/pipeline/hypotheses", json={ "title": "t", "logic": "If a,则 b", "expectedSign": "positive", "falsifiable": "若 a<=0 则证伪", "dataNeeds": ["bars_daily"], "sentence": "s"}) return r.json()["item"]["id"] def _poll_job(client, hyp): """轮询卡片列表项 decomposeJob 到终态(job 契约=前端同款轮询源).""" import time for _ in range(200): items = client.get("/api/v1/pipeline/hypotheses").json()["items"] cur = next(i for i in items if i["id"] == hyp)["decomposeJob"] if cur and cur["status"] != "running": return cur time.sleep(0.05) raise AssertionError("decompose job 未到终态(10s)") def _wait_job(client, hyp): """POST 分解→202 立返(running)→轮询到终态,返回终态 job.""" r = client.post(f"/api/v1/pipeline/hypotheses/{hyp}/decompose") assert r.status_code == 202, r.text assert r.json()["job"]["status"] == "running" return _poll_job(client, hyp) class TestDecompose: def test_decompose_registers_and_moves_card(self, client, fake_llm, tmp_path, monkeypatch): import yaml as _yaml reg = tmp_path / "factor_reg.yaml" monkeypatch.setenv("SANGUO_FACTOR_REGISTRY", str(reg)) _, queue = fake_llm queue.append(DECOMP_GOOD) hyp = _seed_card(client) job = _wait_job(client, hyp) assert job["status"] == "completed" and job["rounds"] == 1 assert len(job["registered"]) == 1 assert job["registered"][0]["source"] == "bars_daily" items = client.get("/api/v1/pipeline/hypotheses").json()["items"] item = next(i for i in items if i["id"] == hyp) assert item["state"] == "building" assert item["factorId"] == "llm_mom20" doc = _yaml.safe_load(reg.read_text())["factors"]["llm_mom20"] assert doc["hypothesis"] == hyp and doc["status"] == "incubating" assert doc["origin"] == "decomposer" # #91① 条目级身份戳建条即落 v = doc["versions"][0] assert v["params"]["expression"].startswith("cs_rank") assert v["params"]["origin"] == "decomposer" assert v["params"]["source"] == "bars_daily" def test_decompose_feedback_second_round(self, client, fake_llm, tmp_path, monkeypatch): monkeypatch.setenv("SANGUO_FACTOR_REGISTRY", str(tmp_path / "fr.yaml")) _, queue = fake_llm queue.append({"factors": [ {"name": "llm_bad", "expression": "cs_rank(pledge_pct)", "justification": "j"}]}) queue.append(DECOMP_GOOD) hyp = _seed_card(client) job = _wait_job(client, hyp) assert job["rounds"] == 2 assert job["registered"][0]["name"] == "llm_mom20" def test_decompose_404_409_503(self, client, fake_llm, monkeypatch): assert client.post("/api/v1/pipeline/hypotheses/hyp-nope/decompose" ).status_code == 404 hyp = _seed_card(client) from sanguo_portfolio import pipeline_store pipeline_store.update_hypothesis_state( pipeline_store.default_pipeline_db_path(), hyp, "graveyard", death_reason="x") assert client.post(f"/api/v1/pipeline/hypotheses/{hyp}/decompose" ).status_code == 409 monkeypatch.delenv("SANGUO_LLM_API_KEY") hyp2 = _seed_card(client) r = client.post(f"/api/v1/pipeline/hypotheses/{hyp2}/decompose") assert r.status_code == 503 def test_decompose_llm_error_job_failed_no_leak(self, client, fake_llm): _, queue = fake_llm from sanguo_api.llm import LLMError queue.append(LLMError("LLM 重试耗尽(HTTP 429): 上游体片段 sk-secret123")) hyp = _seed_card(client) job = _wait_job(client, hyp) assert job["status"] == "failed" # P3-8: job.error 固定文案,不透传上游响应体任何片段 assert job["error"] == "LLM 上游返回异常" assert "sk-secret123" not in job["error"] def test_decompose_twice_building_is_incremental(self, client, fake_llm, tmp_path, monkeypatch): # building 态二次分解:I-1 修复——D3 矩阵无 building→building 自迁移边, # 二次分解=纯增量注册,卡片保持 building(I-1 前为 500 半成功) import yaml as _yaml reg = tmp_path / "fr_twice.yaml" monkeypatch.setenv("SANGUO_FACTOR_REGISTRY", str(reg)) _, queue = fake_llm queue.append(DECOMP_GOOD) hyp = _seed_card(client) job1 = _wait_job(client, hyp) assert job1["status"] == "completed" items = client.get("/api/v1/pipeline/hypotheses").json()["items"] assert next(i for i in items if i["id"] == hyp)["state"] == "building" queue.append({"factors": [ {"name": "llm_vol5", "expression": "cs_rank(ts_std(close, 5))", "justification": "5 日波动承载假设"}]}) job2 = _wait_job(client, hyp) assert job2["status"] == "completed" assert job2["registered"][0]["name"] == "llm_vol5" items = client.get("/api/v1/pipeline/hypotheses").json()["items"] assert next(i for i in items if i["id"] == hyp)["state"] == "building" factors = _yaml.safe_load(reg.read_text())["factors"] assert "llm_mom20" in factors and "llm_vol5" in factors # 两次注册都在 def test_decompose_lock_exists(self): # P2-6:decompose 全程持 _DECOMPOSE_LOCK(仿 _GRADUATE_LOCK,进程内 # 串行化同卡并发,registry 整文件覆盖丢更新窗口闭合)。并发真测不做 # ——线程池+async 端点组合测不划算;锁内注册+转态的行为回归由本类 # 既有各例钉死,此处只钉锁存在防回退。 import threading from sanguo_api import routes_pipeline as rp assert isinstance(rp._DECOMPOSE_LOCK, type(threading.Lock())) assert isinstance(rp._GRADUATE_LOCK, type(threading.Lock())) class TestDecomposeJobs: """job 化新增语义(10-09,QA 范式):202 立返/在飞 409/台账跨重启/僵尸判死.""" def test_post_returns_202_running_immediately(self, client, fake_llm, tmp_path, monkeypatch): import asyncio monkeypatch.setenv("SANGUO_FACTOR_REGISTRY", str(tmp_path / "r.yaml")) hyp = _seed_card(client) async def slow(client_, card, **k): await asyncio.sleep(0.4) return {"passed": [], "failed": [], "rounds": 1} import sanguo_api.routes_pipeline as rp monkeypatch.setattr(rp, "_run_decompose", slow) r = client.post(f"/api/v1/pipeline/hypotheses/{hyp}/decompose") assert r.status_code == 202, r.text job = r.json()["job"] assert job["status"] == "running" and job["jobId"].startswith("job-") # 在飞期间二次 POST → 409(同卡互斥,治重复点击堆积) r2 = client.post(f"/api/v1/pipeline/hypotheses/{hyp}/decompose") assert r2.status_code == 409 assert _poll_job(client, hyp)["status"] == "completed" def test_result_survives_memory_clear(self, client, fake_llm, tmp_path, monkeypatch): """台账跨重启:内存清空(=服务重启)后已完成结果仍可读(run 台账先于页面).""" monkeypatch.setenv("SANGUO_FACTOR_REGISTRY", str(tmp_path / "r2.yaml")) _, queue = fake_llm queue.append(DECOMP_GOOD) hyp = _seed_card(client) assert _wait_job(client, hyp)["status"] == "completed" from sanguo_api import decompose_jobs decompose_jobs._JOBS.clear() decompose_jobs._LATEST.clear() items = client.get("/api/v1/pipeline/hypotheses").json()["items"] cur = next(i for i in items if i["id"] == hyp)["decomposeJob"] assert cur["status"] == "completed" assert cur["registered"][0]["name"] == "llm_mom20" def test_zombie_running_reconciled_failed(self, client, tmp_path, monkeypatch): """台账僵尸 running(重启打断)无内存主→读侧判 failed 并回写.""" from sanguo_portfolio import pipeline_store hyp = _seed_card(client) pipeline_store.upsert_decompose_job( pipeline_store.default_pipeline_db_path(), {"jobId": "job-zombie", "hypId": hyp, "status": "running", "rounds": None, "registered": [], "failed": [], "error": None, # 未来时间戳保证它是最新的(started_at DESC 取这条) "startedAt": "2099-01-01T00:00:00", "finishedAt": None}) items = client.get("/api/v1/pipeline/hypotheses").json()["items"] cur = next(i for i in items if i["id"] == hyp)["decomposeJob"] assert cur["status"] == "failed" assert "中断" in cur["error"] # 回写台账:再读仍 failed(一处真相,非每次读侧重算) row = pipeline_store.latest_decompose_job( pipeline_store.default_pipeline_db_path(), hyp) assert row["status"] == "failed" class TestD7Visibility: """D7 落成边亮灯(10-09 拍板):三件套数据面+工厂 origin+琥珀待办.""" def test_decompose_jobs_history_endpoint(self, client, tmp_path): """件①1b:批次历史端点=台账新→旧(QA tasks/list 同形状 {items}).""" from sanguo_portfolio import pipeline_store db = pipeline_store.default_pipeline_db_path() hyp = _seed_card(client) for i, (jid, st) in enumerate([("job-old", "completed"), ("job-new", "completed")]): pipeline_store.upsert_decompose_job(db, { "jobId": jid, "hypId": hyp, "status": st, "rounds": 1, "registered": [{"name": f"f{i}", "source": "bars_daily", "expression": "e", "justification": "j"}], "failed": [], "error": None, "startedAt": f"2026-10-09T10:1{i}:00", "finishedAt": None}) r = client.get(f"/api/v1/pipeline/hypotheses/{hyp}/decompose-jobs") assert r.status_code == 200 items = r.json()["items"] assert [i["jobId"] for i in items] == ["job-new", "job-old"] assert items[0]["registered"][0]["name"] == "f1" # 不存在的卡 404 assert client.get("/api/v1/pipeline/hypotheses/hyp-x/decompose-jobs" ).status_code == 404 def test_hypotheses_list_includes_factors(self, client, fake_llm, tmp_path, monkeypatch): """件①1a:列表项内联血统反查因子清单(名/表达式/origin/落成日).""" monkeypatch.setenv("SANGUO_FACTOR_REGISTRY", str(tmp_path / "r1.yaml")) _, queue = fake_llm queue.append(DECOMP_GOOD) hyp = _seed_card(client) assert _wait_job(client, hyp)["status"] == "completed" items = client.get("/api/v1/pipeline/hypotheses").json()["items"] item = next(i for i in items if i["id"] == hyp) assert [f["name"] for f in item["factors"]] == ["llm_mom20"] assert item["factors"][0]["origin"] == "decomposer" assert item["factors"][0]["expression"].startswith("cs_rank") assert item["factors"][0]["createdAt"] # 落成日=首版 effective_from def test_progress_reports_rounds_while_running(self, client, fake_llm, tmp_path, monkeypatch): """件①1c:在飞卡片 decomposeJob.progress=进程内真轮数(非 QA 摆设).""" import asyncio monkeypatch.setenv("SANGUO_FACTOR_REGISTRY", str(tmp_path / "r3.yaml")) hyp = _seed_card(client) seen = [] async def slow(client_, card, **k): cb = k.get("on_progress") if cb: cb(round_no=1, total_rounds=3, passed=3, regen=1, message="第 1/3 轮校验完成") await asyncio.sleep(0.15) cb(round_no=2, total_rounds=3, passed=6, regen=1, message="第 2/3 轮失败者反馈重生成中") await asyncio.sleep(0.15) seen.append(True) return {"passed": [], "failed": [], "rounds": 3} import sanguo_api.routes_pipeline as rp monkeypatch.setattr(rp, "_run_decompose", slow) r = client.post(f"/api/v1/pipeline/hypotheses/{hyp}/decompose") assert r.status_code == 202 job = _poll_job(client, hyp) assert job["status"] == "completed" # 进度曾可观测(轮询窗口内读到 progress;0.15s×2>轮询间隔 50ms) items = client.get("/api/v1/pipeline/hypotheses").json()["items"] cur = next(i for i in items if i["id"] == hyp)["decomposeJob"] assert cur.get("progress") is None # 终态不带过程态 def test_progress_visible_mid_flight(self, client, fake_llm, tmp_path, monkeypatch): """件①1c 真断言:running 期间列表项 progress.currentRound 递增可见.""" import asyncio import time monkeypatch.setenv("SANGUO_FACTOR_REGISTRY", str(tmp_path / "r4.yaml")) hyp = _seed_card(client) async def slow(client_, card, **k): cb = k.get("on_progress") if cb: cb(round_no=1, total_rounds=3, passed=3, regen=1, message="第 1/3 轮校验完成") await asyncio.sleep(0.4) return {"passed": [], "failed": [], "rounds": 1} import sanguo_api.routes_pipeline as rp monkeypatch.setattr(rp, "_run_decompose", slow) assert client.post( f"/api/v1/pipeline/hypotheses/{hyp}/decompose").status_code == 202 got = None for _ in range(80): items = client.get("/api/v1/pipeline/hypotheses").json()["items"] cur = next(i for i in items if i["id"] == hyp)["decomposeJob"] if cur and cur.get("progress"): got = cur["progress"] break time.sleep(0.02) assert got and got["currentRound"] == 1 and got["totalRounds"] == 3 assert got["passed"] == 3 and got["regen"] == 1 assert _poll_job(client, hyp)["status"] == "completed" def test_todos_amber_until_decomposed(self, client, fake_llm, tmp_path, monkeypatch): """件③:queued 卡=琥珀待办;分解转 building 即消行.""" monkeypatch.setenv("SANGUO_FACTOR_REGISTRY", str(tmp_path / "r5.yaml")) _, queue = fake_llm hyp = _seed_card(client) def _amber_titles(): return [t["title"] for t in client.get( "/api/v1/pipeline/todos").json()["items"] if t["id"] == f"decompose-{hyp}"] assert _amber_titles(), "queued 卡应上琥珀待办" queue.append(DECOMP_GOOD) assert _wait_job(client, hyp)["status"] == "completed" assert not _amber_titles(), "building 卡不应再上待办" def test_factors_include_origin_and_hypothesis(self, client, fake_llm, tmp_path, monkeypatch): """件②:工厂列表带 origin 身份戳+来源卡血统(manual 缺省).""" monkeypatch.setenv("SANGUO_FACTOR_REGISTRY", str(tmp_path / "r6.yaml")) _, queue = fake_llm queue.append(DECOMP_GOOD) hyp = _seed_card(client) assert _wait_job(client, hyp)["status"] == "completed" items = client.get("/api/v1/pipeline/factors").json()["items"] row = next(i for i in items if i["id"] == "llm_mom20") assert row["origin"] == "decomposer" and row["hypothesis"] == hyp """P3-7: save 端点 source 白名单+上限(此前零白名单零上限,任意串入库).""" class TestSaveSourceGuard: """P3-7: save 端点 source 白名单+上限(此前零白名单零上限,任意串入库).""" def test_save_source_off_whitelist_400(self, client, fake_llm): body = {"title": "t", "logic": "If a,则 b", "expectedSign": "positive", "falsifiable": "若 a<=0 则证伪", "dataNeeds": [], "source": "任意自由文本来源"} r = client.post("/api/v1/pipeline/hypotheses", json=body) assert r.status_code == 400 and "source" in r.json()["detail"] def test_save_source_manual_accepted(self, client, fake_llm): body = {"title": "t", "logic": "If a,则 b", "expectedSign": "positive", "falsifiable": "若 a<=0 则证伪", "dataNeeds": [], "source": "manual"} r = client.post("/api/v1/pipeline/hypotheses", json=body) assert r.status_code == 201 assert r.json()["item"]["source"] == "manual"