Files
sanguo_vnpy_v2/scripts/data_platform/parse_csindex_announce.py
T
claude_dev c2a89d01a4 feat(data): 方案A 数据层落地(spec §14)— DB唯一表+治幸存者偏差+权威源
spec §14 方案A 数据层迁移完成 + E2E 验证(read_db_daily: 在市/退市治偏差/ETF 全OK):
- dbbardata('d') 1826万含退市(治回测幸存者偏差, INSERT OR REPLACE staging迁移, WHERE OHLC NOT NULL+COALESCE)
- constituent_unified 7110行/9指数(300/500/50 baostock全集 + 深证4指 akshare cni union + 中证1000/2000 snapshot)
- pe/pb 不进DB -> valuation_baostock/<year>.parquet 按年宽表(2003-2026)
- 废弃 daily_baostock_full/bs_index_constituent(rename _old 保留); 旧4 schtask disabled
- 新 schtask sanguo-bs-eod 18:05(baostock个股日线+15min+拆pe/pb DAILY_LIMIT 48000) + sanguo-xt-eod 18:40(ETF/基金xtata)
- 权威源: baostock个股日线+估值+15min+复权+300/500/50 / xtata ETF+基金+当天实时 / akshare三表+事件+深证中证成份股
- 全程备份+staging+_old保留可回滚; 脚本 audit/probe/migrate/merge/cleanup/fix_config/verify/bs_eod/xt_eod/wrapper/register_schtasks
- 待办(spec §6 使用层): LocalParquetProvider 接 constituent_unified+valuation_baostock + 实时拼接
2026-07-23 07:20:34 +08:00

581 lines
23 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""csindex 公告解析 - 重建中证1000/2000 历史成份股曾经入选集 (治幸存者偏差)
数据源:
1. 列表 API: POST https://www.csindex.com.cn/csindex-home/announcement/queryAnnouncementByVo
payload: {lang, classlist, indexlist, page:{desc,key,page,rows}, related_topics, typelist}
2. 详情 API: GET https://www.csindex.com.cn/csindex-home/announcement/queryAnnouncementById?id={nid}
{content (HTML), enclosureList:[{fileUrl, fileName}]}
3. 附件: PDF (定期调整 多指数合并) 或 xlsx/xls (临时调整 单/多指数)
实证覆盖 (2026-07-22 实地验证):
CSI 1000 (000852) 2014-09 发布:
- 2018-07-11 起 csindex 有调整公告 (28 份: 7 定期 + 21 临时)
- 2014-2018 期间的 June/Dec 定期调整不在 csindex (gap: ~7 round)
CSI 2000 (932000) 2023-08-10 发布:
- csindex 上无任何 CSI 2000 样本调整公告 (定期调整 PDF 中均无 CSI 2000 section)
- 仅 launch xlsx 有初始 2000 只样本 (2023-08-10)
- 历史/最新调整需查指数详情页 "拟生效样本" 或 wind/choice (不在本脚本范围)
输出:
data/index_const_hist/000852_announce_union.parquet - CSI 1000 曾经入选集
data/index_const_hist/932000_announce_union.parquet - CSI 2000 (initial only)
schema: updateDate / index_code / code / code_name / adjust_type / notice_id / source
约束:
- 直连不走代理 (unset)
- 串行 + sleep(1.0~1.5s)
- UA header
"""
import argparse
import json
import logging
import os
import re
import socket
import sys
import time
import urllib.error
import urllib.parse
import urllib.request
from pathlib import Path
from typing import Dict, List, Optional, Tuple
# ======================== 硬约束 ========================
for _k in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY", "all_proxy", "ALL_PROXY"):
os.environ.pop(_k, None)
socket.setdefaulttimeout(30)
try:
sys.stdout.reconfigure(line_buffering=True)
except (AttributeError, ValueError):
pass
import pandas as pd # noqa: E402
import openpyxl # noqa: E402
import pdfplumber # noqa: E402
# ======================== 配置 ========================
BASE = "https://www.csindex.com.cn/csindex-home"
LIST_URL = f"{BASE}/announcement/queryAnnouncementByVo"
DETAIL_URL = f"{BASE}/announcement/queryAnnouncementById"
UA = "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36"
DEFAULT_OUT_DIR = "/Users/chufeng/.openclaw/sanguo_projects/sanguo_vnpy_v2/data/index_const_hist"
OUT_DIR = Path(os.environ.get("INDEX_HIST_OUT_DIR", DEFAULT_OUT_DIR))
CACHE_DIR = Path(os.environ.get("CSINDEX_CACHE", "/tmp/csindex_raw"))
DETAIL_CACHE_DIR = CACHE_DIR / "detail"
# CSI 1000 相关调整公告 id (实证已发现, 脚本也支持自动发现)
CSI1000_REGULAR_IDS = [14796, 15044, 15267, 15471, 15690, 3006000, 3006137]
CSI1000_TEMP_IDS = [
12446, 12965, 13070, 13212, 13281, 13334, 13765, 14092,
14842, 15019, 15232, 15342, 15357, 15390, 15575, 15613, 15648,
1006042, 3006027, 3006041, 3006120,
]
# CSI 2000 launch xlsx (initial 2000 stocks)
CSI2000_LAUNCH_ID = 14883
# ======================== logging ========================
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s [%(levelname)s] %(message)s",
datefmt="%H:%M:%S",
)
log = logging.getLogger("csindex")
# ======================== HTTP ========================
def _post_json(url: str, payload: dict, timeout: int = 30) -> dict:
data = json.dumps(payload).encode("utf-8")
req = urllib.request.Request(
url, data=data, method="POST",
headers={"Content-Type": "application/json", "User-Agent": UA},
)
with urllib.request.urlopen(req, timeout=timeout) as r:
return json.loads(r.read())
def _get_json(url: str, timeout: int = 30) -> dict:
req = urllib.request.Request(url, headers={"User-Agent": UA})
with urllib.request.urlopen(req, timeout=timeout) as r:
return json.loads(r.read())
def _download(url: str, target: Path, timeout: int = 60) -> int:
"""下载含中文名 URL (自动 quote, 防双重编码)"""
p = urllib.parse.urlsplit(url)
path_q = p.path if "%" in p.path else urllib.request.quote(p.path)
quoted = urllib.parse.urlunsplit((p.scheme, p.netloc, path_q, p.query, p.fragment))
req = urllib.request.Request(quoted, headers={"User-Agent": UA})
with urllib.request.urlopen(req, timeout=timeout) as r:
data = r.read()
target.parent.mkdir(parents=True, exist_ok=True)
with open(target, "wb") as f:
f.write(data)
return len(data)
# ======================== 1. 列表 ========================
def fetch_all_notices(cache_path: Path, force: bool = False) -> List[dict]:
"""拉全量公告列表 (cached)"""
if cache_path.exists() and not force:
log.info(f"使用缓存列表: {cache_path}")
with open(cache_path, encoding="utf-8") as f:
return json.load(f)
log.info(f"分页拉取全量公告: {LIST_URL}")
all_items = []
page = 1
while True:
payload = {
"lang": "cn", "classlist": [], "indexlist": [],
"page": {"desc": "", "key": "", "page": page, "rows": 100},
"related_topics": [], "typelist": [],
}
try:
d = _post_json(LIST_URL, payload)
except Exception as e:
log.error(f"page {page} err: {e}; retry once after 5s")
time.sleep(5)
d = _post_json(LIST_URL, payload)
if d.get("code") != "200":
log.error(f"page {page} API err: {d}")
break
items = d.get("data") or []
all_items.extend(items)
total = d.get("total") or 0
log.info(f" page {page}: +{len(items)} (cum={len(all_items)}/{total})")
if not items or len(all_items) >= total:
break
page += 1
time.sleep(1.0)
cache_path.parent.mkdir(parents=True, exist_ok=True)
with open(cache_path, "w", encoding="utf-8") as f:
json.dump(all_items, f, ensure_ascii=False)
log.info(f"缓存列表: {cache_path} (total={len(all_items)})")
return all_items
def filter_csi1000_notices(notices: List[dict]) -> List[dict]:
"""筛 CSI 1000 调整公告 (theme=指数调样 + title 含 中证1000)"""
out = []
seen_ids = set()
for x in notices:
title = x.get("title") or ""
if x.get("theme") != "指数调样":
continue
if "中证1000" not in title:
continue
if "调整" not in title:
continue
# 排除 "不实施" 通知
if "不实施" in title:
continue
if x["id"] in seen_ids:
continue
seen_ids.add(x["id"])
out.append(x)
out.sort(key=lambda x: x.get("publishDate", ""))
return out
# ======================== 2. 详情 + 下载 ========================
def fetch_detail(nid: int, cache_dir: Path) -> dict:
"""详情 (cached by id)"""
cache_file = cache_dir / f"detail_{nid}.json"
if cache_file.exists():
with open(cache_file, encoding="utf-8") as f:
return json.load(f)
try:
d = _get_json(f"{DETAIL_URL}?id={nid}")
except Exception as e:
log.error(f"detail {nid} err: {e}")
return {}
cache_file.parent.mkdir(parents=True, exist_ok=True)
with open(cache_file, "w", encoding="utf-8") as f:
json.dump(d, f, ensure_ascii=False)
time.sleep(1.0)
return d
def extract_file_urls(detail_data: dict) -> List[Tuple[str, str]]:
"""从 enclosureList + content 内嵌 href 抽 (fileName, fileUrl)"""
urls = []
for e in (detail_data.get("enclosureList") or []):
fu = e.get("fileUrl")
fn = e.get("fileName") or "file"
if fu:
urls.append((fn, fu))
content = detail_data.get("content") or ""
for m in re.finditer(r'href="([^"]+)"', content):
fu = m.group(1)
if re.search(r"\.(xlsx|pdf|xls|csv)(\?|$)", fu, re.I):
urls.append(("inline", fu))
# 去重保序
seen, out = set(), []
for fn, fu in urls:
if fu not in seen:
seen.add(fu)
out.append((fn, fu))
return out
def download_notice_files(nid: int, publish_date: str, urls: List[Tuple[str, str]], cache_dir: Path) -> List[Path]:
"""下载该公告的所有附件,返本地路径列表"""
out = []
for fn, fu in urls:
ext = fu.rsplit(".", 1)[-1].lower().split("?")[0]
local = cache_dir / f"{nid}_{publish_date}.{ext}"
if not local.exists():
try:
sz = _download(fu, local)
log.info(f" dl [{sz}B] {fn} -> {local.name}")
except Exception as ex:
log.error(f" dl ERR {fu}: {ex}")
continue
else:
log.info(f" cache {local.name}")
out.append(local)
return out
# ======================== 3. 解析 ========================
def _norm_code(c) -> Optional[str]:
"""规整为 6 位字符串代码 (前补 0)"""
if c is None:
return None
s = str(c).strip()
# 排除空/NA
if not s or s.lower() in ("nan", "none", ""):
return None
# 纯数字 (可能 int 转 str 失去前导 0)
if s.isdigit():
s = s.zfill(6)
if len(s) > 6:
return None # 异常长
return s
# H开头 (港股代码, CSI 1000 不含, 跳过)
if s.startswith("H") or s.startswith("688") and not s.isdigit():
return None
# 6 位字母数字 (如 SH/SZ 前缀)
m = re.search(r"(\d{6})", s)
return m.group(1) if m else None
def parse_xlsx_adjustments(path: Path, target_index_code: str = "000852") -> Tuple[List[dict], List[dict]]:
"""解析 xlsx/xls -> (add_rows, remove_rows)
每个 row = {code, code_name}
Sheet 名: 调入/调出 或 换入/换出 (反向)
"""
add_rows, remove_rows = [], []
if path.suffix.lower() == ".xls":
xl = pd.ExcelFile(path)
df_map = {sn: xl.parse(sn, dtype=str) for sn in xl.sheet_names}
else:
wb = openpyxl.load_workbook(path, data_only=True)
df_map = {}
for sn in wb.sheetnames:
ws = wb[sn]
rows = list(ws.iter_rows(values_only=True))
if not rows:
continue
df_map[sn] = pd.DataFrame(rows[1:], columns=rows[0])
for sn, df in df_map.items():
if df.empty or len(df.columns) < 4:
continue
# 判定方向
sn_norm = sn.strip()
if sn_norm in ("调入", "换入", "新增"):
direction = "add"
elif sn_norm in ("调出", "换出", "删除"):
direction = "remove"
else:
continue
# 筛 target index
col0 = df.iloc[:, 0].astype(str).str.strip()
mask = col0 == target_index_code
sub = df[mask]
for _, row in sub.iterrows():
code = _norm_code(row.iloc[2])
name = str(row.iloc[3]).strip() if row.iloc[3] is not None else ""
if not code:
continue
r = {"code": code, "code_name": name}
if direction == "add":
add_rows.append(r)
else:
remove_rows.append(r)
return add_rows, remove_rows
def parse_pdf_adjustments(path: Path, target_section: str = "中证1000") -> Tuple[List[dict], List[dict]]:
"""解析 PDF 的指定指数 section
target_section: '中证1000' or '中证2000'
返 (add_rows, remove_rows)
"""
with pdfplumber.open(path) as pdf:
full_text = "\n".join((p.extract_text() or "") for p in pdf.pages)
# 定位所有 section header
header_re = re.compile(
r"(沪深300|中证500|中证1000|中证2000|中证A\d+|上证\d+|科创50|北证\d+)\s*指数样本调整名单[:]?"
)
headers = list(header_re.finditer(full_text))
if not headers:
return [], []
# 找 target section
target_text = None
for i, m in enumerate(headers):
if m.group(1) == target_section:
start = m.end()
end = headers[i + 1].start() if i + 1 < len(headers) else len(full_text)
target_text = full_text[start:end].strip()
break
if not target_text:
return [], []
# 解析 4 列 (调出代码 调出名称 调入代码 调入名称)
add_rows, remove_rows = [], []
in_data = False
last_side = None # 处理 2 列 (只有一侧) 的情况
for ln in target_text.split("\n"):
ln = ln.strip()
if not ln:
continue
# 表头行跳过
if re.search(r"证券代码|股票代码|调出名单|调入名单|指数代码|换出|换入", ln):
in_data = True
if "调出名单" in ln and "调入名单" not in ln:
last_side = "remove"
elif "调入名单" in ln and "调出名单" not in ln:
last_side = "add"
else:
last_side = None
continue
if not in_data:
continue
# 4 列
m4 = re.match(r"^(\d{6})\s+(\S+?)\s+(\d{6})\s+(\S+)", ln)
if m4:
rc, rn, ac, an = m4.group(1), m4.group(2), m4.group(3), m4.group(4)
remove_rows.append({"code": rc, "code_name": rn})
add_rows.append({"code": ac, "code_name": an})
last_side = None
continue
# 2 列 (单侧)
m2 = re.match(r"^(\d{6})\s+(\S+)", ln)
if m2:
r = {"code": m2.group(1), "code_name": m2.group(2)}
if last_side == "add":
add_rows.append(r)
elif last_side == "remove":
remove_rows.append(r)
return add_rows, remove_rows
# ======================== 4. 主流程 ========================
def process_notice(nid: int, detail_cache: Path, file_cache: Path,
target_index_code: str, target_section: str) -> List[dict]:
"""处理单个公告 -> 返回 records 列表
record: {updateDate, index_code, code, code_name, adjust_type, notice_id, source}
"""
detail = fetch_detail(nid, detail_cache)
if not detail or not detail.get("data"):
log.warning(f" {nid}: detail empty")
return []
data = detail["data"]
publish_date = (data.get("publishDate") or "").split("T")[0]
title = data.get("title") or ""
urls = extract_file_urls(data)
files = download_notice_files(nid, publish_date, urls, file_cache)
if not files:
log.warning(f" {nid} {publish_date}: no files; title={title}")
return []
records = []
for f in files:
ext = f.suffix.lower()
try:
if ext == ".pdf":
add, rem = parse_pdf_adjustments(f, target_section=target_section)
elif ext in (".xlsx", ".xls"):
add, rem = parse_xlsx_adjustments(f, target_index_code=target_index_code)
else:
continue
except Exception as e:
log.error(f" parse {f.name} ERR: {e}")
continue
for r in add:
records.append({
"updateDate": publish_date, "index_code": target_index_code,
"code": r["code"], "code_name": r["code_name"],
"adjust_type": "add", "notice_id": nid, "source": f.name,
})
for r in rem:
records.append({
"updateDate": publish_date, "index_code": target_index_code,
"code": r["code"], "code_name": r["code_name"],
"adjust_type": "remove", "notice_id": nid, "source": f.name,
})
log.info(f" {nid} {publish_date}: parsed add={sum(1 for r in records if r['adjust_type']=='add')}, "
f"remove={sum(1 for r in records if r['adjust_type']=='remove')}, title={title}")
return records
def main():
parser = argparse.ArgumentParser()
parser.add_argument("--out-dir", default=str(OUT_DIR))
parser.add_argument("--cache-dir", default=str(CACHE_DIR))
parser.add_argument("--refresh-list", action="store_true", help="强制重新拉全量公告列表")
parser.add_argument("--only", choices=["1000", "2000", "both"], default="both")
args = parser.parse_args()
out_dir = Path(args.out_dir)
out_dir.mkdir(parents=True, exist_ok=True)
cache_dir = Path(args.cache_dir)
detail_cache = cache_dir / "detail"
file_cache = cache_dir / "files"
# 1. 拉/缓存全量列表
notices = fetch_all_notices(cache_dir / "all_notices.json", force=args.refresh_list)
# ========== CSI 1000 ==========
if args.only in ("1000", "both"):
log.info("\n========== CSI 1000 (000852) ==========")
# 过滤 + 附加已知 id
filtered = filter_csi1000_notices(notices)
log.info(f"filtered CSI 1000 adjustment notices: {len(filtered)}")
all_ids = sorted({x["id"] for x in filtered} | set(CSI1000_REGULAR_IDS) | set(CSI1000_TEMP_IDS))
log.info(f"total CSI 1000 notice ids to process: {len(all_ids)}")
records_1000 = []
for nid in all_ids:
try:
recs = process_notice(nid, detail_cache, file_cache, "000852", "中证1000")
records_1000.extend(recs)
except Exception as e:
log.error(f" {nid} FAILED: {e}")
# 加 initial 集合 (launch 2014-09 -> 2014-09-18 id=2998, 无初始样本 xlsx, 跳过)
# 加 current (akshare 现快照)
try:
import akshare as ak
log.info("拉 akshare 中证1000 当前快照...")
df = ak.index_stock_cons_csindex(symbol="000852")
for _, row in df.iterrows():
code = str(row.iloc[0]).zfill(6) if str(row.iloc[0]).isdigit() else str(row.iloc[0])
if not code.isdigit() or len(code) != 6:
continue
records_1000.append({
"updateDate": "current", "index_code": "000852",
"code": code, "code_name": str(row.iloc[1]) if df.shape[1] > 1 else "",
"adjust_type": "current", "notice_id": 0, "source": "akshare.index_stock_cons_csindex",
})
log.info(f" akshare current: {sum(1 for r in records_1000 if r['adjust_type']=='current')} stocks")
except Exception as e:
log.warning(f"akshare 当前快照拉取失败 (非致命): {e}")
df_1000 = pd.DataFrame(records_1000, columns=[
"updateDate", "index_code", "code", "code_name", "adjust_type", "notice_id", "source"])
out_1000 = out_dir / "000852_announce_union.parquet"
df_1000.to_parquet(out_1000, index=False)
distinct_1000 = df_1000["code"].nunique()
n_add = (df_1000["adjust_type"] == "add").sum()
n_rem = (df_1000["adjust_type"] == "remove").sum()
n_cur = (df_1000["adjust_type"] == "current").sum()
log.info(f"\nCSI 1000 result: {out_1000}")
log.info(f" total records: {len(df_1000)} (add={n_add}, remove={n_rem}, current={n_cur})")
log.info(f" distinct codes (ever-included): {distinct_1000}")
log.info(f" date range: {df_1000['updateDate'].min()} ~ {df_1000['updateDate'].max()}")
log.info(f" distinct notice_ids: {df_1000['notice_id'].nunique()}")
# ========== CSI 2000 ==========
if args.only in ("2000", "both"):
log.info("\n========== CSI 2000 (932000) ==========")
log.warning("GAP: csindex 公告系统未发布任何 CSI 2000 样本调整公告")
log.warning(" 仅可从 2023-08-10 launch xlsx 获取初始 2000 只样本")
records_2000 = []
# launch xlsx (initial)
try:
detail = fetch_detail(CSI2000_LAUNCH_ID, detail_cache)
if detail.get("data"):
data = detail["data"]
publish_date = (data.get("publishDate") or "2023-08-10").split("T")[0]
urls = extract_file_urls(data)
files = download_notice_files(CSI2000_LAUNCH_ID, publish_date, urls, file_cache)
for f in files:
if f.suffix.lower() != ".xlsx":
continue
wb = openpyxl.load_workbook(f, data_only=True)
for sn in wb.sheetnames:
ws = wb[sn]
rows = list(ws.iter_rows(values_only=True))
if not rows:
continue
header = rows[0]
# 找 code 列 (通常第 1 列 或 命名"证券代码"/"成分券代码")
for row in rows[1:]:
if not row:
continue
code = _norm_code(row[0] if len(row) > 0 else None)
if not code:
continue
name = str(row[1]).strip() if len(row) > 1 and row[1] else ""
records_2000.append({
"updateDate": publish_date, "index_code": "932000",
"code": code, "code_name": name,
"adjust_type": "initial", "notice_id": CSI2000_LAUNCH_ID, "source": f.name,
})
break # 只用第一个 xlsx
except Exception as e:
log.error(f"CSI 2000 launch xlsx ERR: {e}")
# current snapshot via akshare
try:
import akshare as ak
log.info("拉 akshare 中证2000 当前快照...")
df = ak.index_stock_cons_csindex(symbol="932000")
for _, row in df.iterrows():
code = str(row.iloc[0]).zfill(6) if str(row.iloc[0]).isdigit() else str(row.iloc[0])
if not code.isdigit() or len(code) != 6:
continue
records_2000.append({
"updateDate": "current", "index_code": "932000",
"code": code, "code_name": str(row.iloc[1]) if df.shape[1] > 1 else "",
"adjust_type": "current", "notice_id": 0, "source": "akshare.index_stock_cons_csindex",
})
log.info(f" akshare current: {sum(1 for r in records_2000 if r['adjust_type']=='current')} stocks")
except Exception as e:
log.warning(f"akshare CSI 2000 当前快照失败: {e}")
df_2000 = pd.DataFrame(records_2000, columns=[
"updateDate", "index_code", "code", "code_name", "adjust_type", "notice_id", "source"])
out_2000 = out_dir / "932000_announce_union.parquet"
df_2000.to_parquet(out_2000, index=False)
distinct_2000 = df_2000["code"].nunique()
n_init = (df_2000["adjust_type"] == "initial").sum()
n_cur = (df_2000["adjust_type"] == "current").sum()
log.info(f"\nCSI 2000 result: {out_2000}")
log.info(f" total records: {len(df_2000)} (initial={n_init}, current={n_cur})")
log.info(f" distinct codes: {distinct_2000}")
log.info(f" GAP: 无调整公告,初始集 ∪ 当前集 (中间调整无记录)")
# 总结
log.info("\n========== DONE ==========")
log.info(f"OUT_DIR: {out_dir.resolve()}")
for f in sorted(out_dir.glob("*_announce_union.parquet")):
log.info(f" {f.name}: {f.stat().st_size} bytes")
if __name__ == "__main__":
sys.exit(main())