feat(data): 行业指数治偏差 content HTML 解析(was_removed 8->99)
parse_csindex_announce 加 parse_content_adjustments: 公告无 PDF/xlsx 附件时 fallback 解析 HTML content 嵌入文字表格(1208 公告 2009 中证行业调整名单, 10 行业各一表)。共享 _extract_section_records 给 PDF/content 两路, 非破坏(1000/2000 PDF 路径不变; 实测 000852 was_removed=672 / 000905=1303 等宽基未变)。process_notice 附件空时 fallback content; SHARED_NOTICES_BY_INDEX 补 1208->全 10 行业(csindex API indexCode 漏绑 000936)。 结果: constituent_unified 行业指数 was_removed 8->99(000928-937 全 >0)。 局限: 仅覆盖 2009 那次 + 零星 2016-2022 临时调整; 2010-2025 半年度定期调整 csindex 公告 系统无完整存档, 需 wind/choice/官网另补(本脚本范围外)。
This commit is contained in:
@@ -367,17 +367,13 @@ def parse_xlsx_adjustments(path: Path, target_index_code: str = "000852") -> Tup
|
|||||||
return add_rows, remove_rows
|
return add_rows, remove_rows
|
||||||
|
|
||||||
|
|
||||||
def parse_pdf_adjustments(path: Path, target_section: str = "中证1000") -> Tuple[List[dict], List[dict]]:
|
def _extract_section_records(full_text: str, target_section: str) -> Tuple[List[dict], List[dict]]:
|
||||||
"""解析 PDF 的指定指数 section
|
"""从纯文本中定位 target_section 调整名单 + 抽 4 列/2 列 records。
|
||||||
target_section: '中证1000' / '中证2000' / '中证全指' / '中证能源' 等
|
PDF (pdfplumber extract_text) 和 HTML content (strip 标签后) 共用此逻辑。
|
||||||
返 (add_rows, remove_rows)
|
|
||||||
|
|
||||||
header_re 已泛化: 支持任意"中证XXX"/"沪深XXX"/"上证XXX"/"深证XXX"/"中证N位数字" 等指数简称。
|
header_re 已泛化: 支持任意"中证XXX"/"沪深XXX"/"上证XXX"/"深证XXX"/"中证N位数字" 等指数简称。
|
||||||
target_section 精确匹配 group(1)。
|
target_section 精确匹配 group(1)。
|
||||||
"""
|
"""
|
||||||
with pdfplumber.open(path) as pdf:
|
|
||||||
full_text = "\n".join((p.extract_text() or "") for p in pdf.pages)
|
|
||||||
|
|
||||||
# 定位所有 section header (泛化: 数字代号 + 中文简称均可)
|
# 定位所有 section header (泛化: 数字代号 + 中文简称均可)
|
||||||
header_re = re.compile(
|
header_re = re.compile(
|
||||||
r"(沪深300|沪深[一-龥]{1,10}|"
|
r"(沪深300|沪深[一-龥]{1,10}|"
|
||||||
@@ -406,24 +402,40 @@ def parse_pdf_adjustments(path: Path, target_section: str = "中证1000") -> Tup
|
|||||||
add_rows, remove_rows = [], []
|
add_rows, remove_rows = [], []
|
||||||
in_data = False
|
in_data = False
|
||||||
last_side = None # 处理 2 列 (只有一侧) 的情况
|
last_side = None # 处理 2 列 (只有一侧) 的情况
|
||||||
|
# HTML table 横向表头 ("调出|调入" 同行): 左侧 side 通常='remove', 右侧='add'
|
||||||
|
# 用于:csindex content HTML 表格, 某行 4 列中右侧 2 列为 nbsp 空时, fallback 把左 2 列归到 left_side
|
||||||
|
header_orientation = None # 'horizontal' or None
|
||||||
|
left_side = None
|
||||||
for ln in target_text.split("\n"):
|
for ln in target_text.split("\n"):
|
||||||
ln = ln.strip()
|
ln = ln.strip()
|
||||||
if not ln:
|
if not ln:
|
||||||
continue
|
continue
|
||||||
# 表头行跳过
|
# 表头行跳过 (含横向 "调出 调入" 同行表头, csindex HTML table 典型)
|
||||||
if re.search(r"证券代码|股票代码|调出名单|调入名单|指数代码|换出|换入", ln):
|
if re.search(r"证券代码|股票代码|调出名单|调入名单|指数代码|换出|换入|"
|
||||||
|
r"调出\s+调入|调入\s+调出", ln):
|
||||||
in_data = True
|
in_data = True
|
||||||
if "调出名单" in ln and "调入名单" not in ln:
|
if "调出名单" in ln and "调入名单" not in ln:
|
||||||
last_side = "remove"
|
last_side = "remove"
|
||||||
|
header_orientation = None
|
||||||
elif "调入名单" in ln and "调出名单" not in ln:
|
elif "调入名单" in ln and "调出名单" not in ln:
|
||||||
last_side = "add"
|
last_side = "add"
|
||||||
|
header_orientation = None
|
||||||
|
elif re.search(r"调出.*调入", ln):
|
||||||
|
# 横向表头 "调出 调入" (HTML table): 左 remove 右 add
|
||||||
|
header_orientation = "horizontal"
|
||||||
|
left_side = "remove"
|
||||||
|
last_side = None
|
||||||
|
elif re.search(r"调入.*调出", ln):
|
||||||
|
header_orientation = "horizontal"
|
||||||
|
left_side = "add"
|
||||||
|
last_side = None
|
||||||
else:
|
else:
|
||||||
last_side = None
|
last_side = None
|
||||||
continue
|
continue
|
||||||
if not in_data:
|
if not in_data:
|
||||||
continue
|
continue
|
||||||
# 4 列
|
# 4 列 (调出代码 调出名称 调入代码 调入名称), 名称用 .+? 支持含空格 (如 "ST 锦化")
|
||||||
m4 = re.match(r"^(\d{6})\s+(\S+?)\s+(\d{6})\s+(\S+)", ln)
|
m4 = re.match(r"^(\d{6})\s+(.+?)\s+(\d{6})\s+(.+)$", ln)
|
||||||
if m4:
|
if m4:
|
||||||
rc, rn, ac, an = m4.group(1), m4.group(2), m4.group(3), m4.group(4)
|
rc, rn, ac, an = m4.group(1), m4.group(2), m4.group(3), m4.group(4)
|
||||||
remove_rows.append({"code": rc, "code_name": rn})
|
remove_rows.append({"code": rc, "code_name": rn})
|
||||||
@@ -438,14 +450,65 @@ def parse_pdf_adjustments(path: Path, target_section: str = "中证1000") -> Tup
|
|||||||
add_rows.append(r)
|
add_rows.append(r)
|
||||||
elif last_side == "remove":
|
elif last_side == "remove":
|
||||||
remove_rows.append(r)
|
remove_rows.append(r)
|
||||||
|
elif header_orientation == "horizontal" and left_side in ("add", "remove"):
|
||||||
|
# HTML 横向表头下, 单独 2 列通常左侧 (右侧 nbsp 空)
|
||||||
|
(add_rows if left_side == "add" else remove_rows).append(r)
|
||||||
return add_rows, remove_rows
|
return add_rows, remove_rows
|
||||||
|
|
||||||
|
|
||||||
|
def parse_pdf_adjustments(path: Path, target_section: str = "中证1000") -> Tuple[List[dict], List[dict]]:
|
||||||
|
"""解析 PDF 的指定指数 section
|
||||||
|
target_section: '中证1000' / '中证2000' / '中证全指' / '中证能源' 等
|
||||||
|
返 (add_rows, remove_rows)
|
||||||
|
"""
|
||||||
|
with pdfplumber.open(path) as pdf:
|
||||||
|
full_text = "\n".join((p.extract_text() or "") for p in pdf.pages)
|
||||||
|
return _extract_section_records(full_text, target_section)
|
||||||
|
|
||||||
|
|
||||||
|
def parse_content_adjustments(content_html: str, target_section: str) -> Tuple[List[dict], List[dict]]:
|
||||||
|
"""解析 csindex 公告 detail.content (HTML) 嵌入的调整名单表格。
|
||||||
|
|
||||||
|
用途: csindex 早期公告 (如 2009-12-14 nid=1208 "中证行业指数调整名单") 无 PDF/xlsx 附件,
|
||||||
|
调整名单直接以 HTML <table> 嵌入 content 字段。此函数把 HTML 转 text 后复用
|
||||||
|
_extract_section_records 抽 add/remove。
|
||||||
|
|
||||||
|
HTML → text 策略: </tr> → \n (每个 tr 一行) → strip 所有标签 → → space → 压缩 [ \t]+
|
||||||
|
保留 \n 以让 _extract_section_records 按行扫描。
|
||||||
|
"""
|
||||||
|
if not content_html:
|
||||||
|
return [], []
|
||||||
|
try:
|
||||||
|
# csindex 老公告 HTML 无 </tr> 闭标签 (HTML5 隐式闭合), 用 <tr> 开标签作行边界
|
||||||
|
# \x00 marker 防止 raw HTML 里 \n 缩进被压空白时丢行边界
|
||||||
|
text = re.sub(r"(?i)<\s*tr[^>]*>", "\x00", content_html)
|
||||||
|
# 去所有标签
|
||||||
|
text = re.sub(r"<[^>]+>", " ", text)
|
||||||
|
# HTML entity → space
|
||||||
|
text = re.sub(r"(?i) ", " ", text)
|
||||||
|
# 压所有空白 (含 raw HTML \n 缩进) 为单个空格
|
||||||
|
text = re.sub(r"\s+", " ", text)
|
||||||
|
# \x00 marker 恢复为 \n (每个 tr 一行)
|
||||||
|
text = text.replace("\x00", "\n")
|
||||||
|
# 行内首尾空白
|
||||||
|
text = "\n".join(ln.strip() for ln in text.split("\n"))
|
||||||
|
return _extract_section_records(text, target_section)
|
||||||
|
except Exception as e:
|
||||||
|
log.error(f" parse_content_adjustments ERR: {e}")
|
||||||
|
return [], []
|
||||||
|
|
||||||
|
|
||||||
# ======================== 4. 主流程 ========================
|
# ======================== 4. 主流程 ========================
|
||||||
def process_notice(nid: int, detail_cache: Path, file_cache: Path,
|
def process_notice(nid: int, detail_cache: Path, file_cache: Path,
|
||||||
target_index_code: str, target_section: str) -> List[dict]:
|
target_index_code: str, target_section: str) -> List[dict]:
|
||||||
"""处理单个公告 -> 返回 records 列表
|
"""处理单个公告 -> 返回 records 列表
|
||||||
record: {updateDate, index_code, code, code_name, adjust_type, notice_id, source}
|
record: {updateDate, index_code, code, code_name, adjust_type, notice_id, source}
|
||||||
|
|
||||||
|
流程:
|
||||||
|
1. 取 detail
|
||||||
|
2. 优先解析 PDF/xlsx 附件 (extract_file_urls + download_notice_files)
|
||||||
|
3. 附件 records 空时 fallback: 解析 detail.content HTML 嵌入的调整名单表格
|
||||||
|
(解锁 2009-2014 早期公告无附件但 content 内嵌 table 的情况)
|
||||||
"""
|
"""
|
||||||
detail = fetch_detail(nid, detail_cache)
|
detail = fetch_detail(nid, detail_cache)
|
||||||
if not detail or not detail.get("data"):
|
if not detail or not detail.get("data"):
|
||||||
@@ -458,8 +521,7 @@ def process_notice(nid: int, detail_cache: Path, file_cache: Path,
|
|||||||
urls = extract_file_urls(data)
|
urls = extract_file_urls(data)
|
||||||
files = download_notice_files(nid, publish_date, urls, file_cache)
|
files = download_notice_files(nid, publish_date, urls, file_cache)
|
||||||
if not files:
|
if not files:
|
||||||
log.warning(f" {nid} {publish_date}: no files; title={title}")
|
log.info(f" {nid} {publish_date}: no attachments; will try content HTML fallback; title={title}")
|
||||||
return []
|
|
||||||
|
|
||||||
records = []
|
records = []
|
||||||
for f in files:
|
for f in files:
|
||||||
@@ -486,6 +548,30 @@ def process_notice(nid: int, detail_cache: Path, file_cache: Path,
|
|||||||
"code": r["code"], "code_name": r["code_name"],
|
"code": r["code"], "code_name": r["code_name"],
|
||||||
"adjust_type": "remove", "notice_id": nid, "source": f.name,
|
"adjust_type": "remove", "notice_id": nid, "source": f.name,
|
||||||
})
|
})
|
||||||
|
|
||||||
|
# Fallback: 附件空 或 附件解析 0 records 时, 解析 detail.content HTML 嵌入的表格
|
||||||
|
if not records:
|
||||||
|
content_html = data.get("content") or ""
|
||||||
|
try:
|
||||||
|
add, rem = parse_content_adjustments(content_html, target_section)
|
||||||
|
except Exception as e:
|
||||||
|
log.error(f" content fallback {nid} ERR: {e}")
|
||||||
|
add, rem = [], []
|
||||||
|
if add or rem:
|
||||||
|
log.info(f" content HTML fallback hit: add={len(add)}, remove={len(rem)}")
|
||||||
|
for r in add:
|
||||||
|
records.append({
|
||||||
|
"updateDate": publish_date, "index_code": target_index_code,
|
||||||
|
"code": r["code"], "code_name": r["code_name"],
|
||||||
|
"adjust_type": "add", "notice_id": nid, "source": f"content_html_{nid}",
|
||||||
|
})
|
||||||
|
for r in rem:
|
||||||
|
records.append({
|
||||||
|
"updateDate": publish_date, "index_code": target_index_code,
|
||||||
|
"code": r["code"], "code_name": r["code_name"],
|
||||||
|
"adjust_type": "remove", "notice_id": nid, "source": f"content_html_{nid}",
|
||||||
|
})
|
||||||
|
|
||||||
log.info(f" {nid} {publish_date}: parsed add={sum(1 for r in records if r['adjust_type']=='add')}, "
|
log.info(f" {nid} {publish_date}: parsed add={sum(1 for r in records if r['adjust_type']=='add')}, "
|
||||||
f"remove={sum(1 for r in records if r['adjust_type']=='remove')}, title={title}")
|
f"remove={sum(1 for r in records if r['adjust_type']=='remove')}, title={title}")
|
||||||
return records
|
return records
|
||||||
@@ -595,6 +681,16 @@ INDEX_SECTION_MAP: Dict[str, str] = {
|
|||||||
# 000938 不是行业指数 (中证民企ESG 50 等), 留作可选
|
# 000938 不是行业指数 (中证民企ESG 50 等), 留作可选
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# 多指数合并公告 -> 共享给哪些 index_code 的映射
|
||||||
|
# 实证 (2026-07-28): csindex API 的 indexCode 字段对部分 code 绑定不全
|
||||||
|
# 例: 公告 id=1208 "中证行业指数调整名单" (2009-12-14) 涵盖 000928-000937 全部 10 个行业指数,
|
||||||
|
# 但 fetch_notices_by_index("000936") 返回的列表里没 1208 (csindex 后台绑定漏),
|
||||||
|
# 需在此显式补入, 否则 000936 治偏差完全失效。
|
||||||
|
SHARED_NOTICES_BY_INDEX: Dict[int, List[str]] = {
|
||||||
|
1208: ["000928", "000929", "000930", "000931", "000932",
|
||||||
|
"000933", "000934", "000935", "000936", "000937"],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def process_generic_index(
|
def process_generic_index(
|
||||||
idx_code: str,
|
idx_code: str,
|
||||||
@@ -625,6 +721,12 @@ def process_generic_index(
|
|||||||
idx_code, cache_dir / f"notices_{idx_code}.json", force=refresh_index,
|
idx_code, cache_dir / f"notices_{idx_code}.json", force=refresh_index,
|
||||||
)
|
)
|
||||||
adj = filter_adjustment_notices(notices_idx)
|
adj = filter_adjustment_notices(notices_idx)
|
||||||
|
# 补共享的多指数合并公告 (csindex indexCode 字段对部分 code 绑定不全, 见 SHARED_NOTICES_BY_INDEX)
|
||||||
|
existing_ids = {x["id"] for x in adj}
|
||||||
|
for shared_nid, codes in SHARED_NOTICES_BY_INDEX.items():
|
||||||
|
if idx_code in codes and shared_nid not in existing_ids:
|
||||||
|
adj.insert(0, {"id": shared_nid})
|
||||||
|
log.info(f" 补共享公告 nid={shared_nid} (csindex indexCode 未绑定)")
|
||||||
log.info(f" indexCode={idx_code} theme=指数调样 notices: {len(adj)}")
|
log.info(f" indexCode={idx_code} theme=指数调样 notices: {len(adj)}")
|
||||||
|
|
||||||
# 2. 逐公告解析 (PDF/xlsx), max_notices 截断
|
# 2. 逐公告解析 (PDF/xlsx), max_notices 截断
|
||||||
|
|||||||
Reference in New Issue
Block a user