"""解析目录页与文章页 HTML,构建结构化文章数据。""" import re from bs4 import BeautifulSoup from converter import blocks_to_markdown from config import DATE_OVERRIDES # 「教你炒股票1:不会赢钱的经济人,只是废人!」 TITLE_RE = re.compile(r"^教你炒股票\s*([0-9]+)\s*[::-\-]\s*(.+)$") DATETIME_RE = re.compile(r"(\d{4}-\d{2}-\d{2})(?:\s+(\d{1,2}:\d{2}))?") _WS_RE = re.compile(r"\s+") # 用于自动打标签的缠论核心概念(多字、不易误判) CONCEPTS = [ "走势中枢", "中枢", "背驰", "买卖点", "分型", "线段", "区间套", "第三类买卖点", "第一类买点", "第二类买点", "走势类型", ] def parse_index(html: str) -> list[int]: """从目录页提取所有文章的 p 编号(升序,去重,去掉 p=0 目录本身)。""" soup = BeautifulSoup(html, "lxml") nums: list[int] = [] for a in soup.select("a.chan[href*='chanlun?p=']"): m = re.search(r"p=(\d+)", a.get("href", "")) if m: n = int(m.group(1)) if n > 0: nums.append(n) return sorted(set(nums)) def _is_nav_paragraph(p) -> bool: """识别正文末尾「返回目录」的导航
,排除掉。""" a = p.find("a", href=True) return bool(a and "chanlun" in a["href"] and "p=0" in a["href"]) def _make_tags(corpus: str) -> list[str]: found = [c for c in CONCEPTS if c in corpus] # 去重保序(「走势中枢」与「中枢」可能同时命中) seen: set[str] = set() unique: list[str] = [] for c in found: if c not in seen: seen.add(c) unique.append(c) return ["教你炒股票"] + unique[:2] def parse_article(html: str, p_num: int, resolve_image) -> dict: """解析单篇文章,返回符合 Vue ``Article`` 结构的字典。""" soup = BeautifulSoup(html, "lxml") container = soup.select_one("div.container-narrow") title_full = "" date = DATE_OVERRIDES.get(p_num) # 源站个别篇目日期残缺,按已知准确值纠偏 elements = [] #
与块级 等,按文档顺序)
if container is not None:
h2 = container.find("h2")
if h2:
title_full = h2.get_text(strip=True)
if date is None:
h5 = container.find("h5")
if h5:
m = DATETIME_RE.search(h5.get_text())
if m:
date = m.group(1) + (f" {m.group(2)}" if m.group(2) else "")
seen_hr = False
for el in container.find_all(recursive=False):
if el.name == "hr":
seen_hr = True
continue
if not seen_hr:
continue # 跳过标题/日期
if el.name == "p" and _is_nav_paragraph(el):
continue # 排除文末「返回目录」
elements.append(el)
# 课次号 / 标题:从「教你炒股票N:标题」中拆分
m = TITLE_RE.match(title_full)
if m:
lesson = int(m.group(1))
topic = m.group(2).strip()
else:
lesson = p_num
topic = title_full or f"教你炒股票 {p_num}"
content = blocks_to_markdown(elements, resolve_image)
# 纯文本度量:阅读时长与摘要(折叠源 HTML 的折行空白)
plain_texts = [_WS_RE.sub(" ", el.get_text(" ", strip=True)) for el in elements]
plain_texts = [t for t in plain_texts if t]
plain_len = sum(len(t) for t in plain_texts)
corpus = "".join(plain_texts)
first_text = plain_texts[0] if plain_texts else ""
excerpt = (first_text[:90] + "…") if len(first_text) > 90 else first_text
reading_time = max(1, round(plain_len / 500)) # 中文技术文 ~500 字/分钟
return {
"id": f"cl-{p_num:03d}",
"slug": f"ke-{p_num:03d}",
"category": "chanlun",
"title": topic,
"subtitle": f"第 {lesson} 课 · 教你炒股票",
"lesson": lesson,
"date": date or "",
"tags": _make_tags(corpus),
"excerpt": excerpt,
"readingTime": reading_time,
"content": content,
}