#!/usr/bin/env python3 # -*- coding: utf-8 -*- """ 《卡若的IP财富旅程》PART1-5 重新上传:正文以本地 MD 为准,已入库的图片块按顺序保留(不重新上传)。 - 从 chapters.content 提取现有

(或裸 )块,顺序与 MD 中独立图片行一一对应。 - 无图片的章节:与 batch_upload_ip_book.py 相同(纯文本清洗)。 - 同步 sort_order(与 fix_ip_book_sort_and_images 一致)。 """ from __future__ import annotations import html import importlib.util import re import sys from pathlib import Path ROOT = Path(__file__).resolve().parent.parent BOOK_DIR = Path("/Users/karuo/Documents/个人/2、我写的书/《卡若的IP财富旅程》") PART_ID = "part-new-17755513182" PART_TITLE = "卡若的IP财富旅程" CHAPTERS = { 1: ("chapter-ip-1", "PART1 | 我少年时期的互联网启蒙"), 2: ("chapter-ip-2", "PART2 | 从0-1的过程"), 3: ("chapter-ip-3", "PART3 | 我的创业旅程"), 4: ("chapter-ip-4", "PART4 | 私域流量到AI时代的商业进化"), 5: ("chapter-ip-5", "PART5 | 未来的一些思考"), } FILENAME_RE = re.compile( r"^[✅⬜]?\s*(\d+)\.(\d+)[- ]+(.+?)(?:\.md)?$", re.UNICODE, ) LINE_IMAGE_ONLY = re.compile(r"^\s*!\[([^\]]*)\]\(([^)]+)\)\s*$") def load_db(): mig = ROOT / "scripts" / "migrate_2026_sections.py" spec = importlib.util.spec_from_file_location("_mig_db", mig) mod = importlib.util.module_from_spec(spec) assert spec.loader is not None spec.loader.exec_module(mod) return mod.DB_CONFIG def strip_md_title_line(text: str) -> str: lines = text.splitlines() if lines and lines[0].lstrip().startswith("#"): return "\n".join(lines[1:]).lstrip("\n") return text def for_miniprogram_body_plain(text: str) -> str: out_lines: list[str] = [] for line in text.splitlines(): if line.strip() == "---": out_lines.append("") else: out_lines.append(line) body = "\n".join(out_lines) body = body.replace("**", "") body = re.sub(r"\n{3,}", "\n\n", body) return body.strip() + "\n" def extract_img_blocks_from_db(old_content: str) -> list[str]: """从已入库正文中按文档顺序提取图片块(供复用 URL)。""" if not old_content or "]*>\s*(]+/?>)\s*

", s, re.IGNORECASE | re.DOTALL): blocks.append(m.group(0).strip()) if blocks: return blocks for m in re.finditer(r"]+/?>", s, re.IGNORECASE): blocks.append(f"

{m.group(0).strip()}

") return blocks def body_html_preserve_images(md_path: Path, old_content: str) -> tuple[str, list[str]]: """ 按 fix_ip_book 的 HTML 段落规则生成正文,图片行使用旧库中的块。 返回 (body, warnings) """ warnings: list[str] = [] old_imgs = extract_img_blocks_from_db(old_content) raw = md_path.read_text(encoding="utf-8") lines = strip_md_title_line(raw).splitlines() img_idx = 0 md_img_lines = sum(1 for ln in lines if LINE_IMAGE_ONLY.match(ln)) chunks: list[str] = [] for line in lines: if line.strip() == "---": chunks.append("") continue m = LINE_IMAGE_ONLY.match(line) if m: if img_idx < len(old_imgs): chunks.append(old_imgs[img_idx]) img_idx += 1 else: warnings.append(f"MD 图片行多于库中已存图片 ({md_path.name}),已跳过该行图片") chunks.append("

(图片:库中无对应已上传块,请检查)

") continue stripped = line.replace("**", "").strip() if stripped: chunks.append(f"

{html.escape(stripped)}

") else: chunks.append("") html_parts: list[str] = [] for c in chunks: if c == "": if html_parts and html_parts[-1] != "": html_parts.append("") else: html_parts.append(c) body = "\n".join(html_parts).strip() + "\n" rest = len(old_imgs) - img_idx if rest > 0: warnings.append(f"{md_path.name}: 库中多 {rest} 张图片未在 MD 中引用,已从新正文移除(按 MD 为准)") if md_img_lines and not old_imgs: warnings.append(f"{md_path.name}: MD 含图片行但库中正文无 ,请先跑 fix_ip_book_sort_and_images 上传图片,或使用纯文本上传") return body, warnings def collect_files(part_num: int) -> list[tuple[str, str, Path]]: part_dir = BOOK_DIR / f"PART{part_num}" if not part_dir.exists(): print(f" ⚠ 目录不存在: {part_dir}", file=sys.stderr) return [] seen_sections: dict[str, tuple[str, str, Path]] = {} for f in part_dir.iterdir(): name = f.name clean_name = re.sub(r"\.md(\.md)*$", "", name, flags=re.IGNORECASE) m = FILENAME_RE.match(clean_name) if not m: print(f" ⚠ 跳过无法解析的文件: {name}") continue p_num, s_num, title_raw = m.group(1), m.group(2), m.group(3).strip() if int(p_num) != part_num: print(f" ⚠ PART号不匹配,跳过: {name}") continue ip_id = f"ip{part_num}.{s_num}" section_key = f"{part_num}.{s_num}" title = title_raw if section_key in seen_sections: existing_path = seen_sections[section_key][2] if str(f).lower().endswith(".md") and not str(existing_path).lower().endswith(".md"): seen_sections[section_key] = (ip_id, title, f) else: seen_sections[section_key] = (ip_id, title, f) result = list(seen_sections.values()) result.sort(key=lambda x: float(re.sub(r"ip\d+\.", "", x[0]))) return result def md_has_image_only_lines(md_path: Path) -> bool: raw = md_path.read_text(encoding="utf-8") text = strip_md_title_line(raw) return any(LINE_IMAGE_ONLY.match(ln) for ln in text.splitlines()) def upsert_chapter(cur, chapter_id: str, chapter_title: str, ip_id: str, title: str, body: str, price: float, sort_order: int): cur.execute("SELECT id FROM chapters WHERE id = %s", (ip_id,)) if cur.fetchone(): cur.execute( "UPDATE chapters SET section_title=%s, content=%s, part_id=%s, part_title=%s, " "chapter_id=%s, chapter_title=%s, price=%s, sort_order=%s WHERE id=%s", (title, body, PART_ID, PART_TITLE, chapter_id, chapter_title, price, sort_order, ip_id), ) return "updated" cur.execute( "INSERT INTO chapters (id, section_title, content, part_id, part_title, chapter_id, chapter_title, price, sort_order) " "VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s)", (ip_id, title, body, PART_ID, PART_TITLE, chapter_id, chapter_title, price, sort_order), ) return "created" def main(): try: import pymysql except ImportError: print("需要: pip install pymysql", file=sys.stderr) sys.exit(1) dry_run = "--dry-run" in sys.argv price = 1.0 cfg = load_db() conn = pymysql.connect(**cfg) cur = conn.cursor() total_ok = 0 total_err = 0 all_warnings: list[str] = [] for part_num, (chapter_id, chapter_title) in CHAPTERS.items(): print(f"\n{'='*60}") print(f"PART{part_num} → {chapter_id} | {chapter_title}") print(f"{'='*60}") files = collect_files(part_num) if not files: print(" 无文件,跳过") continue for sort_order, (ip_id, title, path) in enumerate(files, start=1): try: has_imgs = md_has_image_only_lines(path) if has_imgs: cur.execute("SELECT content FROM chapters WHERE id = %s", (ip_id,)) row = cur.fetchone() old_content = row[0] if row and row[0] else "" body, warns = body_html_preserve_images(path, old_content) for w in warns: all_warnings.append(f"{ip_id}: {w}") else: raw = path.read_text(encoding="utf-8") body = for_miniprogram_body_plain(strip_md_title_line(raw)) except Exception as e: print(f" ✗ 处理失败 {ip_id} ({path.name}): {e}") total_err += 1 continue if dry_run: mode = "html+保留图" if has_imgs else "plain" print(f" [dry] {ip_id} sort={sort_order} [{mode}] {title[:36]} ({path.name})") total_ok += 1 continue try: action = upsert_chapter( cur, chapter_id, chapter_title, ip_id, title, body, price, sort_order ) conn.commit() mode = "html" if has_imgs else "plain" print(f" ✓ [{action}] {ip_id} sort={sort_order} [{mode}] {title[:40]}") total_ok += 1 except Exception as e: conn.rollback() print(f" ✗ 上传失败 {ip_id} ({path.name}): {e}") total_err += 1 conn.close() print(f"\n{'='*60}") print(f"完成:成功 {total_ok} | 失败 {total_err}") if all_warnings: print("\n⚠ 提示:") for w in all_warnings[:50]: print(f" - {w}") if len(all_warnings) > 50: print(f" ... 另有 {len(all_warnings) - 50} 条") if dry_run: print("(dry-run 模式,未实际写库)") if __name__ == "__main__": main()