#!/usr/bin/env python3 """Export public Wolai pages listed in scraped_docs/网址.md to Markdown.""" from __future__ import annotations import json import os import re from collections import deque from dataclasses import dataclass from pathlib import Path from typing import Dict, Iterable, List, Sequence from urllib.parse import quote import requests URL_LIST_PATH = Path("scraped_docs/网址.md") OUTPUT_DIR = Path("scraped_docs") COOKIE_ENV = "WOLAI_COOKIE" def load_page_mapping(path: Path) -> Dict[str, str]: content = path.read_text(encoding="utf-8").strip() entries = re.split(r"[;;\n]", content) mapping: Dict[str, str] = {} for entry in entries: entry = entry.strip() if not entry or ":" not in entry: continue name, url = entry.split(":", 1) name = name.strip() url = url.strip() if name and url.startswith("https://"): mapping[name] = url return mapping def sanitize_filename(name: str) -> str: return re.sub(r"[\\/:\*\?\"<>\|]", "_", name) def split_batches(items: Iterable[str], size: int) -> Iterable[List[str]]: batch: List[str] = [] for item in items: batch.append(item) if len(batch) >= size: yield batch batch = [] if batch: yield batch class WolaiClient: API_URL = "https://api.wolai.com/v1/pages/getData" def __init__(self, cookie: str) -> None: headers = { "wolai-client-platform": "web", "wolai-app-version": "1.2.3-15", "wolai-os-platform": "win", "Origin": "https://www.wolai.com", "Referer": "https://www.wolai.com/", "User-Agent": ( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " "AppleWebKit/537.36 (KHTML, like Gecko) " "Chrome/120.0.0.0 Safari/537.36" ), "Accept": "application/json, text/plain, */*", "Cookie": cookie, } self.session = requests.Session() self.session.headers.update(headers) def fetch_blocks(self, page_id: str) -> Dict[str, dict]: blocks: Dict[str, dict] = {} queue: deque[str] = deque([page_id]) seen: set[str] = set() while queue: chunk: List[str] = [] while queue and len(chunk) < 50: bid = queue.popleft() if bid in seen: continue seen.add(bid) chunk.append(bid) if not chunk: continue payload = {"requests": [{"table": "wolai.block", "id": bid} for bid in chunk]} resp = self.session.post(self.API_URL, json=payload, timeout=30) resp.raise_for_status() for entry in resp.json().get("data", []): value = entry.get("value") if not value: continue blocks[value["id"]] = value for child in value.get("sub_nodes") or []: if child not in seen: queue.append(child) return blocks def get_title_fragments(block: dict) -> List[List]: attrs = block.get("attributes") or {} title = attrs.get("title") if isinstance(title, str): return [[title]] return title or [] def rich_text(fragments: Sequence[Sequence]) -> str: result: List[str] = [] for fragment in fragments: if not fragment: continue text = fragment[0] marks = fragment[1] if len(fragment) > 1 else None result.append(apply_marks(text, marks)) return "".join(result).strip() def apply_marks(text: str, marks: Sequence[Sequence] | None) -> str: if not marks: return text decorated = text link_target: str | None = None for mark in marks: kind = mark[0] if kind == "B": decorated = f"**{decorated}**" elif kind == "I": decorated = f"*{decorated}*" elif kind == "S": decorated = f"~~{decorated}~~" elif kind == "<>": decorated = f"`{decorated}`" elif kind == "Link" and len(mark) > 1: link_target = mark[1] elif kind == "BiLink": # 链接到空间内其他块,保留文本即可 continue elif kind == "h": continue else: continue if link_target: decorated = f"[{decorated}]({link_target})" return decorated def build_image_url(block: dict) -> str | None: attrs = block.get("attributes") or {} img = attrs.get("img") or [] if not img or not img[0]: return None path = img[0][0] if path.startswith("http"): return path safe_path = quote(path, safe="/%") return f"https://secure2.wostatic.cn/{safe_path}" def indent_lines(lines: Sequence[str], prefix: str) -> List[str]: indented: List[str] = [] for line in lines: if line: indented.append(f"{prefix}{line}") else: indented.append("") return indented @dataclass class MarkdownRenderer: blocks: Dict[str, dict] def render_page(self, page_id: str) -> str: root = self.blocks.get(page_id) if not root: return "" lines = self.render_children(root.get("sub_nodes") or []) # 移除多余空行并确保结尾有换行符 while lines and not lines[-1].strip(): lines.pop() return "\n".join(lines) + "\n" def render_children(self, child_ids: Sequence[str]) -> List[str]: lines: List[str] = [] i = 0 while i < len(child_ids): block = self.blocks.get(child_ids[i]) i += 1 if not block: continue btype = block["type"] if btype in {"enumList", "bullList", "todoList", "todoListPro"}: seq_lines, new_index = self.render_list_sequence(child_ids, i - 1) lines.extend(seq_lines) i = new_index continue block_lines = self.render_block(block) if not block_lines: continue lines.extend(block_lines) if block_lines[-1].strip(): lines.append("") return lines def render_list_sequence(self, ids: Sequence[str], start: int) -> tuple[List[str], int]: lines: List[str] = [] i = start first_type = self.blocks[ids[start]]["type"] group = { "enumList": {"enumList"}, "bullList": {"bullList"}, "todoList": {"todoList", "todoListPro"}, "todoListPro": {"todoList", "todoListPro"}, }[first_type] counter = 1 while i < len(ids): block = self.blocks.get(ids[i]) if not block or block["type"] not in group: break text = rich_text(get_title_fragments(block)) if block["type"] == "enumList": prefix = f"{counter}. " elif block["type"] in {"todoList", "todoListPro"}: checked = block.get("attributes", {}).get("checked") == "yes" prefix = f"- [{'x' if checked else ' '}] " else: prefix = "- " lines.append(f"{prefix}{text}") child_lines = self.render_children(block.get("sub_nodes") or []) if child_lines: lines.extend(indent_lines(child_lines, " ")) counter += 1 i += 1 lines.append("") return lines, i def render_block(self, block: dict) -> List[str]: btype = block["type"] text = rich_text(get_title_fragments(block)) if btype == "text": return [text] if text else [] if btype in {"midHeader", "subHeader", "tinyHeader"}: level = {"midHeader": "##", "subHeader": "###", "tinyHeader": "####"}[btype] return [f"{level} {text}"] if btype == "quote": quote_lines = [f"> {line}" if line else ">" for line in text.splitlines() or [""]] child_lines = self.render_children(block.get("sub_nodes") or []) if child_lines: quote_lines.extend(f"> {line}" if line else ">" for line in child_lines) return quote_lines if btype == "divider": return ["---"] if btype == "image": url = build_image_url(block) return [f"![]({url})" if url else "![](图片资源需登录查看)"] if btype == "code": language = block.get("attributes", {}).get("language") or "" body = text fence = f"```{language.lower()}" if language else "```" return [fence, body, "```"] if btype in {"row", "column"}: return self.render_children(block.get("sub_nodes") or []) if btype == "enumList": return [f"1. {text}"] if btype == "bullList": return [f"- {text}"] if btype in {"todoList", "todoListPro"}: checked = block.get("attributes", {}).get("checked") == "yes" return [f"- [{'x' if checked else ' '}] {text}"] if btype == "toggleList": title = text or "折叠列表" child_lines = indent_lines(self.render_children(block.get("sub_nodes") or []), " ") return [f"- **{title}**", *child_lines] if btype in {"toggleSubHeader", "toggleTinyHeader"}: level = "####" if btype == "toggleSubHeader" else "#####" title = text + "(可折叠)" child_lines = self.render_children(block.get("sub_nodes") or []) return [f"{level} {title}", *child_lines] if btype == "simpleTable": return self.render_table(block) if btype == "progressBar": progress = block.get("attributes", {}).get("progress", 0) return [f"进度条:{progress}%"] if btype == "xiguaVideo": source = block.get("attributes", {}).get("source") or "" label = text or "西瓜视频" return [f"[{label}]({source})" if source else label] return [text] if text else [] def render_table(self, block: dict) -> List[str]: attrs = block.get("attributes") or {} raw = attrs.get("cells") if not raw: return [] try: cells = json.loads(raw) except json.JSONDecodeError: return [] rows: List[List[str]] = [] for row in cells: cols = [] for cell in row.get("column", []): cols.append(rich_text(cell.get("attributes", {}).get("title") or [])) rows.append(cols) if not rows: return [] widths = max(len(r) for r in rows) for row in rows: if len(row) < widths: row.extend([""] * (widths - len(row))) header = rows[0] separator = ["---" for _ in header] lines = ["| " + " | ".join(header) + " |", "| " + " | ".join(separator) + " |"] for row in rows[1:]: lines.append("| " + " | ".join(row) + " |") return lines def main() -> None: cookie = os.environ.get(COOKIE_ENV) if not cookie: raise SystemExit( f"请先在环境变量 {COOKIE_ENV} 中配置 wolai Cookie,例如:\n" f"$env:{COOKIE_ENV}='wolai_client_id=...; token=...'" ) mapping = load_page_mapping(URL_LIST_PATH) client = WolaiClient(cookie) OUTPUT_DIR.mkdir(parents=True, exist_ok=True) for name, url in mapping.items(): if name == "页面选项": continue page_id = url.rstrip("/").split("/")[-1] print(f"正在导出 {name} ({page_id}) ...") blocks = client.fetch_blocks(page_id) renderer = MarkdownRenderer(blocks) content = renderer.render_page(page_id) out_path = OUTPUT_DIR / f"{sanitize_filename(name)}.md" out_path.write_text(content, encoding="utf-8") print("导出完成。") if __name__ == "__main__": main()