Files
mnote/scripts/wolai_scraper.py
2025-11-23 10:55:04 +08:00

347 lines
12 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Export public Wolai pages listed in scraped_docs/网址.md to Markdown."""
from __future__ import annotations
import json
import os
import re
from collections import deque
from dataclasses import dataclass
from pathlib import Path
from typing import Dict, Iterable, List, Sequence
from urllib.parse import quote
import requests
URL_LIST_PATH = Path("scraped_docs/网址.md")
OUTPUT_DIR = Path("scraped_docs")
COOKIE_ENV = "WOLAI_COOKIE"
def load_page_mapping(path: Path) -> Dict[str, str]:
content = path.read_text(encoding="utf-8").strip()
entries = re.split(r"[;\n]", content)
mapping: Dict[str, str] = {}
for entry in entries:
entry = entry.strip()
if not entry or "" not in entry:
continue
name, url = entry.split("", 1)
name = name.strip()
url = url.strip()
if name and url.startswith("https://"):
mapping[name] = url
return mapping
def sanitize_filename(name: str) -> str:
return re.sub(r"[\\/:\*\?\"<>\|]", "_", name)
def split_batches(items: Iterable[str], size: int) -> Iterable[List[str]]:
batch: List[str] = []
for item in items:
batch.append(item)
if len(batch) >= size:
yield batch
batch = []
if batch:
yield batch
class WolaiClient:
API_URL = "https://api.wolai.com/v1/pages/getData"
def __init__(self, cookie: str) -> None:
headers = {
"wolai-client-platform": "web",
"wolai-app-version": "1.2.3-15",
"wolai-os-platform": "win",
"Origin": "https://www.wolai.com",
"Referer": "https://www.wolai.com/",
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/120.0.0.0 Safari/537.36"
),
"Accept": "application/json, text/plain, */*",
"Cookie": cookie,
}
self.session = requests.Session()
self.session.headers.update(headers)
def fetch_blocks(self, page_id: str) -> Dict[str, dict]:
blocks: Dict[str, dict] = {}
queue: deque[str] = deque([page_id])
seen: set[str] = set()
while queue:
chunk: List[str] = []
while queue and len(chunk) < 50:
bid = queue.popleft()
if bid in seen:
continue
seen.add(bid)
chunk.append(bid)
if not chunk:
continue
payload = {"requests": [{"table": "wolai.block", "id": bid} for bid in chunk]}
resp = self.session.post(self.API_URL, json=payload, timeout=30)
resp.raise_for_status()
for entry in resp.json().get("data", []):
value = entry.get("value")
if not value:
continue
blocks[value["id"]] = value
for child in value.get("sub_nodes") or []:
if child not in seen:
queue.append(child)
return blocks
def get_title_fragments(block: dict) -> List[List]:
attrs = block.get("attributes") or {}
title = attrs.get("title")
if isinstance(title, str):
return [[title]]
return title or []
def rich_text(fragments: Sequence[Sequence]) -> str:
result: List[str] = []
for fragment in fragments:
if not fragment:
continue
text = fragment[0]
marks = fragment[1] if len(fragment) > 1 else None
result.append(apply_marks(text, marks))
return "".join(result).strip()
def apply_marks(text: str, marks: Sequence[Sequence] | None) -> str:
if not marks:
return text
decorated = text
link_target: str | None = None
for mark in marks:
kind = mark[0]
if kind == "B":
decorated = f"**{decorated}**"
elif kind == "I":
decorated = f"*{decorated}*"
elif kind == "S":
decorated = f"~~{decorated}~~"
elif kind == "<>":
decorated = f"`{decorated}`"
elif kind == "Link" and len(mark) > 1:
link_target = mark[1]
elif kind == "BiLink":
# 链接到空间内其他块,保留文本即可
continue
elif kind == "h":
continue
else:
continue
if link_target:
decorated = f"[{decorated}]({link_target})"
return decorated
def build_image_url(block: dict) -> str | None:
attrs = block.get("attributes") or {}
img = attrs.get("img") or []
if not img or not img[0]:
return None
path = img[0][0]
if path.startswith("http"):
return path
safe_path = quote(path, safe="/%")
return f"https://secure2.wostatic.cn/{safe_path}"
def indent_lines(lines: Sequence[str], prefix: str) -> List[str]:
indented: List[str] = []
for line in lines:
if line:
indented.append(f"{prefix}{line}")
else:
indented.append("")
return indented
@dataclass
class MarkdownRenderer:
blocks: Dict[str, dict]
def render_page(self, page_id: str) -> str:
root = self.blocks.get(page_id)
if not root:
return ""
lines = self.render_children(root.get("sub_nodes") or [])
# 移除多余空行并确保结尾有换行符
while lines and not lines[-1].strip():
lines.pop()
return "\n".join(lines) + "\n"
def render_children(self, child_ids: Sequence[str]) -> List[str]:
lines: List[str] = []
i = 0
while i < len(child_ids):
block = self.blocks.get(child_ids[i])
i += 1
if not block:
continue
btype = block["type"]
if btype in {"enumList", "bullList", "todoList", "todoListPro"}:
seq_lines, new_index = self.render_list_sequence(child_ids, i - 1)
lines.extend(seq_lines)
i = new_index
continue
block_lines = self.render_block(block)
if not block_lines:
continue
lines.extend(block_lines)
if block_lines[-1].strip():
lines.append("")
return lines
def render_list_sequence(self, ids: Sequence[str], start: int) -> tuple[List[str], int]:
lines: List[str] = []
i = start
first_type = self.blocks[ids[start]]["type"]
group = {
"enumList": {"enumList"},
"bullList": {"bullList"},
"todoList": {"todoList", "todoListPro"},
"todoListPro": {"todoList", "todoListPro"},
}[first_type]
counter = 1
while i < len(ids):
block = self.blocks.get(ids[i])
if not block or block["type"] not in group:
break
text = rich_text(get_title_fragments(block))
if block["type"] == "enumList":
prefix = f"{counter}. "
elif block["type"] in {"todoList", "todoListPro"}:
checked = block.get("attributes", {}).get("checked") == "yes"
prefix = f"- [{'x' if checked else ' '}] "
else:
prefix = "- "
lines.append(f"{prefix}{text}")
child_lines = self.render_children(block.get("sub_nodes") or [])
if child_lines:
lines.extend(indent_lines(child_lines, " "))
counter += 1
i += 1
lines.append("")
return lines, i
def render_block(self, block: dict) -> List[str]:
btype = block["type"]
text = rich_text(get_title_fragments(block))
if btype == "text":
return [text] if text else []
if btype in {"midHeader", "subHeader", "tinyHeader"}:
level = {"midHeader": "##", "subHeader": "###", "tinyHeader": "####"}[btype]
return [f"{level} {text}"]
if btype == "quote":
quote_lines = [f"> {line}" if line else ">" for line in text.splitlines() or [""]]
child_lines = self.render_children(block.get("sub_nodes") or [])
if child_lines:
quote_lines.extend(f"> {line}" if line else ">" for line in child_lines)
return quote_lines
if btype == "divider":
return ["---"]
if btype == "image":
url = build_image_url(block)
return [f"![]({url})" if url else "![](图片资源需登录查看)"]
if btype == "code":
language = block.get("attributes", {}).get("language") or ""
body = text
fence = f"```{language.lower()}" if language else "```"
return [fence, body, "```"]
if btype in {"row", "column"}:
return self.render_children(block.get("sub_nodes") or [])
if btype == "enumList":
return [f"1. {text}"]
if btype == "bullList":
return [f"- {text}"]
if btype in {"todoList", "todoListPro"}:
checked = block.get("attributes", {}).get("checked") == "yes"
return [f"- [{'x' if checked else ' '}] {text}"]
if btype == "toggleList":
title = text or "折叠列表"
child_lines = indent_lines(self.render_children(block.get("sub_nodes") or []), " ")
return [f"- **{title}**", *child_lines]
if btype in {"toggleSubHeader", "toggleTinyHeader"}:
level = "####" if btype == "toggleSubHeader" else "#####"
title = text + "(可折叠)"
child_lines = self.render_children(block.get("sub_nodes") or [])
return [f"{level} {title}", *child_lines]
if btype == "simpleTable":
return self.render_table(block)
if btype == "progressBar":
progress = block.get("attributes", {}).get("progress", 0)
return [f"进度条:{progress}%"]
if btype == "xiguaVideo":
source = block.get("attributes", {}).get("source") or ""
label = text or "西瓜视频"
return [f"[{label}]({source})" if source else label]
return [text] if text else []
def render_table(self, block: dict) -> List[str]:
attrs = block.get("attributes") or {}
raw = attrs.get("cells")
if not raw:
return []
try:
cells = json.loads(raw)
except json.JSONDecodeError:
return []
rows: List[List[str]] = []
for row in cells:
cols = []
for cell in row.get("column", []):
cols.append(rich_text(cell.get("attributes", {}).get("title") or []))
rows.append(cols)
if not rows:
return []
widths = max(len(r) for r in rows)
for row in rows:
if len(row) < widths:
row.extend([""] * (widths - len(row)))
header = rows[0]
separator = ["---" for _ in header]
lines = ["| " + " | ".join(header) + " |", "| " + " | ".join(separator) + " |"]
for row in rows[1:]:
lines.append("| " + " | ".join(row) + " |")
return lines
def main() -> None:
cookie = os.environ.get(COOKIE_ENV)
if not cookie:
raise SystemExit(
f"请先在环境变量 {COOKIE_ENV} 中配置 wolai Cookie,例如:\n"
f"$env:{COOKIE_ENV}='wolai_client_id=...; token=...'"
)
mapping = load_page_mapping(URL_LIST_PATH)
client = WolaiClient(cookie)
OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
for name, url in mapping.items():
if name == "页面选项":
continue
page_id = url.rstrip("/").split("/")[-1]
print(f"正在导出 {name} ({page_id}) ...")
blocks = client.fetch_blocks(page_id)
renderer = MarkdownRenderer(blocks)
content = renderer.render_page(page_id)
out_path = OUTPUT_DIR / f"{sanitize_filename(name)}.md"
out_path.write_text(content, encoding="utf-8")
print("导出完成。")
if __name__ == "__main__":
main()