Files

648 lines
23 KiB
Python
Raw Permalink Normal View History

2026-02-01 08:47:40 +08:00
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
抓取 Wolai 帮助中心(公开页面)并落盘为可检索语料:
- 通过 Wolai API 拉取 blocks(比 Playwright 更稳定、噪声更少)
- 递归抓取子块,并从内容里的链接继续爬取其他页面
- 下载图片到本地,并在 Markdown 中引用本地路径
用法示例(PowerShell):
python scripts/wolai_help_center/export_wolai_help_center.py `
--seed-file scripts/wolai_help_center/seeds.txt `
--out artifacts/wolai-help-center `
--max-pages 200 `
--download-images
"""
from __future__ import annotations
import argparse
import hashlib
import json
import re
import sys
import time
from collections import deque
from dataclasses import dataclass
from pathlib import Path
from typing import Any, Dict, Iterable, List, Optional, Sequence, Set, Tuple
from urllib.parse import quote, urlparse
import requests
API_URL = "https://api.wolai.com/v1/pages/getData"
# Wolai 页面 URL https://www.wolai.com/wolai/<pageId>
WOLAI_PAGE_ID_RE = re.compile(r"(?:https?://www\.wolai\.com)?/wolai/([A-Za-z0-9]+)")
def _read_text(path: Path) -> str:
return path.read_text(encoding="utf-8")
def _write_text(path: Path, content: str) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(content, encoding="utf-8")
def _write_json(path: Path, payload: Any) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
def sanitize_filename(name: str) -> str:
return re.sub(r"[\\/:\*\?\"<>\|]", "_", name).strip() or "untitled"
def parse_page_id(url_or_id: str) -> Optional[str]:
url_or_id = url_or_id.strip()
if not url_or_id:
return None
if "/" not in url_or_id:
return url_or_id
m = WOLAI_PAGE_ID_RE.search(url_or_id)
if not m:
return None
return m.group(1)
def build_image_url(block: dict) -> Optional[str]:
attrs = block.get("attributes") or {}
# 新版/部分图片块使用 source 字段(例如动态图标、截图等)
source = attrs.get("source")
if isinstance(source, list) and source and isinstance(source[0], str) and source[0].strip():
return source[0].strip()
img = attrs.get("img") or []
if not img or not img[0]:
return None
path = img[0][0]
if isinstance(path, str) and path.startswith("http"):
return path
if not isinstance(path, str):
return None
safe_path = quote(path, safe="/%")
return f"https://secure2.wostatic.cn/{safe_path}"
def guess_extension(content_type: str, url: str) -> str:
content_type = (content_type or "").split(";")[0].strip().lower()
if content_type in {"image/png"}:
return ".png"
if content_type in {"image/jpeg", "image/jpg"}:
return ".jpg"
if content_type in {"image/webp"}:
return ".webp"
if content_type in {"image/gif"}:
return ".gif"
if content_type in {"image/svg+xml"}:
return ".svg"
# 兜底:从 URL 后缀猜
path = urlparse(url).path.lower()
for ext in [".png", ".jpg", ".jpeg", ".webp", ".gif", ".svg"]:
if path.endswith(ext):
return ".jpg" if ext == ".jpeg" else ext
return ".bin"
def sha256_hex(text: str) -> str:
return hashlib.sha256(text.encode("utf-8")).hexdigest()
def iter_strings(obj: Any) -> Iterable[str]:
"""深度遍历任意 JSON 结构,返回其中所有字符串。"""
if isinstance(obj, str):
yield obj
elif isinstance(obj, list):
for item in obj:
yield from iter_strings(item)
elif isinstance(obj, dict):
for v in obj.values():
yield from iter_strings(v)
def extract_mark_links_from_title_fragments(fragments: Sequence[Sequence]) -> Tuple[List[str], List[str]]:
"""从富文本片段中提取 Link(url) 与 BiLink(id)。"""
link_urls: List[str] = []
bilink_ids: List[str] = []
for fragment in fragments:
if not fragment:
continue
marks = fragment[1] if len(fragment) > 1 else None
if not marks:
continue
for mark in marks:
if not mark:
continue
kind = mark[0]
if kind == "Link" and len(mark) > 1 and isinstance(mark[1], str):
link_urls.append(mark[1])
elif kind == "BiLink":
# Wolai 的 BiLink 结构常见为:
# ["BiLink", "<blockId>", "<pageId>", ...]
# 这里优先取第三段(pageId),避免把 blockId 当成页面导致爆炸式爬取。
if len(mark) > 2 and isinstance(mark[2], str):
bilink_ids.append(mark[2])
elif len(mark) > 1 and isinstance(mark[1], str):
# 兜底:少数情况下可能只有一个 id
bilink_ids.append(mark[1])
return link_urls, bilink_ids
def extract_page_ids_from_blocks(blocks: Dict[str, dict]) -> Set[str]:
"""从 blocks 中提取可能的 Wolai 页面 ID(用于爬取下一页)。"""
ids: Set[str] = set()
for block in blocks.values():
attrs = block.get("attributes") or {}
title = attrs.get("title") or []
if isinstance(title, str):
title = [[title]]
if isinstance(title, list):
link_urls, bilink_ids = extract_mark_links_from_title_fragments(title)
for url in link_urls:
pid = parse_page_id(url)
if pid:
ids.add(pid)
for bid in bilink_ids:
# 只接受“看起来像 Wolai id”的 BiLink 目标,避免误把其它资源标识加入队列
if re.fullmatch(r"[A-Za-z0-9]{16,32}", bid):
ids.add(bid)
# 更激进的兜底:扫整个 block 里的字符串,找 /wolai/<id>
for s in iter_strings(block):
for m in WOLAI_PAGE_ID_RE.finditer(s):
ids.add(m.group(1))
return ids
class WolaiApiClient:
def __init__(self) -> None:
self.session = requests.Session()
self.session.headers.update(
{
"wolai-client-platform": "web",
"wolai-app-version": "1.2.3-15",
"wolai-os-platform": "win",
"Origin": "https://www.wolai.com",
"Referer": "https://www.wolai.com/",
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/120.0.0.0 Safari/537.36"
),
"Accept": "application/json, text/plain, */*",
}
)
def fetch_blocks_recursive(self, root_id: str) -> Dict[str, dict]:
"""从 root_id 开始,批量拉取并递归展开 sub_nodes。"""
blocks: Dict[str, dict] = {}
queue: deque[str] = deque([root_id])
seen: Set[str] = set()
while queue:
chunk: List[str] = []
while queue and len(chunk) < 50:
bid = queue.popleft()
if bid in seen:
continue
seen.add(bid)
chunk.append(bid)
if not chunk:
continue
payload = {"requests": [{"table": "wolai.block", "id": bid} for bid in chunk]}
resp = self.session.post(API_URL, json=payload, timeout=60)
resp.raise_for_status()
data = resp.json().get("data") or []
for entry in data:
value = entry.get("value")
if not value or not isinstance(value, dict):
continue
bid = value.get("id")
if not isinstance(bid, str):
continue
blocks[bid] = value
for child in value.get("sub_nodes") or []:
if isinstance(child, str) and child not in seen:
queue.append(child)
return blocks
def get_title_fragments(block: dict) -> List[List]:
attrs = block.get("attributes") or {}
title = attrs.get("title")
if isinstance(title, str):
return [[title]]
return title or []
def apply_marks(text: str, marks: Sequence[Sequence] | None) -> str:
if not marks:
return text
decorated = text
link_target: Optional[str] = None
for mark in marks:
if not mark:
continue
kind = mark[0]
if kind == "B":
decorated = f"**{decorated}**"
elif kind == "I":
decorated = f"*{decorated}*"
elif kind == "S":
decorated = f"~~{decorated}~~"
elif kind == "<>":
decorated = f"`{decorated}`"
elif kind == "Link" and len(mark) > 1 and isinstance(mark[1], str):
link_target = mark[1]
elif kind == "BiLink":
# 链接到空间内其他块/页面:这里只影响显示,不强行转成链接,避免生成无效 URL
continue
else:
continue
if link_target:
decorated = f"[{decorated}]({link_target})"
return decorated
def rich_text(fragments: Sequence[Sequence]) -> str:
result: List[str] = []
for fragment in fragments:
if not fragment:
continue
text = fragment[0]
if not isinstance(text, str):
continue
marks = fragment[1] if len(fragment) > 1 else None
result.append(apply_marks(text, marks))
return "".join(result).strip()
def indent_lines(lines: Sequence[str], prefix: str) -> List[str]:
out: List[str] = []
for line in lines:
out.append(f"{prefix}{line}" if line else "")
return out
@dataclass
class MarkdownRenderer:
blocks: Dict[str, dict]
image_url_to_local: Dict[str, str]
def render_page(self, page_id: str) -> str:
root = self.blocks.get(page_id)
if not root:
return ""
lines = self.render_children(root.get("sub_nodes") or [])
while lines and not lines[-1].strip():
lines.pop()
return "\n".join(lines) + "\n"
def render_children(self, child_ids: Sequence[str]) -> List[str]:
lines: List[str] = []
i = 0
while i < len(child_ids):
block = self.blocks.get(child_ids[i])
i += 1
if not block:
continue
btype = block.get("type")
if btype in {"enumList", "bullList", "todoList", "todoListPro"}:
seq_lines, new_index = self.render_list_sequence(child_ids, i - 1)
lines.extend(seq_lines)
i = new_index
continue
block_lines = self.render_block(block)
if not block_lines:
continue
lines.extend(block_lines)
if block_lines[-1].strip():
lines.append("")
return lines
def render_list_sequence(self, ids: Sequence[str], start: int) -> Tuple[List[str], int]:
lines: List[str] = []
i = start
first = self.blocks.get(ids[start]) or {}
first_type = first.get("type")
group = {
"enumList": {"enumList"},
"bullList": {"bullList"},
"todoList": {"todoList", "todoListPro"},
"todoListPro": {"todoList", "todoListPro"},
}.get(first_type, {first_type})
counter = 1
while i < len(ids):
block = self.blocks.get(ids[i])
if not block or block.get("type") not in group:
break
text = rich_text(get_title_fragments(block))
btype = block.get("type")
if btype == "enumList":
prefix = f"{counter}. "
elif btype in {"todoList", "todoListPro"}:
checked = (block.get("attributes") or {}).get("checked") == "yes"
prefix = f"- [{'x' if checked else ' '}] "
else:
prefix = "- "
lines.append(f"{prefix}{text}".rstrip())
child_lines = self.render_children(block.get("sub_nodes") or [])
if child_lines:
lines.extend(indent_lines(child_lines, " "))
counter += 1
i += 1
lines.append("")
return lines, i
def render_block(self, block: dict) -> List[str]:
btype = block.get("type")
text = rich_text(get_title_fragments(block))
if btype == "text":
return [text] if text else []
if btype in {"midHeader", "subHeader", "tinyHeader"}:
level = {"midHeader": "##", "subHeader": "###", "tinyHeader": "####"}[btype]
return [f"{level} {text}".rstrip()]
if btype == "quote":
quote_lines = [f"> {line}" if line else ">" for line in (text.splitlines() or [""])]
child_lines = self.render_children(block.get("sub_nodes") or [])
if child_lines:
quote_lines.extend(f"> {line}" if line else ">" for line in child_lines)
return quote_lines
if btype == "divider":
return ["---"]
if btype == "image":
url = build_image_url(block)
if not url:
return ["![](图片资源)"]
local = self.image_url_to_local.get(url)
return [f"![]({local or url})"]
if btype == "code":
language = (block.get("attributes") or {}).get("language") or ""
body = text
fence = f"```{str(language).lower()}" if language else "```"
return [fence, body, "```"]
if btype in {"row", "column"}:
return self.render_children(block.get("sub_nodes") or [])
if btype == "toggleList":
title = text or "折叠列表"
child_lines = indent_lines(self.render_children(block.get("sub_nodes") or []), " ")
return [f"- **{title}**", *child_lines]
if btype in {"toggleSubHeader", "toggleTinyHeader"}:
level = "####" if btype == "toggleSubHeader" else "#####"
title = (text + "(可折叠)").strip()
child_lines = self.render_children(block.get("sub_nodes") or [])
return [f"{level} {title}", *child_lines]
if btype == "simpleTable":
return self.render_table(block)
if btype == "progressBar":
progress = (block.get("attributes") or {}).get("progress", 0)
return [f"进度条:{progress}%"]
# 兜底
return [text] if text else []
def render_table(self, block: dict) -> List[str]:
attrs = block.get("attributes") or {}
raw = attrs.get("cells")
if not raw:
return []
try:
cells = json.loads(raw)
except json.JSONDecodeError:
return []
rows: List[List[str]] = []
for row in cells:
cols = []
for cell in row.get("column", []):
cols.append(rich_text((cell.get("attributes") or {}).get("title") or []))
rows.append(cols)
if not rows:
return []
width = max(len(r) for r in rows)
for r in rows:
r.extend([""] * (width - len(r)))
header = rows[0]
sep = ["---"] * len(header)
lines = ["| " + " | ".join(header) + " |", "| " + " | ".join(sep) + " |"]
for r in rows[1:]:
lines.append("| " + " | ".join(r) + " |")
return lines
def load_seed_ids(seed_urls: List[str], seed_file: Optional[Path]) -> List[str]:
ids: List[str] = []
for u in seed_urls:
pid = parse_page_id(u)
if pid:
ids.append(pid)
if seed_file and seed_file.exists():
for line in _read_text(seed_file).splitlines():
line = line.strip()
if not line or line.startswith("#"):
continue
pid = parse_page_id(line)
if pid:
ids.append(pid)
# 去重但保持顺序
seen: Set[str] = set()
out: List[str] = []
for pid in ids:
if pid in seen:
continue
seen.add(pid)
out.append(pid)
return out
def download_images(
session: requests.Session,
image_urls: List[str],
out_images_dir: Path,
) -> Dict[str, str]:
"""下载图片并返回 url -> 相对路径 的映射。"""
mapping: Dict[str, str] = {}
out_images_dir.mkdir(parents=True, exist_ok=True)
for url in image_urls:
if url in mapping:
continue
digest = sha256_hex(url)[:24]
# 若已下载过(不同页面复用),直接复用
existing = list(out_images_dir.glob(f"{digest}.*"))
if existing:
mapping[url] = str(Path("images") / existing[0].name).replace("\\", "/")
continue
try:
resp = session.get(url, timeout=60)
resp.raise_for_status()
ext = guess_extension(resp.headers.get("content-type", ""), url)
filename = f"{digest}{ext}"
out_path = out_images_dir / filename
out_path.write_bytes(resp.content)
mapping[url] = str(Path("images") / filename).replace("\\", "/")
except Exception:
# 下载失败也要留痕,避免重复尝试拖慢整体
mapping[url] = url
return mapping
def main(argv: Optional[List[str]] = None) -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--seed-url", action="append", default=[], help="种子页面 URL 或 ID(可重复指定)")
parser.add_argument("--seed-file", type=str, default="", help="种子列表文件(每行一个 URL 或 ID)")
parser.add_argument("--out", type=str, default="artifacts/wolai-help-center", help="输出目录")
parser.add_argument("--max-pages", type=int, default=200, help="最大抓取页面数(防止无限扩张)")
parser.add_argument("--download-images", action="store_true", help="下载图片到本地并在 Markdown 中引用")
parser.add_argument("--sleep-ms", type=int, default=0, help="每页抓取后的延迟(毫秒)")
parser.add_argument("--resume", action="store_true", help="从 state.json 恢复")
args = parser.parse_args(argv)
out_dir = Path(args.out)
pages_dir = out_dir / "pages"
images_dir = out_dir / "images"
state_path = out_dir / "state.json"
index_path = out_dir / "index.json"
seed_file = Path(args.seed_file) if args.seed_file else None
seeds = load_seed_ids(args.seed_url, seed_file)
if not seeds:
print("未提供 seed--seed-url 或 --seed-file)。", file=sys.stderr)
return 2
api = WolaiApiClient()
allowed_page_id: Optional[str] = None
queue: deque[str] = deque(seeds)
seen: Set[str] = set()
results: List[dict] = []
failures: List[dict] = []
if args.resume and state_path.exists():
try:
state = json.loads(_read_text(state_path))
allowed_page_id = state.get("allowed_page_id") or None
queue = deque(state.get("queue") or [])
seen = set(state.get("seen") or [])
results = state.get("results") or []
failures = state.get("failures") or []
except Exception:
pass
exported = 0
while queue and exported < args.max_pages:
page_id = queue.popleft()
if page_id in seen:
continue
seen.add(page_id)
try:
blocks = api.fetch_blocks_recursive(page_id)
root = blocks.get(page_id)
if not root:
raise RuntimeError("root block 缺失")
if root.get("type") != "page":
# 队列里可能混入 blockId;仅导出“页面”类型,避免爆炸式爬取
continue
if allowed_page_id is None:
allowed_page_id = root.get("page_id") if isinstance(root.get("page_id"), str) else None
if allowed_page_id and root.get("page_id") != allowed_page_id:
# 只抓同一个帮助中心空间,避免爬到用户分享页/其它空间
continue
title = rich_text(get_title_fragments(root)) or page_id
safe_title = sanitize_filename(title)
md_name = f"{page_id}__{safe_title}.md"
json_name = f"{page_id}__{safe_title}.json"
page_json_path = pages_dir / json_name
page_md_path = pages_dir / md_name
image_urls: List[str] = []
if args.download_images:
for b in blocks.values():
if b.get("type") == "image":
url = build_image_url(b)
if url:
image_urls.append(url)
image_map: Dict[str, str] = {}
if args.download_images and image_urls:
image_map = download_images(api.session, image_urls, images_dir)
renderer = MarkdownRenderer(blocks=blocks, image_url_to_local=image_map)
body = renderer.render_page(page_id)
header = {
"source": f"https://www.wolai.com/wolai/{page_id}",
"title": title,
"pageId": page_id,
"exportedAt": int(time.time() * 1000),
"imageCount": len(image_urls),
}
md = "---\n" + json.dumps(header, ensure_ascii=False, indent=2) + "\n---\n\n" + body
_write_text(page_md_path, md)
# blocks 数据落盘(便于后续结构化分析)
_write_json(
page_json_path,
{
"meta": header,
"blocks": blocks,
"images": [{"url": u, "local": image_map.get(u, u)} for u in image_urls],
},
)
results.append(
{
"pageId": page_id,
"title": title,
"source": header["source"],
"md": str(page_md_path.relative_to(out_dir)).replace("\\", "/"),
"json": str(page_json_path.relative_to(out_dir)).replace("\\", "/"),
"imageCount": len(image_urls),
}
)
# 继续爬取下一层链接
next_ids = extract_page_ids_from_blocks(blocks)
for nid in sorted(next_ids):
if nid not in seen:
queue.append(nid)
exported += 1
if args.sleep_ms > 0:
time.sleep(args.sleep_ms / 1000)
except Exception as e:
failures.append({"pageId": page_id, "error": str(e)})
# 持久化状态,避免中断重来
_write_json(
state_path,
{
"allowed_page_id": allowed_page_id,
"queue": list(queue),
"seen": sorted(seen),
"results": results,
"failures": failures,
},
)
_write_json(index_path, {"results": results, "failures": failures, "allowed_page_id": allowed_page_id})
print(f"已导出页面:{exported},输出目录:{out_dir}")
print(f"索引:{index_path}")
if failures:
print(f"失败:{len(failures)}(见 {state_path} / {index_path}")
return 0
if __name__ == "__main__":
raise SystemExit(main())