"""命令行入口:`python -m tools.downloader `。 设计原则(见 README.md): * **只落 staging**(`_out/`),不碰 `wallpapers/`——promote 是后续单独一步。 * **一页 = 全部场景**:每个有内容的场景各产出一个「可直接搬走的壁纸目录」 (`_out/<游戏>/<页面>/<场景id>/`),不再只落被选中的那一个。形状与校验见 `layout.py`。 * **可重跑**:文本走 `_cache/`,产物存在就跳过;`--offline` 下不发起任何请求。 * **抓完就自检**:`fetch` 结尾自动跑 `verify`,有硬伤就非 0 退出。 """ from __future__ import annotations import argparse import datetime as dt import json import shutil import sys from dataclasses import dataclass, field from pathlib import Path from typing import Any import yaml from . import layout as layout_mod from . import promote as promote_mod from . import scene as scene_mod from . import selection as selection_mod from . import verify as verify_mod from .sites import mihoyo HERE = Path(__file__).resolve().parent ROOT = HERE.parents[1] SOURCES = ROOT / "wallpapers" / "sources.yml" CACHE = HERE / "_cache" OUT = HERE / "_out" SELECTION = HERE / "selection.yml" # 旧形状(一页只落一个场景)留在页面根上的东西。新形状里它们是页面根的污染: # `scene.json` 归到每个场景目录里、`spine/` 改名 `spines/` 并下沉到场景目录。 _LEGACY_ENTRIES = ("scene.json", "spine", "scene") # "看着像贴图平面、其实没有独立文件"的 modifier: # cacheContainer —— 渲染进贴图缓冲 # drawScene —— 把**另一个场景**渲染成纹理(diffuse 名就是场景 id,如各页的 scene_ui) _RUNTIME_MODIFIERS = ("cacheContainer", "drawScene") @dataclass class PageEntry: game: str id: str name: str url: str scene: str | None = None spines: list[str] = field(default_factory=list) class SourcesError(RuntimeError): """`sources.yml` 形状不对。""" def load_sources(path: Path = SOURCES) -> list[PageEntry]: """读 `sources.yml`:``<游戏>: [{id, name, url, scene?, spines?}, …]``。""" if not path.exists(): raise SourcesError(f"找不到来源清单:{path}") raw = yaml.safe_load(path.read_text(encoding="utf-8")) if not isinstance(raw, dict): raise SourcesError(f"{path} 顶层必须是「游戏 → 页面列表」的映射") entries: list[PageEntry] = [] for game, pages in raw.items(): if not isinstance(pages, list): raise SourcesError(f"{path} 里 {game} 必须是列表") for i, page in enumerate(pages): where = f"{path} 的 {game}[{i}]" if not isinstance(page, dict): raise SourcesError(f"{where} 不是映射(YAML 里少写了一个 `- `?)") missing = [k for k in ("id", "name", "url") if not page.get(k)] if missing: raise SourcesError(f"{where} 缺少字段:{', '.join(missing)}") spines = page.get("spines") or [] if not isinstance(spines, list): raise SourcesError(f"{where} 的 spines 必须是列表") entries.append( PageEntry( game=str(game), id=str(page["id"]), name=str(page["name"]), url=str(page["url"]), scene=str(page["scene"]) if page.get("scene") else None, spines=[str(s) for s in spines], ) ) return entries def _download(url: str, dest: Path, *, force: bool) -> int: """下载一个资源到 dest(已存在则跳过)。返回落盘字节数。""" if dest.exists() and not force: return 0 dest.parent.mkdir(parents=True, exist_ok=True) data = mihoyo.http_get(url, None) dest.write_bytes(data) return len(data) def _write_spine(scene_dir: Path, entry: PageEntry, data: mihoyo.PageData, name: str, *, force: bool, offline: bool, report: list[str]) -> int: """落一具骨架到 `<场景目录>/spines/<名>/`:json + atlas + 贴图页 + meta.json。 页与 atlas **同居**(布局 A)——spine-ts 按 `/<页名>` 解析页,页跑到别处就读不到。 """ asset = data.spines.get(name) spine_dir = scene_dir / layout_mod.SPINE_DIR / name if asset is None: report.append(f"场景引用了骨架 {name},但页面里没有它的数据(已跳过)") return 0 try: payload = mihoyo.load_spine_json(data, asset, cache_dir=CACHE, offline=offline) except mihoyo.HttpError as exc: # 离线缺缓存 / 网络失败:报出来,别让一条 traceback 把整页的抓取带崩。 report.append(f"骨架 {name} 的数据取不到:{exc}") return 0 original_images = (payload.get("skeleton") or {}).get("images") normalized = dict(payload) skeleton = dict(normalized.get("skeleton") or {}) # 页名按 atlas 所在目录解析:作者目录("../images/" 之类)搬进分发后一定指错。 skeleton["images"] = "" normalized["skeleton"] = skeleton spine_dir.mkdir(parents=True, exist_ok=True) written = 0 json_file = spine_dir / f"{name}.json" if force or not json_file.exists(): json_file.write_text(json.dumps(normalized, ensure_ascii=False, separators=(",", ":")), encoding="utf-8") written += json_file.stat().st_size atlas_file = spine_dir / f"{name}.atlas" if force or not atlas_file.exists(): atlas_file.write_text(asset.atlas_text, encoding="utf-8") written += atlas_file.stat().st_size # 页名读 atlas 自己声明的页(`asset.pages` 就是 atlas 解析出来的),不按 `_N` 猜。 for page_name in asset.pages: stem = page_name.rsplit(".", 1)[0] rel = data.images.get(stem) if rel is None: report.append(f"骨架 {name} 的贴图页 {page_name} 在资源表里找不到 URL") continue try: written += _download(data.asset_url(rel), spine_dir / page_name, force=force) except mihoyo.HttpError as exc: report.append(f"骨架 {name} 的贴图页 {page_name} 下载失败:{exc}") meta = asset.as_meta(entry.url) meta["originalImages"] = original_images meta["fetchedAt"] = dt.datetime.now(dt.timezone.utc).isoformat(timespec="seconds") (spine_dir / "meta.json").write_text(json.dumps(meta, ensure_ascii=False, indent=2), encoding="utf-8") return written def _write_images(scene_dir: Path, data: mihoyo.PageData, parts: list[Any], *, runtime_names: set[str], force: bool, report: list[str], notes: list[str]) -> tuple[int, set[str]]: """落几何平面用的场景图到 `<场景目录>/scene/`。 没有 URL 的平面**一律算运行时纹理**(不下载、不进预设,在 `scene.json` 里标 `runtime`): 资源表就是页面向网络索取资源的完整清单,名字不在表里说明页面自己也不去网上取它。 四种来源:`drawScene` 的场景渲染目标、`cacheContainer` 的贴图缓冲、diffuse 指向同场景骨架 缓存,以及运行时生成的纹理。其中只有最后一种会记一条 note——前三种是页面的正常结构。 """ written = 0 runtime: set[str] = set() for part in parts: if part.kind != "image": continue image_id = part.id rel = data.images.get(image_id) if rel is None: runtime.add(image_id) if not any(m in part.modifiers for m in _RUNTIME_MODIFIERS) and image_id not in runtime_names: notes.append(f"[{scene_dir.name}] 平面 {image_id} 在资源表里没有 URL,按运行时纹理处理") continue ext = mihoyo.image_ext(rel) try: written += _download( data.asset_url(rel), scene_dir / layout_mod.SCENE_DIR / f"{image_id}{ext}", force=force ) except mihoyo.HttpError as exc: report.append(f"几何平面 {image_id} 下载失败:{exc}") return written, runtime def _scene_dir_names(scenes: list[scene_mod.Scene]) -> dict[str, str]: """场景 id → 目录名;同时消掉转义后可能出现的重名(大小写不敏感)。""" used: set[str] = set() out: dict[str, str] = {} for scene in scenes: base = layout_mod.scene_dir_name(scene.id) name, index = base, 2 while name.lower() in used: name = f"{base}_{index}" index += 1 used.add(name.lower()) out[scene.id] = name return out def _scene_has_content(scene: scene_mod.Scene) -> bool: """这个场景有没有**内容**(骨架或贴图平面)。 只有纯色平面的场景(back-moon 的 `动画预览`)与完全没有 part 的场景(`effect_DofBlur`) 不算——它们连一张图都不需要,落出来的目录必然是空的。 注意"有内容"不等于"能搬走":各页的 `scene_ui` 全是 `drawScene` 渲染目标,一落地就是 没有 part 的目录(`usable: false`),它进 `_out` 只是为了让"页面上有几个场景"这件事 在产物里是完整的。 """ return any(p.kind in ("spine", "image") for p in scene.parts) def _write_scene(page_dir: Path, entry: PageEntry, data: mihoyo.PageData, scene: scene_mod.Scene, *, dir_name: str, force: bool, offline: bool, report: list[str], notes: list[str], fetched_at: str) -> dict[str, Any]: """把一个场景落成「可直接搬走的壁纸目录」。返回它的摘要(写进 page.json)。""" scene_dir = page_dir / dir_name (scene_dir / layout_mod.AUDIO_DIR).mkdir(parents=True, exist_ok=True) before = len(report) written = 0 kept_spines = scene.spine_ids # 场景内的全部骨架,按出现序去重(同一骨架被引用多次只存一份) for name in kept_spines: written += _write_spine(scene_dir, entry, data, name, force=force, offline=offline, report=report) image_bytes, runtime_images = _write_images( scene_dir, data, scene.parts, runtime_names=set(kept_spines), force=force, report=report, notes=notes, ) written += image_bytes payload = scene.as_dict(page=entry.id, game=entry.game) payload["parts"] = [] for part in scene.parts: item = part.as_dict() if part.kind == "image" and part.id in runtime_images: item["runtime"] = True # 由骨架 / 别的场景渲染出来,没有独立文件 payload["parts"].append(item) written += layout_mod.write_json(scene_dir / "scene.json", payload) preset, preset_notes = layout_mod.build_preset(payload, scene_dir) notes.extend(f"[{dir_name}] {note}" for note in preset_notes) for rel in layout_mod.missing_preset_paths(preset, scene_dir): report.append(f"[{dir_name}] preset.template.json 引用的 {rel} 不在磁盘上") written += layout_mod.write_json(scene_dir / "preset.template.json", preset) # 该落盘却没落进预设 = 硬伤(上面的 report 已经写了原因),别让目录看起来是好的。 planned = [p for p in scene.parts if p.kind == "spine" or (p.kind == "image" and p.id not in runtime_images)] usable = bool(preset["sceneConfig"]["parts"]) reason = "" if not usable: reason = "场景里只有运行时纹理(drawScene / 贴图缓冲),没有可搬走的资源" if not usable and planned: report.append(f"[{dir_name}] 有 {len(planned)} 件该落盘的 part 却没进预设(见上面的缺失报告)") meta = layout_mod.build_meta( wallpaper_id=dir_name, name=f"{entry.name}({scene.id})", title=f"{entry.name}({scene.id})", description=f"{entry.name} · 场景 {scene.id}\n来源:{entry.url}", game=entry.game, page=entry.id, scene=scene.id, source=entry.url, fetched_at=fetched_at, ) meta["usable"] = usable if reason: meta["reason"] = reason written += layout_mod.write_json(scene_dir / "meta.json", meta) return { "dir": dir_name, "scene": scene.id, "usable": usable, "reason": reason or None, "spines": len(kept_spines), "images": len({p.id for p in scene.parts if p.kind == "image"}), "runtimeImages": sorted(runtime_images), "solids": sum(1 for p in scene.parts if p.kind == "solid"), "parts": len(preset["sceneConfig"]["parts"]), "bytes": written, "problems": len(report) - before, } def cmd_fetch(args: argparse.Namespace) -> int: entries = load_sources() if args.page: wanted = set(args.page) entries = [e for e in entries if e.id in wanted] missing = wanted - {e.id for e in entries} if missing: print(f"来源清单里没有这些页面:{', '.join(sorted(missing))}", file=sys.stderr) return 2 if not entries: print("没有要抓的页面。", file=sys.stderr) return 2 selection = selection_mod.load_selection(SELECTION) problems: list[str] = [] total_bytes = 0 for entry in entries: print(f"\n=== {entry.game}/{entry.id} {entry.name}") def fail(message: str, _id: str = entry.id) -> None: """抓取期的硬伤当场打印——攒到最后再报会让人以为"这页没东西"。""" problems.append(f"{_id}:{message}") print(f" ✗ {message}") try: data = mihoyo.fetch_page(entry.id, entry.game, entry.url, cache_dir=CACHE, offline=args.offline) except mihoyo.HttpError as exc: fail(f"抓取失败 {exc}") continue scenes = data.scenes if not scenes: fail("页面里一个场景都没有") continue # 默认场景仍然按"骨架最多"算,但它现在只用于「哪一档是推荐的」—— # 全部有内容的场景都会落盘,选择不再裁剪产物(选择记录也只剩这个用途)。 fallback = scene_mod.pick_default_scene(scenes) choice = selection_mod.resolve( entry.id, scenes, sources_entry={"scene": entry.scene, "spines": entry.spines}, selection=selection, default_scene=fallback.id if fallback else None, default_spines=fallback.spine_ids if fallback else [], interactive=args.interactive, ) chosen = choice.scene or (fallback.id if fallback else None) page_dir = OUT / entry.game / entry.id page_dir.mkdir(parents=True, exist_ok=True) for legacy in _LEGACY_ENTRIES: stale = page_dir / legacy if not stale.exists(): continue # 旧形状的残留:留着会被 verify 当成形状不对的场景目录。 if stale.is_dir(): shutil.rmtree(stale, ignore_errors=True) else: stale.unlink() print(f" · 清掉旧形状的 {legacy}") names = _scene_dir_names(scenes) wanted_scenes = [s for s in scenes if _scene_has_content(s)] wanted_ids = {s.id for s in wanted_scenes} skipped = [ {"scene": s.id, "reason": "只有纯色平面或完全没有 part,不需要任何资源"} for s in scenes if s.id not in wanted_ids ] if not wanted_scenes: fail("所有场景都不需要资源(没有骨架、也没有贴图平面)") continue print(f" 场景 {len(wanted_scenes)}/{len(scenes)} 个有内容,逐个落盘:") fetched_at = dt.datetime.now(dt.timezone.utc).isoformat(timespec="seconds") summaries: list[dict[str, Any]] = [] notes: list[str] = [] for scene in wanted_scenes: before = len(problems) summary = _write_scene( page_dir, entry, data, scene, dir_name=names[scene.id], force=args.force, offline=args.offline, report=problems, notes=notes, fetched_at=fetched_at, ) summaries.append(summary) total_bytes += int(summary["bytes"]) for message in problems[before:]: print(f" ✗ {message}") mark = "✓" if summary["usable"] else "○" print( f" {mark} {summary['dir']}/(原 id {summary['scene']}):骨架 {summary['spines']}、" f"贴图平面 {summary['images']}、纯色 {summary['solids']}、预设 part {summary['parts']}," f"{int(summary['bytes']) / 1024:.1f} KB" + (f"(不可搬走:{summary['reason']})" if not summary["usable"] else "") ) for note in notes: print(f" · {note}") # 推荐的场景要真的落了盘——推荐到一个被跳过的场景会让 promote 的默认值指向空气。 usable_dirs = {s["dir"] for s in summaries if s["usable"]} chosen_dir = names.get(chosen, "") if chosen else "" if chosen_dir not in usable_dirs: chosen = next((s["scene"] for s in summaries if s["usable"]), None) chosen_dir = names.get(chosen, "") if chosen else "" page_payload = { "id": entry.id, "game": entry.game, "name": entry.name, "url": entry.url, "site": data.site.id, "entry": data.entry_url, "bundles": sorted(data.bundles), "scenes": [ { "id": s.id, "dir": names.get(s.id) if s.id in wanted_ids else None, "spines": len(s.spine_ids), "images": len({p.id for p in s.parts if p.kind == "image"}), "solids": sum(1 for p in s.parts if p.kind == "solid"), } for s in scenes ], "chosenScene": chosen, "chosenDir": chosen_dir or None, "sceneDirs": summaries, "skippedScenes": skipped, "totalBytes": sum(int(s["bytes"]) for s in summaries), "warnings": data.warnings + [w for w in problems if w.startswith(entry.id)] + notes, "fetchedAt": fetched_at, } layout_mod.write_json(page_dir / "page.json", page_payload) for warning in data.warnings: print(f" ! {warning}") usable_count = sum(1 for s in summaries if s["usable"]) print( f" 页面合计 {page_payload['totalBytes'] / 1024 / 1024:.2f} MB → " f"{len(summaries)} 个场景目录(其中 {usable_count} 个可直接搬走)" ) selection[entry.id] = choice.as_dict() selection_mod.save_selection(SELECTION, selection) print(f"\n合计新增 {total_bytes / 1024 / 1024:.2f} MB;选择记录 → {SELECTION.relative_to(ROOT)}") print("\n=== verify") found, stats = verify_mod.verify_all(OUT, pages=[e.id for e in entries]) for problem in found: print(f" ✗ {problem}") print( f" 页面 {stats['pages']}、场景目录 {stats['sceneDirs']}、骨架 {stats['spines']}、" f"几何平面 {stats['images']}、体积 {stats['bytes'] / 1024 / 1024:.2f} MB" ) if stats["versions"]: versions = "、".join(f"{v}×{n}" for v, n in sorted(stats["versions"].items())) print(f" 骨架版本:{versions}") if found: print(f"\n{len(found)} 处硬伤,未通过。", file=sys.stderr) return 1 return 0 def cmd_select(args: argparse.Namespace) -> int: """只记录「推荐哪个场景」——产物不再按选择裁剪(fetch 落全部有内容的场景)。""" entries = load_sources() if args.page: entries = [e for e in entries if e.id in set(args.page)] selection = selection_mod.load_selection(SELECTION) for entry in entries: try: data = mihoyo.fetch_page(entry.id, entry.game, entry.url, cache_dir=CACHE, offline=args.offline) except mihoyo.HttpError as exc: print(f"{entry.id}:抓取失败 {exc}", file=sys.stderr) return 1 fallback = scene_mod.pick_default_scene(data.scenes) choice = selection_mod.resolve( entry.id, data.scenes, sources_entry={"scene": entry.scene, "spines": entry.spines}, selection=selection, default_scene=fallback.id if fallback else None, default_spines=fallback.spine_ids if fallback else [], interactive=True, ) selection[entry.id] = choice.as_dict() print(f"[{entry.id}] 已记录推荐场景:{choice.scene}") selection_mod.save_selection(SELECTION, selection) print(f"\n选择记录 → {SELECTION.relative_to(ROOT)}") return 0 def cmd_promote(args: argparse.Namespace) -> int: """把 staging 的一个**场景目录**转成发布形状(`wallpapers/<游戏>/<壁纸id>/`)。""" staged = Path(args.page) if not staged.is_absolute(): candidate = OUT / args.page staged = candidate if candidate.exists() else Path(args.page) try: result = promote_mod.promote_page( staged, ROOT / "wallpapers", game=args.game, wallpaper_id=args.id, scene=args.scene, name=args.name, title=args.title, description=args.description, cover=args.cover, force=args.force, ) except (FileNotFoundError, FileExistsError) as exc: print(str(exc), file=sys.stderr) return 2 print( f"promote {result.scene_dir.name} → {result.target.relative_to(ROOT)}\n" f" 预设 part {result.parts}(骨架 {result.spines}、贴图平面 {result.images})" f",跳过纯色平面 {result.solids_skipped},落盘 {result.bytes / 1024 / 1024:.1f} MB" ) print(" 下一步:pnpm build(或 pnpm dev)让构建把它烘焙进 preset.js") return 0 def cmd_verify(args: argparse.Namespace) -> int: root = Path(args.root).resolve() if args.root else OUT found, stats = verify_mod.verify_all(root, pages=args.page) for problem in found: print(f"✗ {problem}") print( f"页面 {stats['pages']}、场景目录 {stats['sceneDirs']}、骨架 {stats['spines']}、" f"几何平面 {stats['images']}、体积 {stats['bytes'] / 1024 / 1024:.2f} MB" ) if stats["versions"]: versions = "、".join(f"{v}×{n}" for v, n in sorted(stats["versions"].items())) print(f"骨架版本:{versions}") return 1 if found else 0 def build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser(prog="python -m tools.downloader", description="米哈游活动页 Spine 抓取器") sub = parser.add_subparsers(dest="command", required=True) def common(p: argparse.ArgumentParser) -> None: p.add_argument("--page", action="append", help="只处理指定页面 id(可重复)") fetch = sub.add_parser("fetch", help="抓取并落 staging:每个有内容的场景一个壁纸目录(结尾自动 verify)") common(fetch) fetch.add_argument("--interactive", action="store_true", help="逐页交互式选「推荐哪个场景」") fetch.add_argument("--offline", action="store_true", help="只用 _cache/,不发起任何请求") fetch.add_argument("--force", action="store_true", help="重下已存在的产物") fetch.set_defaults(func=cmd_fetch) pick = sub.add_parser("select", help="只做交互式选择(推荐场景),写入 selection.yml") common(pick) pick.add_argument("--offline", action="store_true", help="只用 _cache/,不发起任何请求") pick.set_defaults(func=cmd_select) promote = sub.add_parser("promote", help="把一个场景目录转成 wallpapers/<游戏>/<壁纸id>/") promote.add_argument("--page", required=True, help="页面 id(如 kv45)、「页面/场景」(如 kv45/scene_ava)或场景目录的路径") promote.add_argument("--game", required=True, help="游戏 id(wallpapers/<游戏>/)") promote.add_argument("--id", required=True, help="壁纸 id(全局唯一,与 WE combo 的 value 一致)") promote.add_argument("--scene", help="页面目录下要 promote 的场景(默认取 page.json 的 chosenScene)") promote.add_argument("--name", help="显示名(默认沿用场景目录 meta.json 的)") promote.add_argument("--title", help="创意工坊标题(默认沿用场景目录 meta.json 的)") promote.add_argument("--description", help="WE 的 BBCode 文案(默认沿用场景目录 meta.json 的)") promote.add_argument("--cover", help="封面图逻辑名(scene/<名>.*),写进 backgroundImage") promote.add_argument("--force", action="store_true", help="目标目录已存在时覆盖") promote.set_defaults(func=cmd_promote) check = sub.add_parser("verify", help="自检 _out/ 的引用闭包与文件齐全") common(check) check.add_argument("--root", help="检查别的产物目录(默认 tools/downloader/_out)") check.set_defaults(func=cmd_verify) return parser def main(argv: list[str] | None = None) -> int: args = build_parser().parse_args(argv) try: return int(args.func(args)) except SourcesError as exc: print(f"sources.yml 有问题:{exc}", file=sys.stderr) return 2 if __name__ == "__main__": raise SystemExit(main())