feat: TS 构建管线、Spine 抓取器与 wallpapers/ 唯一真相来源

把项目从「手写 dist/」改成「wallpapers/ 是唯一真相来源,dist/ 由 pnpm build 生成」,
并补上配套的类型、门禁与抓取器。一次提交落地整条管线,因为拆开会留下不能构建的中间态。

- src/:运行时与模拟器源码(TS,strict),编译到 build/ 再拷进各分发
- tools/:build / dev / check-{syntax,paths,dist},以及抓取器与回归门禁 tools/checks/
  (.scratch/ 下那批一次性脚本移入 tools/checks/ 并入库为长期门禁)
- wallpapers/:七档壁纸的源数据 + README.md(id/音频/预设的完整规范)
- docs/adr/0005-0008:构建管线与分发拓扑、模拟器契约、自包含 sim、每骨架资源布局
- .gitignore:排除 .scratch/ 的参考资料副本(上游 spine 整仓克隆 ~1.2 GB、
  抓取侦查数据 ~680 MB)与调试转储;这些是本地调查材料,补偿会让仓库无法克隆
- 归一化 .gitignore/CONTEXT.md 行尾(工作区 CRLF、索引 LF 造成的整文件假 diff)

同时修掉三档卡住构建的未完工壁纸:
- kv45 的 meta.json 里 id 还是抓取期场景名 scene_main,经 downloader promote 正名为 kv45
- shajin / zhigengniao_juheye 的 meta.json 误用了骨架描述文件(name/spine/animations/pages)、
  且都缺 preset.template.json;现按规范重建:骨架沉到 spines/<名>/(spine-ts 按 atlas 所在
  目录解析贴图页)、补上元数据与单骨架预设,并清掉 zhigengniao 骨架里指向作者机的绝对路径
- 顺带 promote 已在 sources.yml 里的 kv46(月升之前,与兽共舞)

pnpm check 五道门全绿:10 个分发 / 7 档壁纸 / 141 处引用自包含。
This commit is contained in:
Shuery committed 2026-10-02 01:27:02 +08:00
1 parent b8eee05d78
commit 3f11426964
297 files changed
+216627 -1926

No files matched your search

+217
View File
@@ -0,0 +1,217 @@
"""JS 字面量 → Python 对象。
米哈游活动页把骨架数据与场景树以 **JS 对象字面量** 内联在 bundle 里,不是合法 JSON:
{skeleton:{hash:"KR8Ibf8DXTI",spine:"4.2.43",x:-110.93},bones:[{name:"root"}],scaleX:.6487}
键不带引号、小数可以省略前导 0、布尔写作 ``!0`` / ``!1``、末尾可以有逗号。
这里只做**词法层**的规范化(把字面量改写成 JSON 文本),再交给 ``json.loads``——
不 ``eval``、不执行页面代码。
唯一被容忍的"语义"是未知裸标识符(例如 ``undefined``):默认替换为 ``null`` 并计数,
``strict=True`` 时抛错。骨架数据里出现别的裸标识符说明提取边界错了,值得知道。
"""
from __future__ import annotations
import json
import re
from typing import Any
__all__ = ["loads", "match_literal", "first_object_value", "unescape", "JsLitError"]
# 按优先级排列的词法单元。字符串/数字/``!0`` 必须排在 ``other`` 之前。
_TOKEN = re.compile(
r"""
(?P<ws>\s+)
| (?P<dquote>"(?:[^"\\]|\\.)*")
| (?P<squote>'(?:[^'\\]|\\.)*')
| (?P<template>`(?:[^`\\]|\\.)*`)
| (?P<negnot>!0|!1)
| (?P<num>-?(?:\d+\.\d*|\.\d+|\d+)(?:[eE][+-]?\d+)?)
| (?P<ident>[A-Za-z_$][\w$]*)
| (?P<other>.)
""",
re.VERBOSE | re.DOTALL,
)
_LITERAL_IDENTS = {"true": "true", "false": "false", "null": "null"}
_NULL_IDENTS = {"undefined", "NaN", "Infinity", "void"}
_ESCAPES = {"n": "\n", "t": "\t", "r": "\r", "b": "\b", "f": "\f", "v": "\v", "0": "\0"}
class JsLitError(ValueError):
"""字面量无法规范化成 JSON。"""
def unescape(body: str) -> str:
"""还原 JS 字符串体里的转义(``\\n`` / ``\\uXXXX`` / ``\\'`` / ``\\\\`` …)。
Family B 的骨架是 ``e.exports=JSON.parse('…')``:外层是**单引号** JS 字符串,
必须先按 JS 语义还原,再交给 ``json.loads``。
"""
return _unescape(body, "'")
def _unescape(body: str, quote: str) -> str:
"""还原 JS 字符串体(不含两端引号)里的转义。"""
out: list[str] = []
i = 0
while i < len(body):
ch = body[i]
if ch != "\\":
out.append(ch)
i += 1
continue
nxt = body[i + 1] if i + 1 < len(body) else ""
if nxt == "u" and i + 6 <= len(body):
out.append(chr(int(body[i + 2 : i + 6], 16)))
i += 6
elif nxt == "x" and i + 4 <= len(body):
out.append(chr(int(body[i + 2 : i + 4], 16)))
i += 4
elif nxt == "\n": # 行延续
i += 2
else:
out.append(_ESCAPES.get(nxt, nxt))
i += 2
return "".join(out)
def _num_key(value: str) -> str:
"""把数字字面量还原成 JS 当键时用的字符串(``0`` → ``"0"``、``.5`` → ``"0.5"``)。"""
try:
as_float = float(value)
except ValueError:
return value
if as_float.is_integer():
return str(int(as_float))
return repr(as_float)
def _normalize(text: str, *, strict: bool) -> tuple[str, int]:
out: list[str] = []
unknown = 0
pos = 0
for m in _TOKEN.finditer(text):
kind = m.lastgroup
raw = m.group()
if kind == "ws" or kind == "other":
out.append(raw)
pos = m.end()
continue
if kind in ("dquote", "squote", "template"):
body = raw[1:-1]
quote = raw[0]
out.append(json.dumps(_unescape(body, quote), ensure_ascii=False))
elif kind == "negnot":
out.append("true" if raw == "!0" else "false")
elif kind == "num":
value = raw
if value.startswith("-."):
value = "-0" + value[1:]
elif value.startswith("."):
value = "0" + value
if value.endswith("."):
value += "0"
# 数字也能当键(动画名就叫 "0" 的骨架真实存在:`animations:{0:{…}}`)。
# JS 会把数字字面量转成字符串当键,所以这里要按同一语义还原。
if re.match(r"\s*:", text[m.end() :]):
out.append(json.dumps(_num_key(value), ensure_ascii=False))
else:
out.append(value)
elif kind == "ident":
# 后面(跳过空白)跟冒号的标识符是键,否则是值。
tail = text[m.end() :]
is_key = bool(re.match(r"\s*:", tail))
if is_key:
out.append(json.dumps(raw, ensure_ascii=False))
elif raw in _LITERAL_IDENTS:
out.append(_LITERAL_IDENTS[raw])
elif raw in _NULL_IDENTS:
out.append("null")
unknown += 1
if strict:
raise JsLitError(f"未知标识符 {raw!r} @ {m.start()}")
else:
out.append("null")
unknown += 1
if strict:
raise JsLitError(f"未知标识符 {raw!r} @ {m.start()}")
pos = m.end()
if pos < len(text):
out.append(text[pos:])
normalized = "".join(out)
# 尾逗号:`,}` / `,]`
normalized = re.sub(r",(\s*[}\]])", r"\1", normalized)
return normalized, unknown
def loads(text: str, *, strict: bool = False) -> Any:
"""把 JS 字面量文本解析成 Python 对象。"""
normalized, _ = _normalize(text, strict=strict)
try:
return json.loads(normalized)
except json.JSONDecodeError as exc:
head = text[max(0, exc.pos - 120) : exc.pos + 120].replace("\n", " ")
raise JsLitError(f"字面量规范化后仍不是 JSON:{exc.msg} @ {exc.pos}\n…{head}…") from exc
def match_literal(text: str, start: int) -> int:
"""返回 ``text[start]`` 处那个配平字面量的**闭括号下标**;找不到返回 -1。
与 JS 侧同名的辅助函数一一对应:跳过字符串/模板串/注释,按开闭括号配平。
它是所有提取器的地基——正则数不清嵌套括号,只有这个能。
"""
if start >= len(text):
return -1
open_ch = text[start]
close_ch = {"{": "}", "[": "]", "(": ")"}.get(open_ch)
if close_ch is None:
return -1
depth = 0
i = start
while i < len(text):
ch = text[i]
if ch in "\"'`":
quote = ch
i += 1
while i < len(text):
if text[i] == "\\":
i += 2
elif text[i] == quote:
break
else:
i += 1
i += 1
continue
if ch == "/" and i + 1 < len(text) and text[i + 1] == "/":
while i < len(text) and text[i] != "\n":
i += 1
continue
if ch == "/" and i + 1 < len(text) and text[i + 1] == "*":
i += 2
while i + 1 < len(text) and not (text[i] == "*" and text[i + 1] == "/"):
i += 1
i += 2
continue
if ch == open_ch:
depth += 1
elif ch == close_ch:
depth -= 1
if depth == 0:
return i
i += 1
return -1
def first_object_value(text: str, brace_index: int) -> Any:
"""``Object.values(Object.assign({k: v}))[0]`` 的取值语义:解析 ``{...}`` 并返回第一个值。"""
end = match_literal(text, brace_index)
if end < 0:
raise JsLitError(f"未配平的对象字面量 @ {brace_index}")
obj = loads(text[brace_index : end + 1])
if not isinstance(obj, dict) or not obj:
raise JsLitError(f"期望非空对象字面量 @ {brace_index}")
return next(iter(obj.values()))