Files
SpineWallpaper/tools/downloader/jslit.py
T
Shuery 3f11426964 feat: TS 构建管线、Spine 抓取器与 wallpapers/ 唯一真相来源
把项目从「手写 dist/」改成「wallpapers/ 是唯一真相来源,dist/ 由 pnpm build 生成」,
并补上配套的类型、门禁与抓取器。一次提交落地整条管线,因为拆开会留下不能构建的中间态。

- src/:运行时与模拟器源码(TS,strict),编译到 build/ 再拷进各分发
- tools/:build / dev / check-{syntax,paths,dist},以及抓取器与回归门禁 tools/checks/
  (.scratch/ 下那批一次性脚本移入 tools/checks/ 并入库为长期门禁)
- wallpapers/:七档壁纸的源数据 + README.md(id/音频/预设的完整规范)
- docs/adr/0005-0008:构建管线与分发拓扑、模拟器契约、自包含 sim、每骨架资源布局
- .gitignore:排除 .scratch/ 的参考资料副本(上游 spine 整仓克隆 ~1.2 GB、
  抓取侦查数据 ~680 MB)与调试转储;这些是本地调查材料,补偿会让仓库无法克隆
- 归一化 .gitignore/CONTEXT.md 行尾(工作区 CRLF、索引 LF 造成的整文件假 diff)

同时修掉三档卡住构建的未完工壁纸:
- kv45 的 meta.json 里 id 还是抓取期场景名 scene_main,经 downloader promote 正名为 kv45
- shajin / zhigengniao_juheye 的 meta.json 误用了骨架描述文件(name/spine/animations/pages)、
  且都缺 preset.template.json;现按规范重建:骨架沉到 spines/<名>/(spine-ts 按 atlas 所在
  目录解析贴图页)、补上元数据与单骨架预设,并清掉 zhigengniao 骨架里指向作者机的绝对路径
- 顺带 promote 已在 sources.yml 里的 kv46(月升之前,与兽共舞)

pnpm check 五道门全绿:10 个分发 / 7 档壁纸 / 141 处引用自包含。
2026-10-02 01:27:02 +08:00

218 lines
7.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""JS 字面量 → Python 对象。
米哈游活动页把骨架数据与场景树以 **JS 对象字面量** 内联在 bundle 里,不是合法 JSON:
{skeleton:{hash:"KR8Ibf8DXTI",spine:"4.2.43",x:-110.93},bones:[{name:"root"}],scaleX:.6487}
键不带引号、小数可以省略前导 0、布尔写作 ``!0`` / ``!1``、末尾可以有逗号。
这里只做**词法层**的规范化(把字面量改写成 JSON 文本),再交给 ``json.loads``——
不 ``eval``、不执行页面代码。
唯一被容忍的"语义"是未知裸标识符(例如 ``undefined``):默认替换为 ``null`` 并计数,
``strict=True`` 时抛错。骨架数据里出现别的裸标识符说明提取边界错了,值得知道。
"""
from __future__ import annotations
import json
import re
from typing import Any
__all__ = ["loads", "match_literal", "first_object_value", "unescape", "JsLitError"]
# 按优先级排列的词法单元。字符串/数字/``!0`` 必须排在 ``other`` 之前。
_TOKEN = re.compile(
r"""
(?P<ws>\s+)
| (?P<dquote>"(?:[^"\\]|\\.)*")
| (?P<squote>'(?:[^'\\]|\\.)*')
| (?P<template>`(?:[^`\\]|\\.)*`)
| (?P<negnot>!0|!1)
| (?P<num>-?(?:\d+\.\d*|\.\d+|\d+)(?:[eE][+-]?\d+)?)
| (?P<ident>[A-Za-z_$][\w$]*)
| (?P<other>.)
""",
re.VERBOSE | re.DOTALL,
)
_LITERAL_IDENTS = {"true": "true", "false": "false", "null": "null"}
_NULL_IDENTS = {"undefined", "NaN", "Infinity", "void"}
_ESCAPES = {"n": "\n", "t": "\t", "r": "\r", "b": "\b", "f": "\f", "v": "\v", "0": "\0"}
class JsLitError(ValueError):
"""字面量无法规范化成 JSON。"""
def unescape(body: str) -> str:
"""还原 JS 字符串体里的转义(``\\n`` / ``\\uXXXX`` / ``\\'`` / ``\\\\`` …)。
Family B 的骨架是 ``e.exports=JSON.parse('…')``:外层是**单引号** JS 字符串,
必须先按 JS 语义还原,再交给 ``json.loads``。
"""
return _unescape(body, "'")
def _unescape(body: str, quote: str) -> str:
"""还原 JS 字符串体(不含两端引号)里的转义。"""
out: list[str] = []
i = 0
while i < len(body):
ch = body[i]
if ch != "\\":
out.append(ch)
i += 1
continue
nxt = body[i + 1] if i + 1 < len(body) else ""
if nxt == "u" and i + 6 <= len(body):
out.append(chr(int(body[i + 2 : i + 6], 16)))
i += 6
elif nxt == "x" and i + 4 <= len(body):
out.append(chr(int(body[i + 2 : i + 4], 16)))
i += 4
elif nxt == "\n": # 行延续
i += 2
else:
out.append(_ESCAPES.get(nxt, nxt))
i += 2
return "".join(out)
def _num_key(value: str) -> str:
"""把数字字面量还原成 JS 当键时用的字符串(``0`` → ``"0"``、``.5`` → ``"0.5"``)。"""
try:
as_float = float(value)
except ValueError:
return value
if as_float.is_integer():
return str(int(as_float))
return repr(as_float)
def _normalize(text: str, *, strict: bool) -> tuple[str, int]:
out: list[str] = []
unknown = 0
pos = 0
for m in _TOKEN.finditer(text):
kind = m.lastgroup
raw = m.group()
if kind == "ws" or kind == "other":
out.append(raw)
pos = m.end()
continue
if kind in ("dquote", "squote", "template"):
body = raw[1:-1]
quote = raw[0]
out.append(json.dumps(_unescape(body, quote), ensure_ascii=False))
elif kind == "negnot":
out.append("true" if raw == "!0" else "false")
elif kind == "num":
value = raw
if value.startswith("-."):
value = "-0" + value[1:]
elif value.startswith("."):
value = "0" + value
if value.endswith("."):
value += "0"
# 数字也能当键(动画名就叫 "0" 的骨架真实存在:`animations:{0:{…}}`)。
# JS 会把数字字面量转成字符串当键,所以这里要按同一语义还原。
if re.match(r"\s*:", text[m.end() :]):
out.append(json.dumps(_num_key(value), ensure_ascii=False))
else:
out.append(value)
elif kind == "ident":
# 后面(跳过空白)跟冒号的标识符是键,否则是值。
tail = text[m.end() :]
is_key = bool(re.match(r"\s*:", tail))
if is_key:
out.append(json.dumps(raw, ensure_ascii=False))
elif raw in _LITERAL_IDENTS:
out.append(_LITERAL_IDENTS[raw])
elif raw in _NULL_IDENTS:
out.append("null")
unknown += 1
if strict:
raise JsLitError(f"未知标识符 {raw!r} @ {m.start()}")
else:
out.append("null")
unknown += 1
if strict:
raise JsLitError(f"未知标识符 {raw!r} @ {m.start()}")
pos = m.end()
if pos < len(text):
out.append(text[pos:])
normalized = "".join(out)
# 尾逗号:`,}` / `,]`
normalized = re.sub(r",(\s*[}\]])", r"\1", normalized)
return normalized, unknown
def loads(text: str, *, strict: bool = False) -> Any:
"""把 JS 字面量文本解析成 Python 对象。"""
normalized, _ = _normalize(text, strict=strict)
try:
return json.loads(normalized)
except json.JSONDecodeError as exc:
head = text[max(0, exc.pos - 120) : exc.pos + 120].replace("\n", " ")
raise JsLitError(f"字面量规范化后仍不是 JSON:{exc.msg} @ {exc.pos}\n…{head}…") from exc
def match_literal(text: str, start: int) -> int:
"""返回 ``text[start]`` 处那个配平字面量的**闭括号下标**;找不到返回 -1。
与 JS 侧同名的辅助函数一一对应:跳过字符串/模板串/注释,按开闭括号配平。
它是所有提取器的地基——正则数不清嵌套括号,只有这个能。
"""
if start >= len(text):
return -1
open_ch = text[start]
close_ch = {"{": "}", "[": "]", "(": ")"}.get(open_ch)
if close_ch is None:
return -1
depth = 0
i = start
while i < len(text):
ch = text[i]
if ch in "\"'`":
quote = ch
i += 1
while i < len(text):
if text[i] == "\\":
i += 2
elif text[i] == quote:
break
else:
i += 1
i += 1
continue
if ch == "/" and i + 1 < len(text) and text[i + 1] == "/":
while i < len(text) and text[i] != "\n":
i += 1
continue
if ch == "/" and i + 1 < len(text) and text[i + 1] == "*":
i += 2
while i + 1 < len(text) and not (text[i] == "*" and text[i + 1] == "/"):
i += 1
i += 2
continue
if ch == open_ch:
depth += 1
elif ch == close_ch:
depth -= 1
if depth == 0:
return i
i += 1
return -1
def first_object_value(text: str, brace_index: int) -> Any:
"""``Object.values(Object.assign({k: v}))[0]`` 的取值语义:解析 ``{...}`` 并返回第一个值。"""
end = match_literal(text, brace_index)
if end < 0:
raise JsLitError(f"未配平的对象字面量 @ {brace_index}")
obj = loads(text[brace_index : end + 1])
if not isinstance(obj, dict) or not obj:
raise JsLitError(f"期望非空对象字面量 @ {brace_index}")
return next(iter(obj.values()))