Files
MediaOrganizer/src/lib/pipeline/process.sh
T
Shuery 0df07ad0ae refactor: 迁移至 bashly CLI 与功能域模块结构
- CLI 定义迁移至 src/bashly.yml(bashly 1.4 生成参数解析),入口收敛为 src/root_command.sh
- 源码按功能域重组:config/storage/integrate/media/pipeline,消除扁平散落
- 新增零依赖测试框架 tests/(123 用例)与 CI 流水线(lint + test + build + 产物一致性)
- build.sh 支持版本单一来源注入与 --check 产物校验;新增 lint.sh(bash -n + shellcheck 零容忍)
- 删除旧单文件 media_organizer.sh 与旧扁平模块
2026-08-14 22:41:28 +08:00

419 lines
17 KiB
Bash
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# shellcheck shell=bash
# shellcheck disable=SC2034,SC2141 # 跨文件共享全局变量(POOL_THRESHOLD_CHECK);IFS=$'...\...' 反斜杠为有意分隔符
# 入口目录校验:源目录存在可读;目的目录存在可写或可创建(干运行不创建目录)。
validate_directories() {
if [[ -n "$SOURCE_DIR" && (! -d "$SOURCE_DIR" || ! -r "$SOURCE_DIR") ]]; then
_log 错误 "源目录不存在或不可读:$SOURCE_DIR"
exit 1
fi
if [[ -z "$DESTINATION_DIR" ]]; then
return 0
fi
if [[ ! -e "$DESTINATION_DIR" ]]; then
if [[ "$DRY_RUN" == "true" ]]; then
_log 错误 "目的目录不存在(干运行不会创建目录):$DESTINATION_DIR"
exit 1
fi
if ! mkdir -p "$DESTINATION_DIR" 2>/dev/null; then
_log 错误 "无法创建目的目录:$DESTINATION_DIR"
exit 1
fi
elif [[ ! -d "$DESTINATION_DIR" || ! -w "$DESTINATION_DIR" ]]; then
_log 错误 "目的目录不可用(需为可写目录):$DESTINATION_DIR"
exit 1
fi
}
################################################################################
# 环境检查
################################################################################
# 检查硬链接支持(源与目标必须在同一文件系统)。
check_hardlink_support() {
if ! ${SKIP_HARDLINK_CHECK,,}; then
src_test="$SOURCE_DIR/.hardlink_test_src_$$"
dst_test="$DESTINATION_DIR/.hardlink_test_dst_$$"
touch "$src_test" 2>/dev/null || {
_log 错误 "无法在源目录创建测试文件"
exit 1
}
if ! ln "$src_test" "$dst_test" 2>/dev/null; then
rm -f "$src_test"
_log 错误 "无法创建硬链接"
exit 1
fi
rm -f "$dst_test" "$src_test"
fi
}
# 初始化缓存目录,处理 --refresh-cache(仅清空 tmdb/ 缓存,保留脚本自身学习数据)。
# 干运行模式只解析路径、不创建目录(零持久化副作用)。
init_cache_dir() {
# 解析默认缓存路径(优先级:环境变量 > 执行目录 mo_cache > 脚本目录 mo_cache)
if [[ -z "${CACHE_DIR:-}" ]]; then
if [[ -d "$PWD/mo_cache" ]]; then
CACHE_DIR="$PWD/mo_cache"
else
CACHE_DIR="$SCRIPT_DIR/mo_cache"
fi
fi
if [[ "${DRY_RUN:-false}" != "true" ]]; then
if ! mkdir -p "$CACHE_DIR" 2>/dev/null; then
CACHE_DIR=""
fi
fi
if [[ "$REFRESH_CACHE" == "true" ]] && [[ -n "$CACHE_DIR" ]]; then
cache_clear
[[ "${DRY_RUN:-false}" != "true" ]] && _log 信息 "TMDB 缓存已清除"
fi
# 旧版缓存布局(mo_cache/{movie,search,tv} 平铺)迁移到 tmdb/ 子目录
if [[ -n "$CACHE_DIR" ]]; then
migrate_tmdb_cache
fi
# 创建缓存子目录:tmdb/(镜像 TMDB API 路径,可再生)+ media_organizer/
# (脚本运行状态:已链接账本/失败冷却 ledger;用户配置类文件位于 mo_config/,
# 由 init_special_map/init_special_words/init_skip_dirs/init_season_offset 各自创建)
if [[ "${DRY_RUN:-false}" != "true" ]]; then
mkdir -p "$CACHE_DIR/tmdb/search/movie" "$CACHE_DIR/tmdb/search/tv" \
"$CACHE_DIR/tmdb/tv" "$CACHE_DIR/tmdb/movie" \
"$CACHE_DIR/media_organizer" 2>/dev/null || true
fi
}
################################################################################
# 主处理流水线
################################################################################
# 扫描源目录,收集视频文件列表。
# 按去除压制组/字幕组等方括号信息后的名称排序,
# 保证同一部剧先按季、再按集有序处理(先第一季再第二季)。
scan_files() {
_log 信息 "正在扫描源目录..."
local file ext ext_lower key entry
# 进程替换保证 while 在父 shell 执行,VIDEO_FILES 修改生效
# 用换行分隔 + sort(BusyBox sort 仅支持 -nru,不支持 -z/-k)
while IFS= read -r entry; do
file="${entry#*$'\t'}"
VIDEO_FILES+=("$file")
done < <(
find "$SOURCE_DIR" -type f -print0 |
while IFS= read -r -d '' file; do
ext="${file##*.}"
ext_lower="${ext,,}"
if [[ " ${VIDEO_EXTS[*]} " =~ [[:space:]]${ext_lower}[[:space:]] ]]; then
key=$(clean_name "$(basename "$file")")
printf '%s\t%s\n' "$key" "$file"
fi
done |
sort
)
# 收集音频文件(CD 音乐,归入 Music 分类)
while IFS= read -r entry; do
file="${entry#*$'\t'}"
AUDIO_FILES+=("$file")
done < <(
find "$SOURCE_DIR" -type f -print0 |
while IFS= read -r -d '' file; do
ext="${file##*.}"
ext_lower="${ext,,}"
if [[ " ${AUDIO_EXTS[*]} " =~ [[:space:]]${ext_lower}[[:space:]] ]]; then
printf '%s\t%s\n' "$(basename "$file")" "$file"
fi
done |
sort
)
_log 信息 "共发现 ${#VIDEO_FILES[@]} 个视频文件、${#AUDIO_FILES[@]} 个音频文件,开始识别..."
}
# 第一阶段:解析文件名并识别,构建 MEDIA_DESTINATION_MAP。
# 按扩展名分发单个文件到视频/音频处理(识别池 worker 入口)。
# 入口先做增量跳过:已链接(ledger 有效)与失败冷却(未到期)条目直接 SKIP,
# 不再走 parse/识别(干运行不启用——干运行展示全貌;--rerun 不经此路径)。
process_one_file() {
local f="$1"
if [[ "${DRY_RUN:-false}" != "true" ]]; then
if ledger_is_linked "$f"; then
_log 跳过 "增量跳过(已链接): $(basename "$f")"
emit "$f" SKIP "already_linked"
return 0
fi
fi
# 纠错学习命中:--rerun 手动修正过的同命名文件,直接使用用户指定的目标
# (key=去扩展名 + clean_name 小写,与 correction_learn 写入算法一致;
# dest 必须位于当前 DESTINATION_DIR 下才适用;
# 优先于失败冷却——用户显式纠错是强信号,无需再等冷却;
# 干运行同样展示命中结果——学习写入才受 dry-run 约束)
local ckey corr_dest
ckey="${f##*/}"
ckey="${ckey%.*}"
ckey=$(clean_name "$ckey")
ckey="${ckey,,}"
corr_dest="${CORRECTIONS_MAP[$ckey]:-}"
if [[ -n "$corr_dest" ]] && [[ -n "$DESTINATION_DIR" ]] && [[ "$corr_dest" == "$DESTINATION_DIR"/* ]]; then
local corr_subdir corr_filename
corr_subdir="${corr_dest#"$DESTINATION_DIR"/}"
corr_filename="${corr_subdir##*/}"
corr_subdir="${corr_subdir%/*}"
if [[ -n "$corr_subdir" && -n "$corr_filename" ]]; then
_log 信息 "纠错学习命中: $(basename "$f") -> $corr_subdir/"
emit "$f" DEST "$corr_subdir|$corr_filename"
return 0
fi
fi
if [[ "${DRY_RUN:-false}" != "true" ]]; then
if cooldown_active "$f"; then
_log 跳过 "冷却中跳过(失败冷却未到期): $(basename "$f")"
emit "$f" SKIP "cooldown"
return 0
fi
fi
local ext="${f##*.}"
ext="${ext,,}"
if [[ " ${VIDEO_EXTS[*]} " =~ [[:space:]]${ext}[[:space:]] ]]; then
process_one_video "$f"
elif [[ " ${AUDIO_EXTS[*]} " =~ [[:space:]]${ext}[[:space:]] ]]; then
process_one_audio "$f"
fi
}
# 单个视频文件的识别:parse → 分类 → 识别 → 协议输出。
# 协议:DEST(识别成功)/ PENDING(无结果或 unknown,待 AI)/ REQ_FAILED(请求失败)/
# SKIP(类型跳过)。待定登记全部由父进程合并时执行。
process_one_video() {
local video="$1"
_log 信息 "处理视频: $(basename "$video")"
push_indent
local info type title year season episode episode_end special_fragment base ext dest="" rc=0
info=$(parse_media_filename "$video")
IFS='|' read -r type title year season episode episode_end special_fragment <<<"$info"
base="${video##*/}"
base="${base%.*}"
ext="${video##*.}"
if [[ "$type" == "movie" ]]; then
dest=$(identify_movie "$base" "$ext" "$video" "$title" "$year") || rc=$?
elif [[ "$type" == "tv" ]]; then
dest=$(identify_tv_show "$title" "${season:-1}" "${episode:-0}" "${episode_end:-}" \
"$base" "$ext" "$special_fragment" "$video") || rc=$?
elif [[ "$type" == "musicvideo" ]]; then
# Music Videos(v9.6):本地规则识别(无网络);恒成功(artist 有 Unknown 兜底)
dest=$(identify_musicvideo "$video" "$title" "$ext") || rc=$?
fi
if [[ "$type" == "skip" ]]; then
# 电影域特典/附带:不整理(TMDB/Jellyfin 均不收录电影特典,避免误判为剧集特典)
_log 跳过 "电影特典/附带,不整理: $(basename "$video")"
if $AUTOMATED && [[ "${DRY_RUN:-false}" != "true" ]]; then
echo "$(date '+%Y-%m-%d %H:%M:%S') | $video | 电影特典不整理" >>"$SKIP_LOG_FILE"
fi
emit "$video" SKIP "skip_type"
pop_indent
return 0
elif [[ "$type" == "unknown" ]]; then
# unknown:识别不出类别,记录待 AI 判断(type=unknown,AI 返回 media_type 决定 movie/tv)
_log 信息 "类别未定,交 AI 判断: $(basename "$video")"
emit "$video" PENDING "unknown $(dir_chain "$video")"
pop_indent
return 0
elif [[ $rc -eq 2 ]]; then
# 请求失败(网络/密钥问题,重试耗尽)——不登记 AI 待定,语义与"无结果"分离
_log 错误 "TMDB 请求失败,跳过: $(basename "$video")"
emit "$video" REQ_FAILED
elif [[ $rc -eq 3 ]]; then
# 需甄别:identify 已 emit MATCH(交 AI 匹配)
_log 信息 "需甄别,交 AI 匹配: $(basename "$video")"
elif [[ $rc -eq 0 && -n "$dest" ]]; then
local destination_subdir destination_filename
IFS='|' read -r destination_subdir destination_filename <<<"$dest"
emit "$video" DEST "$destination_subdir|$destination_filename"
else
# 查询无结果(rc=1)→ 待 AI 搜索纠正
_log 信息 "未匹配,交 AI 搜索: $(basename "$video")"
emit "$video" PENDING "${type} $(dir_chain "$video")"
fi
pop_indent
}
# 第一阶段:识别池处理全部视频文件(并发)。
process_video() {
[[ ${#VIDEO_FILES[@]} -eq 0 ]] && return 0
_log 信息 "开始识别视频(${#VIDEO_FILES[@]} 个,并发 ${MEDIA_WORKERS})..."
POOL_THRESHOLD_CHECK=true
pool_run "${VIDEO_FILES[@]}"
}
# 解析 CD 目录名,输出:歌手|专辑。
# 目录名格式:[日期] 「专辑名」/歌手 [规格] (格式),如 "[250110] 「幸せのレシピ」/平井大 [24bit_48kHz] (flac)"。
# 解析失败时歌手回退为 "Unknown",专辑用原始目录名。
parse_cd_dir() {
local dir="$1"
local artist album
# 去掉 [日期] 前缀
local name="${dir#*] }"
# 循环去掉末尾的 [规格] 与 (格式) 后缀(可能多个)
local prev=""
while [[ "$name" != "$prev" ]]; do
prev="$name"
name=$(echo "$name" | sed -E 's/[[:space:]]*\[[^]]*\]$//; s/[[:space:]]*\([^)]*\)$//')
done
name=$(echo "$name" | sed -E 's/^[[:space:]]+|[[:space:]]+$//g')
# 按 / 分割:「专辑」/歌手
if [[ "$name" == *"/"* ]]; then
album="${name%%/*}"
artist="${name##*/}"
else
album="$name"
artist="Unknown"
fi
# 专辑名去掉 「」 引号
album=$(echo "$album" | sed -E 's/^「//; s/」$//' | sed -E 's/^[[:space:]]+|[[:space:]]+$//g')
artist=$(echo "$artist" | sed -E 's/^[[:space:]]+|[[:space:]]+$//g')
[[ -z "$artist" ]] && artist="Unknown"
[[ -z "$album" ]] && album="$name"
echo "$artist|$album"
}
# 单个音频文件的整理:伴随优先级(同名视频→音轨排除)→ 艺术家归类链(元数据)→ 协议输出。
# 协议:DEST(音频及其歌词/封面伴随,可链接);ARTIST_PENDING(多艺术家待 AI 判定)。
# 艺术家归类链:专辑艺术家 → 单歌曲艺术家 → 多艺术家交 AI → 目录名解析 → Unknown。
process_one_audio() {
local audio="$1"
# 向上找到 CD 目录(目录名含日期/专辑/格式特征,如 [250110] ...、「专辑」/歌手、…(flac))
local dir cd_root bn
dir=$(dirname "$audio")
cd_root="$dir"
while [[ "$cd_root" != "/" && "$cd_root" != "." ]]; do
bn=$(basename "$cd_root")
if echo "$bn" | grep -qE '\[[0-9]{6}\]|/|「|\(flac|\(mp3|\(wav|\(webp'; then
break
fi
cd_root=$(dirname "$cd_root")
done
# 未找到 CD 目录(走到 / 或 .)→ 以音频所在目录为专辑根,避免 parse_cd_dir("/") 产出垃圾
[[ "$cd_root" == "/" || "$cd_root" == "." ]] && cd_root="$dir"
local base_name="${audio%.*}"
# 伴随优先级(视频 > 音频):同名视频存在 → 该音频是视频的伴随音轨,
# 由视频的伴随逻辑处理(link 时跟随),不独立整理
local same_video="" v vext
for v in "$base_name".*; do
[[ ! -f "$v" ]] && continue
[[ "$v" == "$audio" ]] && continue
vext="${v##*.}"
if [[ " ${VIDEO_EXTS[*]} " =~ [[:space:]]${vext,,}[[:space:]] ]]; then
same_video="$v"
break
fi
done
if [[ -n "$same_video" ]]; then
_log 跳过 "音频为视频伴随音轨,不独立整理: $(basename "$audio")"
return 0
fi
# 元数据读取(ffprobe 读格式标签;仅整理不增删文件)
local meta album_artist meta_artist meta_album
meta=$(ffprobe -v quiet -show_entries format_tags=album_artist,artist,album \
-of default=noprint_wrappers=1 "$audio" 2>/dev/null) || meta=""
# 注意:pipefail 下 grep 无匹配会使管道整体返回非零,必须 || true(set -e 保护)
# ffmpeg 6.x 输出 "TAG:ALBUM=" 前缀且键全大写;grep -ioE 从键处提取(大小写不敏感),
# sed 去键("^[^=]*=" 兼容 TAG: 前缀与全大写键,busybox 无 sed -I)
album_artist=$(echo "$meta" | grep -ioE 'album_artist=.*' | head -1 | sed -E 's/^[^=]*=//') || true
meta_artist=$(echo "$meta" | grep -viE 'album_artist=' | grep -ioE 'artist=.*' | head -1 | sed -E 's/^[^=]*=//') || true
meta_album=$(echo "$meta" | grep -viE 'album_artist=' | grep -ioE 'album=.*' | head -1 | sed -E 's/^[^=]*=//') || true
# CD 目录名解析(元数据缺失时的回退源)
local parsed cd_artist cd_album
parsed=$(parse_cd_dir "$(basename "$cd_root")")
IFS='|' read -r cd_artist cd_album <<<"$parsed"
# CD 目录下的相对子路径(保留碟片子目录,如 THCA-60298-1)
local rel rel_dir
rel="${audio#"$cd_root"/}"
rel_dir=$(dirname "$rel")
[[ "$rel_dir" == "." ]] && rel_dir=""
local artist="" album="${meta_album:-$cd_album}"
if [[ -n "$album_artist" ]]; then
artist="$album_artist"
elif [[ -n "$meta_artist" ]]; then
# 拆分多艺术家(ID3v2 常用 / \ 分隔;日系曲目常用 、;,分隔)。
# 注意:& 不拆("MYTH & ROID" 是乐队名,误拆会破坏艺术家)
local -a arts=() cleaned_arts=() a
IFS=$'/\\;、,,' read -ra arts <<<"$meta_artist"
for a in "${arts[@]}"; do
a=$(echo "$a" | sed -E 's/^[[:space:]]+|[[:space:]]+$//g')
[[ -n "$a" ]] && cleaned_arts+=("$a")
done
if ((${#cleaned_arts[@]} <= 1)); then
artist="${cleaned_arts[0]:-$meta_artist}"
else
# 多艺术家 → 交 AI 判定(AI 判断不出时拼接,见 run_ai_batch)。
# 字段用 \t 分隔(艺术家名可含 |;PENDING_AI_SEARCH 同约定);
# 艺术家内部用顿号连接(日系命名惯例,与校准约定一致,避免 | 与字段分隔冲突)。
_log 信息 "多艺术家,交 AI 判定: $(basename "$audio")"
emit "$audio" ARTIST_PENDING \
"$(
IFS='; '
echo "${cleaned_arts[*]}"
)"$'\t'"${album:-}"$'\t'"${rel_dir}"
return 0
fi
fi
# 艺术家仍未解析 → 目录名解析 → Unknown
[[ -z "$artist" ]] && artist="$cd_artist"
[[ -z "$artist" ]] && artist="$FOLDER_UNKNOWN"
[[ -z "$album" ]] && album="$FOLDER_UNKNOWN"
local destination_subdir music_root
# 音乐目录结构走 NAMING_MUSIC 模板(默认 {artist}/{album};用户可自定义)
music_root=$(render_naming NAMING_MUSIC \
"artist=$(sanitize "$artist")" "album=$(sanitize "$album")")
[[ -n "$music_root" ]] || music_root="$(sanitize "$artist")/$(sanitize "$album")"
destination_subdir="${FOLDER_MUSIC}/$music_root"
[[ -n "$rel_dir" ]] && destination_subdir="$destination_subdir/$rel_dir"
emit "$audio" DEST "$destination_subdir|$(basename "$audio")"
_log 信息 "音频: $(basename "$audio") -> $destination_subdir/"
emit_audio_companions "$audio" "$cd_root" "$destination_subdir" "$artist" "$album"
}
# 音频伴随文件发射(链接既有文件,不提取/不生成):歌词(同名 lrc/elrc/txt)+ 专辑级封面。
# 歌词随曲目(含碟片相对路径);封面在源专辑根目录(目标专辑根)。
emit_audio_companions() {
local audio="$1" cd_root="$2" destination_subdir="$3" artist="$4" album="$5"
local base_name="${audio%.*}"
local c
for c in "${base_name}.lrc" "${base_name}.elrc" "${base_name}.txt"; do
[[ -f "$c" ]] && emit "$c" DEST "$destination_subdir|$(basename "$c")"
done
# 专辑封面:源专辑根目录的类型名图片(cover/folder/poster/jacket/thumb/default)
local cover_file cb cname music_root
music_root=$(render_naming NAMING_MUSIC \
"artist=$(sanitize "$artist")" "album=$(sanitize "$album")")
[[ -n "$music_root" ]] || music_root="$(sanitize "$artist")/$(sanitize "$album")"
for cover_file in "$cd_root"/*.jpg "$cd_root"/*.png "$cd_root"/*.webp; do
[[ ! -f "$cover_file" ]] && continue
cb=$(basename "$cover_file")
cname="${cb%.*}"
if echo "$cname" | grep -qiE '^(cover|folder|poster|jacket|thumb|default)$'; then
emit "$cover_file" DEST \
"${FOLDER_MUSIC}/$music_root|$cb"
break
fi
done
}
# 处理音频文件(CD 音乐):识别池并发处理,归入 Music/歌手/专辑/曲目。
process_audio() {
[[ ${#AUDIO_FILES[@]} -eq 0 ]] && return 0
_log 信息 "处理音频文件(CD 音乐,共 ${#AUDIO_FILES[@]} 个)..."
POOL_THRESHOLD_CHECK=false
pool_run "${AUDIO_FILES[@]}"
}