1593 lines
68 KiB
Python
1593 lines
68 KiB
Python
# -*- coding: utf-8 -*-
|
||
"""Anisong Organizer v5
|
||
|
||
扫描音乐文件夹,自动识别并分类整理:
|
||
- 识别到动漫(本地映射表或四源联网) → Anisong 分类,按系列首播年份排序
|
||
- 非动漫 → MusicBrainz 查歌手地区 → 日韩 / 中国港澳台 / 中国大陆 / 欧美
|
||
- 查不到 → 其他(未被识别)
|
||
每个分类下含四个同级子文件夹: 歌曲(解密后) / 加密原件(.ncm) / 歌词 / 封面,
|
||
编号分类内连续且三处同号;空分类不创建文件夹。
|
||
|
||
工具包结构(分层):
|
||
app/ 程序(本文件 / ncm_decrypt.py / web_server.py / web_ui.html)
|
||
data/ 用户数据(anime_map.json / artist_map.json / 设置.json / 导览.txt)
|
||
python/ 内置 Python 嵌入式运行时(免安装)
|
||
工具包可整体复制到任意音乐文件夹使用,双击 启动网页版.bat 即开浏览器界面。
|
||
|
||
用法:
|
||
python app/anime_sorter.py # 预览(dry-run),确认后输入 y 执行
|
||
python app/anime_sorter.py --apply # 跳过确认直接执行
|
||
python app/anime_sorter.py --offline # 禁用联网,只查本地映射表/缓存
|
||
python app/anime_sorter.py 文件夹路径 # 处理指定文件夹
|
||
python app/anime_sorter.py 下载文件夹 --recursive # 递归整理子目录(网易云下载结构)
|
||
python app/anime_sorter.py --web # 启动浏览器交互界面(关闭网页即退出)
|
||
|
||
数据源(可在 data/设置.json 中启停):
|
||
animethemes AnimeThemes.moe,存罗马字/英文标题,含年份季度,无需代理
|
||
anison anison.info 动漫歌曲库,日文曲名,需要加速器/代理
|
||
wikipedia 日语维基百科,从歌曲条目提取动漫,需要加速器/代理
|
||
moegirl 萌娘百科,中日文曲名,无需代理
|
||
musicbrainz MusicBrainz 歌手地区/成立时间,需要加速器/代理(离线可用本地缓存)
|
||
"""
|
||
|
||
import datetime
|
||
import json
|
||
import os
|
||
import re
|
||
import ssl
|
||
import struct
|
||
import sys
|
||
import time
|
||
import unicodedata
|
||
import urllib.error
|
||
import urllib.parse
|
||
import urllib.request
|
||
from email.utils import parsedate_to_datetime
|
||
|
||
# 音乐文件扩展名(FLAC 可读标签,其他格式回退文件名解析)
|
||
AUDIO_EXTS = ('.flac', '.mp3', '.m4a', '.wav', '.ape', '.ogg')
|
||
# 可处理的全部扩展名(含待解密的 ncm,用于判断目标文件夹是否有内容)
|
||
PROC_EXTS = AUDIO_EXTS + ('.ncm',)
|
||
HTTP_HEADERS = {"User-Agent": "Mozilla/5.0 AnisongOrganizer/4.0"}
|
||
|
||
|
||
def tool_root():
|
||
"""工具包根目录(本文件位于根目录下的 app 子目录)。"""
|
||
return os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
|
||
# Windows 文件名禁用字符 -> 全角
|
||
FORBIDDEN = str.maketrans({
|
||
'\\': '\', '/': '/', ':': ':', '*': '*', '?': '?',
|
||
'"': '"', '<': '<', '>': '>', '|': '|',
|
||
})
|
||
|
||
# 中文简体混入日文标题时的常见替换(谜->謎 等)
|
||
CN2JP = str.maketrans({
|
||
'谜': '謎', '风': '風', '云': '雲', '气': '気', '决': '決',
|
||
'剧': '劇', '剑': '剣', '梦': '夢', '语': '語', '经': '経',
|
||
'难': '難', '个': '個', '发': '発',
|
||
})
|
||
|
||
SEASON_MONTH = {'Winter': '01', 'Spring': '04', 'Summer': '07', 'Fall': '10'}
|
||
|
||
# 数据源注册表: 名称 -> (查询函数, 是否需要加速器/代理)
|
||
SOURCES = {
|
||
'animethemes': ('animethemes_lookup', False),
|
||
'anison': ('anison_lookup', True),
|
||
'wikipedia': ('wikipedia_lookup', True),
|
||
'moegirl': ('moegirl_lookup', False),
|
||
}
|
||
SOURCE_ORDER = ['animethemes', 'anison', 'wikipedia', 'moegirl']
|
||
# 未提供 设置.json 时的默认启用源(均为无需代理的源)
|
||
DEFAULT_ENABLED = {'animethemes': True, 'anison': False, 'wikipedia': False, 'moegirl': True}
|
||
|
||
# 动漫源策略: 按曲名语言调整优先级(动漫歌曲独特适配)——假名=日文曲名,以日文库
|
||
# anison 优先;纯汉字(中文或日文汉字)以覆盖中日文的萌娘优先;纯 ASCII 以存罗马字的
|
||
# animethemes 优先。各源并行查询、按此优先级取首个命中。
|
||
SOURCE_PRIORITY = {
|
||
'kana': ['anison', 'moegirl', 'wikipedia', 'animethemes'],
|
||
'cjk': ['moegirl', 'anison', 'wikipedia', 'animethemes'],
|
||
'ascii': ['animethemes', 'moegirl', 'anison', 'wikipedia'],
|
||
}
|
||
# 源熔断: 连续失败达此数后本次运行停用该源(避免无梯子时每首歌都白等超时)
|
||
SOURCE_MAX_FAILS = 2
|
||
|
||
|
||
def _title_lang(s):
|
||
"""曲名语言类型: kana(含假名) / cjk(纯汉字) / ascii(拉丁字母等)。"""
|
||
if re.search(r'[-ヿ]', s):
|
||
return 'kana'
|
||
if re.search(r'[一-鿿]', s):
|
||
return 'cjk'
|
||
return 'ascii'
|
||
|
||
# 六个输出分类(键稳定,显示名可在 设置.json 配置)
|
||
CATEGORY_KEYS = ['anisong', 'jp_kr', 'hk_mo_tw', 'cn_mainland', 'western', 'other']
|
||
# MusicBrainz 国家/地区码 -> 分类键
|
||
MB_COUNTRY_CAT = {'JP': 'jp_kr', 'KR': 'jp_kr',
|
||
'HK': 'hk_mo_tw', 'MO': 'hk_mo_tw', 'TW': 'hk_mo_tw',
|
||
'CN': 'cn_mainland'}
|
||
|
||
|
||
# ---------- 基础工具 ----------
|
||
|
||
def read_flac_tags(path):
|
||
"""读取 FLAC 的 Vorbis 标签,返回 {KEY: [值, ...]};无标签返回空 dict;失败返回 None。"""
|
||
try:
|
||
with open(path, "rb") as f:
|
||
head = f.read(10)
|
||
if head.startswith(b"ID3"): # 跳过前置 ID3v2 标签
|
||
size = (head[6] << 21) | (head[7] << 14) | (head[8] << 7) | head[9]
|
||
f.seek(10 + size)
|
||
head = f.read(4)
|
||
else:
|
||
f.seek(4)
|
||
if head[:4] != b"fLaC":
|
||
return None
|
||
last = False
|
||
while not last:
|
||
hdr = f.read(4)
|
||
if len(hdr) < 4:
|
||
break
|
||
last = bool(hdr[0] & 0x80)
|
||
btype = hdr[0] & 0x7F
|
||
blen = struct.unpack(">I", b"\x00" + hdr[1:4])[0]
|
||
data = f.read(blen)
|
||
if len(data) < blen:
|
||
break
|
||
if btype == 4: # Vorbis comment 块
|
||
pos = 0
|
||
|
||
def u32():
|
||
nonlocal pos
|
||
v = struct.unpack_from("<I", data, pos)[0]
|
||
pos += 4
|
||
return v
|
||
|
||
pos += u32() + 4 # vendor
|
||
n = u32()
|
||
out = {}
|
||
for _ in range(n):
|
||
l = u32()
|
||
if pos + l > len(data):
|
||
break
|
||
kv = data[pos:pos + l].decode("utf-8", "replace")
|
||
pos += l
|
||
if "=" in kv:
|
||
k, v = kv.split("=", 1)
|
||
out.setdefault(k, []).append(v)
|
||
return out
|
||
if last:
|
||
break
|
||
return {}
|
||
except Exception:
|
||
return None
|
||
|
||
|
||
def normalize(s):
|
||
"""曲名规范化(用于匹配):简中->日文、全角空格、去尾部注释组、合并空格。"""
|
||
s = s.translate(CN2JP)
|
||
s = s.replace(' ', ' ')
|
||
while True: # 去掉尾部的 (注释) / <注释> / 【注释】
|
||
m = re.search(r'[\((【<\[][^()()【\]<>]*[\))】>\]](?:\s*)$', s)
|
||
if not m:
|
||
break
|
||
s = s[:m.start()].rstrip()
|
||
# 去掉 feat./ft. 及其后的合作者标注
|
||
s = re.sub(r'\s+(?:feat\.?|ft\.?)\s+.+$', '', s, flags=re.I)
|
||
# 去掉尾部的版本标注(Director's Edit. / TV Size / Short Ver. 等)
|
||
s = re.sub(
|
||
r"\s+(?:Director's Edit\..*|TV\s*Size.*|Short\s*Ver\..*|Original\s*Ver\..*|Live\s*Ver\..*)$",
|
||
'', s, flags=re.I)
|
||
return re.sub(r'\s+', ' ', s).strip()
|
||
|
||
|
||
def sanitize(title):
|
||
"""最终文件名清理:禁用字符转全角、尾随点转全角、去尾随空格。"""
|
||
t = title.translate(FORBIDDEN)
|
||
t = t.rstrip(' ')
|
||
if t.endswith('.'):
|
||
t = t.rstrip('.') + '.' # Windows 会静默丢弃尾随点,用全角保留
|
||
if len(t) > 120:
|
||
t = t[:120].rstrip(' ')
|
||
return t
|
||
|
||
|
||
_SSL_CTX = ssl.create_default_context()
|
||
_SSL_CTX.set_ciphers("DEFAULT@SECLEVEL=1") # 兼容使用老旧 TLS 的站点
|
||
|
||
|
||
def http_get_json(url, timeout=8):
|
||
req = urllib.request.Request(url, headers=HTTP_HEADERS)
|
||
with urllib.request.urlopen(req, timeout=timeout, context=_SSL_CTX) as r:
|
||
return json.load(r)
|
||
|
||
|
||
def http_get_text(url, timeout=8):
|
||
req = urllib.request.Request(url, headers=HTTP_HEADERS)
|
||
with urllib.request.urlopen(req, timeout=timeout, context=_SSL_CTX) as r:
|
||
return r.read().decode("utf-8", "replace")
|
||
|
||
|
||
def fold(s):
|
||
"""NFKC 归一化(全角标点转半角等)+小写,用于宽松比较。"""
|
||
return unicodedata.normalize('NFKC', s).casefold().strip()
|
||
|
||
|
||
_KKS = None
|
||
|
||
|
||
def _to_romaji(s):
|
||
"""日文转罗马字(依赖 pykakasi,未安装则返回 None)。"""
|
||
global _KKS
|
||
try:
|
||
if _KKS is None:
|
||
import pykakasi
|
||
_KKS = pykakasi.kakasi()
|
||
text = ''
|
||
for item in _KKS.convert(s):
|
||
text += item.get('hepburn') or item.get('orig') or ''
|
||
return text
|
||
except Exception:
|
||
return None
|
||
|
||
|
||
def _lev(a, b):
|
||
"""编辑距离(用于歌手名的罗马字模糊比较)。"""
|
||
if abs(len(a) - len(b)) > 2:
|
||
return 99
|
||
prev = list(range(len(b) + 1))
|
||
for i, ca in enumerate(a, 1):
|
||
cur = [i]
|
||
for j, cb in enumerate(b, 1):
|
||
cur.append(min(prev[j] + 1, cur[j - 1] + 1, prev[j - 1] + (ca != cb)))
|
||
prev = cur
|
||
return prev[-1]
|
||
|
||
|
||
# ---------- 联网数据源 ----------
|
||
# 约定: 每个 lookup 函数返回 (动漫名, 日期'YYYY-MM'或None, 候选列表) 或 (None, None, [])
|
||
# 网络异常向上抛出,由主循环记录并提示用户。
|
||
|
||
def animethemes_lookup(title, artist):
|
||
"""AnimeThemes.moe: 存罗马字/英文标题,日文曲名自动转罗马字再查。"""
|
||
def fetch(q):
|
||
url = ("https://api.animethemes.moe/song?filter[title]=" + urllib.parse.quote(q)
|
||
+ "&page[size]=8&include=animethemes.anime,artists")
|
||
return http_get_json(url).get("songs", [])
|
||
|
||
def score(song):
|
||
s = 0
|
||
if fold(song['title']) == fold(title):
|
||
s += 50
|
||
al = fold(artist)
|
||
al_romaji = fold(_to_romaji(artist) or artist)
|
||
for a in song.get('artists', []):
|
||
af = fold(a.get('name') or '')
|
||
if af and al and (af in al or al in af
|
||
or af in al_romaji or al_romaji in af
|
||
or (len(af) >= 3 and _lev(af, al_romaji) <= 2)):
|
||
s += 100
|
||
break
|
||
return s
|
||
|
||
query = unicodedata.normalize('NFKC', title)
|
||
songs = fetch(query)
|
||
if not songs:
|
||
romaji = _to_romaji(query)
|
||
if romaji: # 依次尝试: 罗马字全文 -> 逐级缩短前缀
|
||
tried = set()
|
||
for f in (1.0, 0.7, 0.6, 0.5, 0.4, 0.35, 0.3):
|
||
q = romaji[:max(4, int(len(romaji) * f))].strip()
|
||
if q in tried:
|
||
continue
|
||
tried.add(q)
|
||
songs = fetch(q)
|
||
if songs:
|
||
break
|
||
if not songs:
|
||
return None, None, []
|
||
|
||
ranked = sorted(songs, key=score, reverse=True)
|
||
best, second = ranked[0], (ranked[1] if len(ranked) > 1 else None)
|
||
cands = []
|
||
for s in ranked[:3]:
|
||
animes = []
|
||
for at in s.get('animethemes', []):
|
||
an = at.get('anime') or {}
|
||
if an.get('name'):
|
||
animes.append((an.get('name'), an.get('year'), an.get('season')))
|
||
cands.append((s['title'], [a.get('name') for a in s.get('artists', [])], animes))
|
||
s_best, s_second = score(best), score(second) if second else -1
|
||
# 高置信: 歌手匹配,或唯一候选(罗马字查询后唯一命中也算),或标题精确且明显领先
|
||
if (s_best >= 100 or len(ranked) == 1
|
||
or (s_best >= 50 and s_best - s_second >= 50)):
|
||
for at in best.get('animethemes', []):
|
||
an = at.get('anime') or {}
|
||
if an.get('name'):
|
||
year = an.get('year')
|
||
mm = SEASON_MONTH.get(an.get('season'), '01')
|
||
return an['name'], (f"{year}-{mm}" if year else None), cands
|
||
return None, None, cands
|
||
|
||
|
||
def anison_lookup(title, artist):
|
||
"""anison.info: 动漫歌曲数据库,搜索结果的表格行直接包含动漫名与 OP/ED 类型。"""
|
||
page = http_get_text("http://anison.info/data/n.php?m=song&q=" + urllib.parse.quote(title))
|
||
rows = [] # (曲名, 歌手, 动漫名, 类型, 作品页id)
|
||
for tr in re.findall(r'<tr[^>]*>(.*?)</tr>', page, re.S):
|
||
prog = re.search(r"link\(\s*['\"]program['\"]\s*,\s*['\"](\d+)['\"]\s*\)", tr)
|
||
if not prog:
|
||
continue # 没有作品链接 = 非动漫曲
|
||
cells = [re.sub(r'<[^>]+>', '', c).strip() for c in re.findall(r'<td[^>]*>(.*?)</td>', tr, re.S)]
|
||
if len(cells) < 4 or fold(cells[0]) != fold(title):
|
||
continue
|
||
rows.append((cells[0], cells[1], cells[3], cells[4] if len(cells) > 4 else '', prog.group(1)))
|
||
if not rows:
|
||
return None, None, []
|
||
|
||
af = fold(artist)
|
||
if af: # 歌手匹配优先
|
||
hits = [r for r in rows if af in fold(r[1]) or fold(r[1]) in af]
|
||
if hits:
|
||
rows = hits
|
||
|
||
cands = [(r[0], [r[1]], [(r[2],)]) for r in rows[:3]]
|
||
if len(rows) == 1:
|
||
anime = rows[0][2]
|
||
year = None
|
||
try: # 作品页取年份(取第一个出现的年份)
|
||
ppage = http_get_text(f"http://anison.info/data/program/{rows[0][4]}.html")
|
||
ym = re.search(r'(\d{4})年', ppage)
|
||
if ym:
|
||
year = ym.group(1)
|
||
except Exception:
|
||
pass
|
||
return anime, (f"{year}-01" if year else None), cands
|
||
return None, None, cands
|
||
|
||
|
||
def wikipedia_lookup(title, artist):
|
||
"""日语维基百科: 从歌曲条目的「テレビアニメ『X』のOP/ED」句式提取动漫。"""
|
||
def search(q):
|
||
url = ("https://ja.wikipedia.org/w/api.php?action=query&list=search&srsearch="
|
||
+ urllib.parse.quote(q) + "&srlimit=5&format=json")
|
||
return http_get_json(url)["query"]["search"]
|
||
|
||
def extract(r):
|
||
url = ("https://ja.wikipedia.org/w/api.php?action=query&prop=extracts&explaintext=1"
|
||
"&titles=" + urllib.parse.quote(r['title']) + "&format=json")
|
||
pages = http_get_json(url)["query"]["pages"]
|
||
return list(pages.values())[0].get("extract", "")
|
||
|
||
tf, af = fold(title), fold(artist)
|
||
results = search(f"{title} {artist}")
|
||
if not results:
|
||
results = search(title)
|
||
for r in results[:4]:
|
||
rt = fold(r['title'])
|
||
if not (tf in rt or rt in tf): # 文章标题与曲名应互相包含(如「嘘 (シドの曲)」)
|
||
continue
|
||
text = extract(r)
|
||
# 严格句式: テレビアニメ『X』之后80字内出现 OP/ED/主题曲 关键词
|
||
for m in re.finditer(
|
||
r'テレビアニメ[『「](.+?)[』」].{0,80}?(?:オープニング|エンディング|主題歌|テーマソング|テーマ曲)', text):
|
||
anime = m.group(1)
|
||
pre = text[max(0, m.start() - 80):m.start()]
|
||
years = re.findall(r'(\d{4})年', pre)
|
||
year = years[-1] if years else None
|
||
return anime, (f"{year}-01" if year else None), []
|
||
# 宽松句式: アニメ『X』 + 关键词
|
||
m2 = re.search(r'アニメ[『「](.+?)[』」].{0,80}?(?:オープニング|エンディング|主題歌|テーマ)', text)
|
||
if m2:
|
||
return m2.group(1), None, []
|
||
return None, None, []
|
||
|
||
|
||
def moegirl_lookup(title, artist):
|
||
"""萌娘百科: 支持日文/中文标题。先搜「曲名 歌手」,不足时再搜「intitle:歌手」。"""
|
||
|
||
def search_heads(query):
|
||
html = http_get_text(
|
||
"https://zh.moegirl.org.cn/index.php?title=Special:%E6%90%9C%E7%B4%A2&search="
|
||
+ urllib.parse.quote(query) + "&fulltext=1")
|
||
return re.findall(r'<div class="mw-search-result-heading"><a href="(/[^"]+)"[^>]*>(.*?)</a>', html, re.S)
|
||
|
||
def examine(heads):
|
||
for href, text in heads[:5]:
|
||
t = fold(re.sub(r'<[^>]+>', '', text)).strip()
|
||
if '歧义' in t or '消歧义' in t or 'disambiguation' in t:
|
||
continue
|
||
page = http_get_text("https://zh.moegirl.org.cn" + href)
|
||
# 强验证: 页面标题以曲名开头(允许带括号注释),或正文含 lang=ja 曲名标记
|
||
heading_ok = (t == tf or t.startswith(tf + '(') or t.startswith(tf + '('))
|
||
if not heading_ok and not jp_pat.search(page):
|
||
continue
|
||
# 提取动漫: 优先 meta 描述 "本条目介绍的是动画《X》的OP2"
|
||
m = re.search(r'动画《(.{1,40}?)》的', page)
|
||
if not m:
|
||
# 正文 "(是)动画《<a title="X">》的片头曲/片尾曲"
|
||
m = re.search(r'是(?:电视)?动画《\s*<a[^>]*title="([^"]+)"', page)
|
||
if not m and heading_ok:
|
||
# 分类兜底(仅当页面标题已确认是这首歌时): Category:X音乐
|
||
m = re.search(r'"title":"Category:([^"]{1,40})音乐"', page)
|
||
if m:
|
||
name = m.group(1).strip()
|
||
if name:
|
||
y = re.search(r'(?:发行|发售)(?:时间|日期)[^\d]{0,6}(\d{4})', page)
|
||
return name, (y.group(1) if y else None)
|
||
return None
|
||
|
||
tf, af = fold(title), fold(artist)
|
||
# 页面精确验证: 正文中的日文原名标记 <span lang="ja">曲名</span>
|
||
jp_pat = re.compile(r'lang="ja"[^>]*>\s*' + re.escape(title) + r'\s*</span>')
|
||
heads = search_heads(f"{title} {artist}")
|
||
result = examine(heads)
|
||
if result:
|
||
return result
|
||
if af: # 二次搜索: 按歌手名搜标题
|
||
heads = search_heads(f"intitle:{artist}")
|
||
result = examine(heads)
|
||
if result:
|
||
return result
|
||
return None, None, []
|
||
|
||
|
||
# ---------- MusicBrainz 歌手地区/成立时间 ----------
|
||
|
||
class MBError(Exception):
|
||
"""MusicBrainz 查询失败(网络/限流/无结果/歧义)。"""
|
||
|
||
|
||
def _parse_retry_after(v, default=2):
|
||
"""解析 Retry-After 头(秒数或 HTTP 日期),返回等待秒数(上限 30)。"""
|
||
if not v:
|
||
return default
|
||
try:
|
||
return max(1, min(30, int(v)))
|
||
except ValueError:
|
||
try:
|
||
dt = parsedate_to_datetime(v)
|
||
now = datetime.datetime.now(dt.tzinfo)
|
||
return max(1, min(30, (dt - now).total_seconds()))
|
||
except Exception:
|
||
return default
|
||
|
||
|
||
def mb_http_json(url, ua, timeout=10):
|
||
"""MusicBrainz 请求(必须自定义 UA,Python-urllib 默认 UA 会被封 IP)。
|
||
|
||
503(共享出口 IP 限流常见,Retry-After 常为 0)时做最多 2 次退避重试,
|
||
间隔 = max(Retry-After, 1.5s)——实测间歇性 503 下重试成功率较高。
|
||
"""
|
||
req = urllib.request.Request(url, headers={'User-Agent': ua})
|
||
for attempt in range(3):
|
||
try:
|
||
with urllib.request.urlopen(req, timeout=timeout, context=_SSL_CTX) as r:
|
||
return json.load(r)
|
||
except urllib.error.HTTPError as e:
|
||
if e.code == 503 and attempt < 2:
|
||
time.sleep(max(_parse_retry_after(e.headers.get('Retry-After')), 1.5))
|
||
continue
|
||
raise MBError(f'HTTP {e.code}') from e
|
||
except Exception as e:
|
||
raise MBError(str(e)[:80]) from e
|
||
|
||
|
||
def musicbrainz_artist_lookup(artist, ctx):
|
||
"""联网查歌手(写缓存),返回 {'name','type','country','begin','area'};失败抛 MBError。"""
|
||
cfg = ctx['cfg']
|
||
mb = ctx['mb_state']
|
||
# 自节流(官方限速 1 请求/秒)
|
||
wait = mb['last_ts'] + cfg['musicbrainz']['rate_limit'] - time.time()
|
||
if wait > 0:
|
||
time.sleep(wait)
|
||
url = ('https://musicbrainz.org/ws/2/artist/?query=artist%3A%22'
|
||
+ urllib.parse.quote(artist) + '%22&fmt=json&limit=10')
|
||
data = mb_http_json(url, cfg['musicbrainz']['user_agent'])
|
||
mb['last_ts'] = time.time()
|
||
artists = data.get('artists', [])
|
||
best = None
|
||
for a in artists:
|
||
if not a.get('name'):
|
||
continue
|
||
if best is None or a.get('score', 0) > best.get('score', 0):
|
||
best = a
|
||
if best is None:
|
||
raise MBError('无结果')
|
||
# 名称校验: 精确匹配或高分,避免同名歧义歌手
|
||
if fold(best['name']) != fold(artist) and best.get('score', 0) < 90:
|
||
raise MBError(f'歧义(最佳 {best["name"]},分数 {best.get("score", 0)})')
|
||
info = {
|
||
'name': best.get('name', artist),
|
||
'type': best.get('type') or '',
|
||
'country': best.get('country') or '',
|
||
'begin': (best.get('life-span') or {}).get('begin') or '',
|
||
'area': (best.get('area') or {}).get('name') or '',
|
||
}
|
||
if not info['country'] and (best.get('area') or {}).get('id'):
|
||
wait = mb['last_ts'] + cfg['musicbrainz']['rate_limit'] - time.time()
|
||
if wait > 0:
|
||
time.sleep(wait)
|
||
adata = mb_http_json('https://musicbrainz.org/ws/2/area/' + best['area']['id'] + '?fmt=json',
|
||
cfg['musicbrainz']['user_agent'])
|
||
mb['last_ts'] = time.time()
|
||
codes = adata.get('iso-3166-1-codes') or []
|
||
if codes:
|
||
info['country'] = codes[0]
|
||
ctx['artist_cache'][artist] = info
|
||
return info
|
||
|
||
|
||
def mb_classify(artist, ctx):
|
||
"""歌手地区 -> 分类键。缓存命中零请求;未启用(离线/停用/连续失败禁用)且无缓存 -> ('other', None)。"""
|
||
if not artist:
|
||
return 'other', None
|
||
if artist in ctx['artist_cache']:
|
||
ctx['source_use']['musicbrainz']['cache'] += 1
|
||
info = ctx['artist_cache'][artist]
|
||
elif ctx['mb_state']['enabled']:
|
||
info = musicbrainz_artist_lookup(artist, ctx) # 失败抛 MBError,由调用侧计数
|
||
ctx['source_use']['musicbrainz']['ok'] += 1
|
||
ctx['mb_state']['consecutive'] = 0 # 成功清零连续失败(间歇性限流不应累计误禁用)
|
||
else:
|
||
return 'other', None
|
||
cc = (info.get('country') or '').upper()
|
||
return MB_COUNTRY_CAT.get(cc, 'western' if cc else 'other'), info
|
||
|
||
|
||
# ---------- 控制台输出 ----------
|
||
|
||
COLOR = False
|
||
_ANSI = {'reset': '\033[0m', 'bold': '\033[1m',
|
||
'green': '\033[32m', 'yellow': '\033[33m', 'red': '\033[31m', 'cyan': '\033[36m'}
|
||
|
||
|
||
def setup_console():
|
||
"""Windows 开启 VT 处理以支持 ANSI 颜色;非终端/重定向/NO_COLOR 时保持纯文本。"""
|
||
global COLOR
|
||
if not sys.stdout.isatty() or os.environ.get('NO_COLOR'):
|
||
return
|
||
if os.name == 'nt':
|
||
try:
|
||
import ctypes
|
||
h = ctypes.windll.kernel32.GetStdHandle(-11)
|
||
mode = ctypes.c_uint32()
|
||
ctypes.windll.kernel32.GetConsoleMode(h, ctypes.byref(mode))
|
||
# ENABLE_VIRTUAL_TERMINAL_PROCESSING = 0x0004
|
||
ctypes.windll.kernel32.SetConsoleMode(h, mode.value | 0x0004)
|
||
except Exception:
|
||
return
|
||
COLOR = True
|
||
|
||
|
||
def c(name, text):
|
||
"""彩色文本(COLOR=False 时原样返回)。name: green/yellow/red/cyan/bold。"""
|
||
if not COLOR:
|
||
return text
|
||
return _ANSI.get(name, '') + text + _ANSI['reset']
|
||
|
||
|
||
def print_section(title):
|
||
print()
|
||
print(c('cyan', f'=== {title} ==='))
|
||
|
||
|
||
# ---------- 配置与缓存 ----------
|
||
|
||
def load_settings(tool_dir):
|
||
"""读取设置(设置.json 各段合并进默认值,损坏/缺失回退默认;--sources= 参数可临时覆盖)。"""
|
||
cfg = {
|
||
'decrypt': True, 'save_cover': True,
|
||
'output': '',
|
||
'categories': {'anisong': 'Anisong', 'jp_kr': '日韩', 'hk_mo_tw': '中国港澳台',
|
||
'cn_mainland': '中国大陆', 'western': '欧美', 'other': '其他(未被识别)'},
|
||
'folders': {'songs': '歌曲', 'originals': '加密原件', 'lyrics': '歌词', 'covers': '封面'},
|
||
'musicbrainz': {'enabled': True, 'rate_limit': 1.05,
|
||
'user_agent': 'AnisongOrganizer/1.0 ( contact )', 'max_failures': 3},
|
||
'sources': dict(DEFAULT_ENABLED),
|
||
}
|
||
settings_path = os.path.join(tool_dir, 'data', '设置.json')
|
||
try:
|
||
with open(settings_path, encoding='utf-8-sig') as fp:
|
||
data = json.load(fp)
|
||
for k in cfg['sources']:
|
||
s = (data.get('sources') or {}).get(k) or {}
|
||
if 'enabled' in s:
|
||
cfg['sources'][k] = bool(s['enabled'])
|
||
d = data.get('decrypt') or {}
|
||
if isinstance(d, dict):
|
||
if 'enabled' in d:
|
||
cfg['decrypt'] = bool(d['enabled'])
|
||
if 'save_cover' in d:
|
||
cfg['save_cover'] = bool(d['save_cover'])
|
||
if data.get('output'):
|
||
cfg['output'] = str(data['output'])
|
||
fo = data.get('folders') or {}
|
||
if isinstance(fo, dict):
|
||
for sub in cfg['folders']:
|
||
if fo.get(sub):
|
||
cfg['folders'][sub] = str(fo[sub])
|
||
ca = data.get('categories') or {}
|
||
if isinstance(ca, dict):
|
||
for k in cfg['categories']:
|
||
if ca.get(k):
|
||
cfg['categories'][k] = str(ca[k])
|
||
mb = data.get('musicbrainz') or {}
|
||
if isinstance(mb, dict):
|
||
if 'enabled' in mb:
|
||
cfg['musicbrainz']['enabled'] = bool(mb['enabled'])
|
||
if 'rate_limit' in mb:
|
||
cfg['musicbrainz']['rate_limit'] = float(mb['rate_limit'])
|
||
if 'user_agent' in mb:
|
||
cfg['musicbrainz']['user_agent'] = str(mb['user_agent'])
|
||
if 'max_failures' in mb:
|
||
cfg['musicbrainz']['max_failures'] = int(mb['max_failures'])
|
||
except Exception:
|
||
pass
|
||
for a in sys.argv: # --sources= 覆盖
|
||
if a.startswith('--sources='):
|
||
val = a.split('=', 1)[1].strip()
|
||
if val == 'all':
|
||
cfg['sources'] = {k: True for k in cfg['sources']}
|
||
elif val == 'off':
|
||
cfg['sources'] = {k: False for k in cfg['sources']}
|
||
else:
|
||
sel = [x.strip() for x in val.split(',') if x.strip() in cfg['sources']]
|
||
cfg['sources'] = {k: (k in sel) for k in cfg['sources']}
|
||
return cfg
|
||
|
||
|
||
def load_anime_map(tool_dir):
|
||
"""读取映射表,返回 (lookup, series_list)。
|
||
|
||
映射表 = 用户维护的 anime_map.json(种子 + ✎ 手动标注) + 识别缓存.json
|
||
(上次联网运行的识别结果,每次联网运行重新生成、不累积历史)。
|
||
lookup: 规范化曲名 -> (系列名, 最终曲名, 'seed'|'cache'),种子优先。
|
||
"""
|
||
lookup = {}
|
||
series_list = []
|
||
|
||
def load_one(path, kind):
|
||
with open(path, encoding='utf-8-sig') as fp:
|
||
sers = json.load(fp)['series']
|
||
for s in sers:
|
||
series_list.append(s)
|
||
for song in s['songs']:
|
||
for k in song.get('keys', [song['title']]):
|
||
key = normalize(k)
|
||
if kind == 'seed' or key not in lookup: # 种子优先,缓存只补缺
|
||
lookup[key] = (s['name'], song['title'], kind)
|
||
|
||
load_one(os.path.join(tool_dir, 'data', 'anime_map.json'), 'seed')
|
||
cache_path = os.path.join(tool_dir, 'data', '识别缓存.json')
|
||
if os.path.exists(cache_path):
|
||
try:
|
||
load_one(cache_path, 'cache')
|
||
except Exception:
|
||
pass
|
||
return lookup, series_list
|
||
|
||
|
||
def write_online_cache(tool_dir, recs):
|
||
"""本次运行的联网识别结果写入 识别缓存.json——每次联网运行重新生成,不累积历史。"""
|
||
hits = [(r['group'], r['title'], r['date']) for r in recs
|
||
if r['category'] == 'anisong'
|
||
and r['source'] in ('animethemes', 'anison', 'wikipedia', 'moegirl')]
|
||
p = os.path.join(tool_dir, 'data', '识别缓存.json')
|
||
if not hits:
|
||
try: # 无联网命中则清空旧缓存(不堆积过期对应关系)
|
||
if os.path.exists(p):
|
||
os.remove(p)
|
||
except OSError:
|
||
pass
|
||
return
|
||
series = []
|
||
for sn, title, date in hits:
|
||
s = next((x for x in series if x['name'] == sn), None)
|
||
if s is None:
|
||
s = {'name': sn, 'date': ((date or '') + '-01' if date else '9999-01-01'), 'songs': []}
|
||
series.append(s)
|
||
if not any(song['title'] == title for song in s['songs']):
|
||
s['songs'].append({'title': title, 'keys': [title]})
|
||
try:
|
||
with open(p, 'w', encoding='utf-8') as fp:
|
||
json.dump({'_说明': '联网识别结果缓存(每次联网运行自动重新生成,可安全删除)。',
|
||
'series': series}, fp, ensure_ascii=False, indent=2)
|
||
except Exception:
|
||
pass
|
||
|
||
|
||
def load_artist_map(tool_dir):
|
||
"""读取歌手缓存(容忍缺失/损坏)。"""
|
||
p = os.path.join(tool_dir, 'data', 'artist_map.json')
|
||
try:
|
||
with open(p, encoding='utf-8-sig') as fp:
|
||
return json.load(fp).get('artists', {})
|
||
except Exception:
|
||
return {}
|
||
|
||
|
||
def save_artist_map(tool_dir, artists):
|
||
"""保存歌手缓存(纯旁路元数据,识别阶段结束即存,离线复用)。"""
|
||
if not artists:
|
||
return
|
||
p = os.path.join(tool_dir, 'data', 'artist_map.json')
|
||
try:
|
||
with open(p, 'w', encoding='utf-8') as fp:
|
||
json.dump({'_说明': 'MusicBrainz 歌手查询结果缓存。country: ISO 国家码(JP/KR/HK/MO/TW/CN/US…,空=未查到);begin: 成立/出生年份(分类内排序用);type: Person/Group 等;area: 地区名。可手工编辑/预填,离线时仍生效。',
|
||
'artists': artists}, fp, ensure_ascii=False, indent=2)
|
||
except Exception:
|
||
pass
|
||
|
||
|
||
def write_anime_map(tool_dir, recs):
|
||
"""手动标注(✎)的动漫系列写入 anime_map.json(用户维护文件;联网结果改走 识别缓存.json)。"""
|
||
new_entries = [(r['group'], r['title']) for r in recs
|
||
if r['category'] == 'anisong' and r['source'] == '手动']
|
||
if not new_entries:
|
||
return
|
||
map_path = os.path.join(tool_dir, 'data', 'anime_map.json')
|
||
try:
|
||
with open(map_path, encoding='utf-8-sig') as fp:
|
||
map_data = json.load(fp)
|
||
all_series = map_data['series']
|
||
for sn, title in new_entries:
|
||
s = next((x for x in all_series if x['name'] == sn), None)
|
||
if s is None:
|
||
rec = next((r for r in recs if r['group'] == sn), None)
|
||
d = (rec or {}).get('date') or ''
|
||
date = f'{d}-01' if d else '9999-01-01'
|
||
s = {'name': sn, 'date': date, 'songs': []}
|
||
all_series.append(s)
|
||
if not any(song['title'] == title for song in s['songs']):
|
||
s['songs'].append({'title': title, 'keys': [title]})
|
||
with open(map_path, 'w', encoding='utf-8') as fp:
|
||
json.dump(map_data, fp, ensure_ascii=False, indent=2)
|
||
except Exception as e:
|
||
print(f'(缓存写入失败: {e})')
|
||
|
||
|
||
# ---------- 分类路径助手 ----------
|
||
|
||
def cat_dir(root, cfg, key):
|
||
return os.path.join(root, sanitize(cfg['categories'][key]))
|
||
|
||
|
||
def cat_sub(root, cfg, key, sub):
|
||
return os.path.join(cat_dir(root, cfg, key), sanitize(cfg['folders'][sub]))
|
||
|
||
|
||
# ---------- 流水线 ----------
|
||
|
||
def find_target_folder(tool_dir):
|
||
"""默认处理工具包上级文件夹;若上级没有音频/ncm 文件则处理工具包所在文件夹。"""
|
||
parent = os.path.dirname(tool_dir)
|
||
has_audio = lambda d: any(f.lower().endswith(PROC_EXTS) for f in os.listdir(d)) if os.path.isdir(d) else False
|
||
return parent if has_audio(parent) else tool_dir
|
||
|
||
|
||
def _walk_media(folder, cfg):
|
||
"""递归收集媒体文件相对路径(跳过分类目录与工具包文件夹)。"""
|
||
skip = {sanitize(cfg['categories'][k]) for k in CATEGORY_KEYS}
|
||
skip.add(os.path.basename(tool_root()))
|
||
out = []
|
||
for root, dirs, files in os.walk(folder):
|
||
dirs[:] = [d for d in dirs if d not in skip]
|
||
rel_root = os.path.relpath(root, folder)
|
||
for fn in files:
|
||
out.append(os.path.join(rel_root, fn) if rel_root != '.' else fn)
|
||
return out
|
||
|
||
|
||
def remove_empty_dirs(folder):
|
||
"""自底向上删除空子目录(递归整理后清理源文件夹的空壳),返回删除数。"""
|
||
removed = 0
|
||
for root, dirs, files in os.walk(folder, topdown=False):
|
||
for d in dirs:
|
||
try:
|
||
os.rmdir(os.path.join(root, d)) # 仅空目录可删除
|
||
removed += 1
|
||
except OSError:
|
||
pass
|
||
return removed
|
||
|
||
|
||
def collect_files(folder, cfg=None, recursive=False, lookup=None, metas=None):
|
||
"""收集音频文件,返回 REC 列表(src/raw/artist/title/ext 已填)。
|
||
recursive=True 时递归扫描子目录(网易云下载目录结构),src 为相对路径。
|
||
|
||
曲名/歌手解析优先级: FLAC 标签 > ncm 元数据(文件名只有歌曲名等格式时补全歌手)
|
||
> 文件名解析。文件名含「 - 」时双方向尝试(歌曲名-歌手 与 歌手-歌曲名),
|
||
优先取能命中本地映射表的方向(适配网易云「歌手-歌曲名」命名设置)。"""
|
||
if recursive:
|
||
candidates = sorted(p for p in _walk_media(folder, cfg)
|
||
if p.lower().endswith(AUDIO_EXTS))
|
||
else:
|
||
candidates = sorted(f for f in os.listdir(folder) if f.lower().endswith(AUDIO_EXTS))
|
||
recs = []
|
||
for fn in candidates:
|
||
tags = None
|
||
if fn.lower().endswith('.flac'):
|
||
tags = read_flac_tags(os.path.join(folder, fn))
|
||
meta = (metas or {}).get(fn)
|
||
raw = None
|
||
artist = ''
|
||
if tags is not None and tags.get('TITLE'):
|
||
raw = tags['TITLE'][0]
|
||
elif meta and meta.get('musicName'):
|
||
raw = meta['musicName']
|
||
if raw is None: # 文件名解析(去掉可能已有的编号前缀)
|
||
stem = re.sub(r'^\d+_', '', os.path.splitext(fn)[0])
|
||
if ' - ' in stem:
|
||
a, b = stem.split(' - ', 1)
|
||
# 双向尝试: 优先能命中本地映射表的解析(曲名-歌手 / 歌手-曲名)
|
||
if lookup and normalize(b) in lookup and normalize(a) not in lookup:
|
||
raw, artist = b, a.strip()
|
||
else:
|
||
raw, artist = a, b.strip()
|
||
else:
|
||
raw = stem
|
||
if tags is not None and tags.get('ARTIST'):
|
||
artist = tags['ARTIST'][0].split('/')[0].strip()
|
||
elif not artist and meta: # ncm 元数据补全歌手
|
||
arr = meta.get('artist') or []
|
||
if isinstance(arr, list) and arr:
|
||
first = arr[0]
|
||
artist = (first[0] if isinstance(first, list) else first).strip()
|
||
recs.append({'src': fn, 'orig_src': fn, 'raw': raw, 'artist': artist, 'title': normalize(raw),
|
||
'category': '', 'group': '', 'date': '', 'source': '',
|
||
'ext': os.path.splitext(fn)[1], 'ncm_pair': None,
|
||
'lrc_pair': None, 'cover': None, 'number': 0, 'final_name': ''})
|
||
return recs
|
||
|
||
|
||
def decrypt_ncm(folder, cfg, recursive=False):
|
||
"""解密 .ncm(幂等),返回 (ncm_pairs, covers, metas, msgs)。
|
||
recursive=True 时递归扫描子目录,键为相对路径。
|
||
ncm_pairs: 解密产物相对路径 -> 原 .ncm 相对路径;covers: 解密产物相对路径 -> (封面字节, 目标文件名);
|
||
metas: 解密产物相对路径 -> 元数据 dict(文件名为纯歌曲名等格式时用于补全歌手)。"""
|
||
ncm_pairs = {}
|
||
covers = {}
|
||
metas = {}
|
||
msgs = []
|
||
if recursive:
|
||
ncm_files = sorted(p for p in _walk_media(folder, cfg) if p.lower().endswith('.ncm'))
|
||
else:
|
||
ncm_files = sorted(f for f in os.listdir(folder) if f.lower().endswith('.ncm'))
|
||
if not ncm_files:
|
||
return ncm_pairs, covers, metas, msgs
|
||
try:
|
||
import ncm_decrypt
|
||
except ImportError:
|
||
if cfg['decrypt']:
|
||
msgs.append(f'检测到 {len(ncm_files)} 个 .ncm 文件,但缺少 ncm_decrypt.py 模块,本次跳过解密。')
|
||
return ncm_pairs, covers, metas, msgs
|
||
if not cfg['decrypt']:
|
||
msgs.append(f'检测到 {len(ncm_files)} 个 .ncm 文件(设置.json 中 decrypt.enabled=false,跳过解密)。')
|
||
return ncm_pairs, covers, metas, msgs
|
||
msgs.append(f'共 {len(ncm_files)} 个:')
|
||
for fn in ncm_files:
|
||
src = os.path.join(folder, fn)
|
||
try:
|
||
_, meta, cover, _ = ncm_decrypt.parse_ncm(src)
|
||
out_fn = os.path.splitext(fn)[0] + '.' + (meta.get('format') or 'mp3')
|
||
out_path = os.path.join(folder, out_fn)
|
||
if os.path.exists(out_path):
|
||
msgs.append(f' · {fn}: 已存在 {out_fn},跳过解密')
|
||
else:
|
||
ncm_decrypt.decrypt_audio(src, out_path)
|
||
msgs.append(f' {c("green", "✓")} {fn} -> {out_fn}')
|
||
ncm_pairs[out_fn] = fn
|
||
metas[out_fn] = meta
|
||
if cover and cfg['save_cover']:
|
||
name = sanitize(meta.get('album') or '') or sanitize(meta.get('musicName') or '')
|
||
if not name:
|
||
name = os.path.splitext(fn)[0]
|
||
cover_ext = '.png' if cover[:8] == b'\x89PNG\r\n\x1a\n' else '.jpg'
|
||
covers[out_fn] = (cover, sanitize(name) + cover_ext)
|
||
except Exception as e:
|
||
msgs.append(f' {c("red", "✗")} 解密失败 {fn}: {e}')
|
||
return ncm_pairs, covers, metas, msgs
|
||
|
||
|
||
def self_heal(root, cfg):
|
||
"""自愈: 各分类「加密原件」中缺少同茎解密音频的 .ncm 重新解密回「歌曲」夹。返回消息列表。"""
|
||
msgs = []
|
||
for key in CATEGORY_KEYS:
|
||
originals_dir = cat_sub(root, cfg, key, 'originals')
|
||
if not os.path.isdir(originals_dir):
|
||
continue
|
||
songs_dir = cat_sub(root, cfg, key, 'songs')
|
||
for fn in sorted(os.listdir(originals_dir)):
|
||
if not fn.lower().endswith('.ncm'):
|
||
continue
|
||
stem = os.path.splitext(fn)[0]
|
||
if not re.match(r'^\d+_', stem):
|
||
msgs.append(f' · {cfg["categories"][key]}\\{cfg["folders"]["originals"]}\\{fn}: 无编号前缀,跳过自愈')
|
||
continue
|
||
has_audio = any(
|
||
f.lower().startswith(stem.lower() + '.') and f.lower().endswith(AUDIO_EXTS)
|
||
for f in os.listdir(songs_dir)) if os.path.isdir(songs_dir) else False
|
||
if has_audio:
|
||
continue
|
||
src = os.path.join(originals_dir, fn)
|
||
try:
|
||
import ncm_decrypt
|
||
_, meta, cover, _ = ncm_decrypt.parse_ncm(src)
|
||
fmt = meta.get('format') or 'mp3'
|
||
dst = os.path.join(songs_dir, stem + '.' + fmt)
|
||
ncm_decrypt.decrypt_audio(src, dst)
|
||
msgs.append(f' {c("green", "✓")} {cfg["categories"][key]}\\{cfg["folders"]["originals"]}\\{fn} '
|
||
f'缺少解密音频,已重新解密 -> {cfg["folders"]["songs"]}\\{stem}.{fmt}')
|
||
except Exception as e:
|
||
msgs.append(f' {c("red", "✗")} 自愈失败 {fn}: {e}')
|
||
return msgs
|
||
|
||
|
||
def _query_source(name, key, artist):
|
||
"""单个源的查询(在后台线程中执行),返回 (name, 'ok', anime, date, cands) 或 (name, 'err', msg)。"""
|
||
try:
|
||
anime, date, cands = globals()[SOURCES[name][0]](key, artist)
|
||
return name, 'ok', anime, date, cands
|
||
except Exception as e:
|
||
return name, 'err', str(e)[:80]
|
||
|
||
|
||
def identify(recs, ctx, progress_cb=None, stop_check=None):
|
||
"""逐文件识别分类,返回联网动漫命中数。
|
||
|
||
动漫源策略: 按曲名语言调整优先级(动漫歌曲独特适配),所有启用源并行查询
|
||
(墙钟时间 = 最慢源而非各源之和),按优先级取首个命中;源连续失败达
|
||
SOURCE_MAX_FAILS 后本次运行熔断停用(无梯子时不再每首歌白等超时)。
|
||
"""
|
||
import concurrent.futures
|
||
lookup = ctx['lookup']
|
||
series_list = ctx['series_list']
|
||
enabled = ctx['enabled']
|
||
online_hits = 0
|
||
mb = ctx['mb_state']
|
||
broken = {n: 0 for n in SOURCES}
|
||
total = len(recs)
|
||
executor = concurrent.futures.ThreadPoolExecutor(max_workers=len(SOURCES))
|
||
try:
|
||
for i, rec in enumerate(recs, 1):
|
||
if stop_check and stop_check():
|
||
break
|
||
key = rec['title']
|
||
result = ''
|
||
# 1) 本地映射表(「其他(非动漫)」系列/未知日期不算动漫,走地区分类)
|
||
if key in lookup:
|
||
series_name, map_title, kind = lookup[key]
|
||
s_entry = next((s for s in series_list if s['name'] == series_name), None)
|
||
s_date = s_entry.get('date', '9999-01-01') if s_entry else '9999-01-01'
|
||
if not s_date.startswith('9999') and series_name != '其他(非动漫)':
|
||
rec['category'] = 'anisong'
|
||
rec['group'] = series_name
|
||
rec['title'] = map_title
|
||
rec['date'] = s_date[:7]
|
||
rec['source'] = '本地' if kind == 'seed' else '缓存'
|
||
result = series_name
|
||
else:
|
||
# 用户已确认非动漫(其他(非动漫)系列/未知日期): 不再查询动漫源,
|
||
# 直接地区分类(歌手缓存命中零网络,避免固定几个文件每次都白等)
|
||
rec['skip_anime'] = True
|
||
# 2) 联网动漫源: 并行查询,按语言优先级取首个命中
|
||
if not rec['category'] and not rec.get('skip_anime'):
|
||
priority = SOURCE_PRIORITY[_title_lang(key)]
|
||
actives = [n for n in priority if enabled.get(n) and broken[n] < SOURCE_MAX_FAILS]
|
||
hit = None
|
||
if actives:
|
||
futures = {executor.submit(_query_source, n, key, rec['artist']): n
|
||
for n in actives}
|
||
outcomes = {n: f.result() for f, n in futures.items()} # 并发等待(墙钟=最慢源)
|
||
for n in priority:
|
||
if n not in outcomes:
|
||
continue
|
||
_, status, *rest = outcomes[n]
|
||
if status == 'ok':
|
||
ctx['source_use'][n]['ok'] += 1
|
||
broken[n] = 0 # 成功清零连续失败
|
||
# 防误识别: 动漫名与歌手名相同多半是抓取错位(如把歌手名当作品名),视为未命中
|
||
if rest[0] and not (rec['artist'] and fold(rest[0]) == fold(rec['artist'])):
|
||
hit = (n, rest[0], rest[1])
|
||
break
|
||
else:
|
||
ctx['source_use'][n]['fail'] += 1
|
||
broken[n] += 1
|
||
if ctx['source_use'][n]['err'] is None:
|
||
ctx['source_use'][n]['err'] = rest[0]
|
||
if hit:
|
||
name, anime, date = hit
|
||
rec['category'] = 'anisong'
|
||
rec['group'] = anime
|
||
rec['date'] = (date or '')[:7]
|
||
rec['source'] = name
|
||
online_hits += 1
|
||
result = anime
|
||
# 并入系列表(供排序与缓存写回)
|
||
s = next((x for x in series_list if x['name'] == anime), None)
|
||
if s is None:
|
||
d = (date or '')[:7] # 日期规范化,防 'YYYY-MM-DD-01' 畸形值
|
||
series_list.append({'name': anime, 'date': (d + '-01' if d else '9999-01-01'),
|
||
'songs': []})
|
||
s = series_list[-1]
|
||
if not any(song['title'] == rec['title'] for song in s['songs']):
|
||
s['songs'].append({'title': rec['title'], 'keys': [rec['title']]})
|
||
# 3) MusicBrainz 地区分类兜底
|
||
if not rec['category']:
|
||
cat = 'other'
|
||
if rec['artist']:
|
||
try:
|
||
cat, info = mb_classify(rec['artist'], ctx)
|
||
except MBError as e:
|
||
cat = 'other'
|
||
ctx['source_use']['musicbrainz']['fail'] += 1
|
||
mb['consecutive'] += 1
|
||
if mb['err'] is None:
|
||
mb['err'] = str(e)[:80]
|
||
if mb['consecutive'] >= ctx['cfg']['musicbrainz']['max_failures']:
|
||
mb['enabled'] = False
|
||
except Exception as e:
|
||
cat = 'other'
|
||
ctx['source_use']['musicbrainz']['fail'] += 1
|
||
mb['consecutive'] += 1
|
||
if mb['err'] is None:
|
||
mb['err'] = str(e)[:80]
|
||
else:
|
||
if cat != 'other':
|
||
rec['source'] = 'musicbrainz'
|
||
rec['group'] = rec['artist']
|
||
rec['date'] = (info or {}).get('begin') or ''
|
||
result = ctx['cfg']['categories'][cat]
|
||
rec['category'] = cat
|
||
if not result:
|
||
result = ctx['cfg']['categories'][rec['category']]
|
||
if not rec['group']:
|
||
rec['group'] = rec['artist'] or ''
|
||
if progress_cb:
|
||
label = f'{rec["raw"]} - {rec["artist"]}' if rec['artist'] else rec['raw']
|
||
# 附带结构化信息(供 Web 界面罗列: 分类/系列/组建年份)
|
||
progress_cb('identify', i, total, label, result,
|
||
{'category': rec['category'], 'group': rec['group'],
|
||
'date': rec['date'] or ''})
|
||
time.sleep(0.3) # 对接口保持礼貌
|
||
finally:
|
||
executor.shutdown(wait=False)
|
||
ctx['broken'] = broken # 供源状态汇总显示熔断情况
|
||
return online_hits
|
||
|
||
|
||
def group_and_order(recs, ctx):
|
||
"""按分类分组并排序,返回 {cat_key: [REC]}。
|
||
Anisong: 系列按首播日期排序,系列内按映射表歌曲顺序;
|
||
日韩/中国港澳台/中国大陆/欧美: 按歌手分组(同歌手相邻),歌手按成立年份(未知置后)排序,组内按曲名;
|
||
其他: 按 (歌手, 曲名)。"""
|
||
by_cat = {k: [] for k in CATEGORY_KEYS}
|
||
for rec in recs:
|
||
by_cat[rec['category']].append(rec)
|
||
|
||
# Anisong: 系列排序
|
||
series_list = ctx['series_list']
|
||
series_list.sort(key=lambda s: (s.get('date', '9999-01-01')[:7], s['name']))
|
||
consumed = set()
|
||
ordered = []
|
||
for s in series_list:
|
||
for song in s['songs']:
|
||
for i, rec in enumerate(by_cat['anisong']):
|
||
if i not in consumed and rec['title'] == song['title'] and rec['group'] == s['name']:
|
||
ordered.append(rec)
|
||
consumed.add(i)
|
||
for i, rec in enumerate(by_cat['anisong']):
|
||
if i not in consumed:
|
||
ordered.append(rec)
|
||
by_cat['anisong'] = ordered
|
||
|
||
# 地区分类: 歌手分组 + 成立年份排序
|
||
def artist_begin(artist):
|
||
return ((ctx['artist_cache'].get(artist) or {}).get('begin') or '').strip()
|
||
|
||
for key in ('jp_kr', 'hk_mo_tw', 'cn_mainland', 'western'):
|
||
groups = {}
|
||
for rec in by_cat[key]:
|
||
groups.setdefault(fold(rec['group']), []).append(rec)
|
||
for gfold in groups:
|
||
groups[gfold].sort(key=lambda r: fold(r['title']))
|
||
def sort_key(gfold):
|
||
begin = artist_begin(groups[gfold][0]['group'])
|
||
return (not begin, begin, gfold)
|
||
by_cat[key] = [r for gfold in sorted(groups, key=sort_key) for r in groups[gfold]]
|
||
|
||
by_cat['other'].sort(key=lambda r: (fold(r['group']), fold(r['title'])))
|
||
return by_cat
|
||
|
||
|
||
def scan_existing_numbers(songs_dir):
|
||
"""读取歌曲夹现有最大编号(无文件/夹不存在返回 0)。"""
|
||
if not os.path.isdir(songs_dir):
|
||
return 0
|
||
mx = 0
|
||
for fn in os.listdir(songs_dir):
|
||
m = re.match(r'^(\d+)_', fn)
|
||
if m:
|
||
mx = max(mx, int(m.group(1)))
|
||
return mx
|
||
|
||
|
||
def assign_numbers(by_cat, root, cfg):
|
||
"""每分类独立编号(续接现有最大号),lrc/ncm 目标名取自 final_name 茎(永远同号)。"""
|
||
for key in CATEGORY_KEYS:
|
||
recs = by_cat[key]
|
||
if not recs:
|
||
continue
|
||
songs_dir = cat_sub(root, cfg, key, 'songs')
|
||
start = scan_existing_numbers(songs_dir) + 1
|
||
width = max(2, len(str(start + len(recs) - 1)))
|
||
used = {fn.lower() for fn in os.listdir(songs_dir)} if os.path.isdir(songs_dir) else set()
|
||
for i, rec in enumerate(recs):
|
||
base = sanitize(rec['title'])
|
||
n = 1
|
||
name = base
|
||
while f'{name}{rec["ext"]}'.lower() in used:
|
||
n += 1
|
||
name = f'{base} ({n})'
|
||
rec['final_name'] = f'{start + i:0{width}d}_{name}{rec["ext"]}'
|
||
rec['number'] = start + i
|
||
used.add(rec['final_name'].lower())
|
||
return by_cat
|
||
|
||
|
||
def build_move_plan(by_cat, ctx):
|
||
"""构建移动清单,返回 (moves, skips, warnings)。只对含移动条目的分类创建目录(执行时)。"""
|
||
folder = ctx['folder']
|
||
root = ctx['root']
|
||
cfg = ctx['cfg']
|
||
moves = []
|
||
warnings = []
|
||
cat_covers_used = {k: set() for k in CATEGORY_KEYS}
|
||
|
||
for key in CATEGORY_KEYS:
|
||
for rec in by_cat[key]:
|
||
stem = os.path.splitext(rec['final_name'])[0]
|
||
moves.append({'kind': 'audio', 'src': os.path.join(folder, rec['src']),
|
||
'dst': os.path.join(cat_sub(root, cfg, key, 'songs'), rec['final_name']),
|
||
'category': key, 'number': rec['number'], 'rec_src': rec['src']})
|
||
if rec.get('ncm_pair'):
|
||
moves.append({'kind': 'ncm', 'src': os.path.join(folder, rec['ncm_pair']),
|
||
'dst': os.path.join(cat_sub(root, cfg, key, 'originals'), stem + '.ncm'),
|
||
'category': key, 'number': rec['number'], 'rec_src': rec['src']})
|
||
old_lrc = os.path.splitext(rec['src'])[0] + '.lrc'
|
||
if os.path.exists(os.path.join(folder, old_lrc)):
|
||
rec['lrc_pair'] = old_lrc
|
||
moves.append({'kind': 'lrc', 'src': os.path.join(folder, old_lrc),
|
||
'dst': os.path.join(cat_sub(root, cfg, key, 'lyrics'), stem + '.lrc'),
|
||
'category': key, 'number': rec['number'], 'rec_src': rec['src']})
|
||
if rec.get('cover'):
|
||
cbytes, cname = rec['cover']
|
||
cdst = os.path.join(cat_sub(root, cfg, key, 'covers'), cname)
|
||
if cname in cat_covers_used[key] or os.path.exists(cdst):
|
||
continue # 分类内同名封面去重
|
||
cat_covers_used[key].add(cname)
|
||
moves.append({'kind': 'cover', 'src': '', 'dst': cdst,
|
||
'category': key, 'number': 0, 'data': cbytes, 'rec_src': rec['src']})
|
||
|
||
# 旧歌词文件夹(v3.1 结构)迁移: 去编号 + 规范化匹配曲名,唯一匹配才迁移
|
||
old_lyrics_dir = os.path.join(folder, cfg['folders']['lyrics'])
|
||
if os.path.isdir(old_lyrics_dir):
|
||
for lfn in sorted(os.listdir(old_lyrics_dir)):
|
||
if not lfn.lower().endswith('.lrc'):
|
||
continue
|
||
lstem = re.sub(r'^\d+_', '', os.path.splitext(lfn)[0])
|
||
lkey = fold(normalize(lstem))
|
||
hits = []
|
||
for key in CATEGORY_KEYS:
|
||
for rec in by_cat[key]:
|
||
if rec.get('lrc_pair') is None:
|
||
src_stem = re.sub(r'^\d+_', '', os.path.splitext(rec['src'])[0])
|
||
if lkey == fold(rec['title']) or lkey == fold(normalize(src_stem)):
|
||
hits.append((key, rec))
|
||
if len(hits) == 1:
|
||
key, rec = hits[0]
|
||
rec['lrc_pair'] = lfn
|
||
moves.append({'kind': 'lrc', 'src': os.path.join(old_lyrics_dir, lfn),
|
||
'dst': os.path.join(cat_sub(root, cfg, key, 'lyrics'),
|
||
os.path.splitext(rec['final_name'])[0] + '.lrc'),
|
||
'category': key, 'number': rec['number'], 'rec_src': rec['src']})
|
||
else:
|
||
warnings.append(f'旧歌词文件夹中 {lfn} 无法唯一对应歌曲,保持原位。')
|
||
|
||
# 旧顶层「封面」文件夹: 无法可靠映射分类,保持原位
|
||
old_covers_dir = os.path.join(folder, cfg['folders']['covers'])
|
||
if os.path.isdir(old_covers_dir):
|
||
warnings.append(f'检测到旧版顶层「{cfg["folders"]["covers"]}」文件夹:'
|
||
f'封面无法可靠对应分类,保持原位不动。')
|
||
|
||
# 目标已存在/重复目标/原地不动 -> 跳过
|
||
final_moves, skips = [], []
|
||
seen = set()
|
||
for m in moves:
|
||
if m['kind'] == 'cover':
|
||
final_moves.append(m)
|
||
continue
|
||
if os.path.normcase(m['src']) == os.path.normcase(m['dst']):
|
||
skips.append({**m, 'reason': '已是最新位置'})
|
||
elif os.path.exists(m['dst']):
|
||
skips.append({**m, 'reason': '目标已存在'})
|
||
elif m['dst'] in seen:
|
||
skips.append({**m, 'reason': '重复目标'})
|
||
else:
|
||
seen.add(m['dst'])
|
||
final_moves.append(m)
|
||
return final_moves, skips, warnings
|
||
|
||
|
||
def render_plan(by_cat, moves, ctx, absolute=False):
|
||
"""渲染预览文本。absolute=True 输出完整绝对路径(撤销日志用)。"""
|
||
root = ctx['root']
|
||
folder = ctx['folder']
|
||
cfg = ctx['cfg']
|
||
lines = []
|
||
for key in CATEGORY_KEYS:
|
||
recs = by_cat[key]
|
||
if not recs:
|
||
continue
|
||
existing = scan_existing_numbers(cat_sub(root, cfg, key, 'songs'))
|
||
lines.append('')
|
||
lines.append(f'【{cfg["categories"][key]}】(现有 {existing} 首,新增 {len(recs)} 首)')
|
||
for rec in recs:
|
||
mark = {'本地': '', '缓存': '[缓存] ', 'animethemes': '[网] ', 'anison': '[网] ',
|
||
'wikipedia': '[网] ', 'moegirl': '[网] ', 'musicbrainz': '[MB] ',
|
||
'手动': '[手动] '}.get(rec['source'], '')
|
||
if absolute:
|
||
lines.append(f' {os.path.join(cat_sub(root, cfg, key, "songs"), rec["final_name"])}'
|
||
f' ← {os.path.join(folder, rec["src"])}')
|
||
else:
|
||
rel = os.path.join(cfg['categories'][key], cfg['folders']['songs'], rec['final_name'])
|
||
lines.append(f' {mark}{rel} ← {rec["src"]}')
|
||
ncm_n = sum(1 for m in moves if m['category'] == key and m['kind'] == 'ncm')
|
||
lrc_n = sum(1 for m in moves if m['category'] == key and m['kind'] == 'lrc')
|
||
cover_n = sum(1 for m in moves if m['category'] == key and m['kind'] == 'cover')
|
||
note = []
|
||
if ncm_n:
|
||
note.append(f'{cfg["folders"]["originals"]} {ncm_n}')
|
||
if lrc_n:
|
||
note.append(f'{cfg["folders"]["lyrics"]} {lrc_n}')
|
||
if cover_n:
|
||
note.append(f'{cfg["folders"]["covers"]} {cover_n}')
|
||
if note:
|
||
lines.append(f' 同步: ' + ', '.join(note) + '(编号与歌曲一致)')
|
||
if absolute: # 撤销日志: 附上非音频移动明细
|
||
sub = [m for m in moves if m['kind'] in ('ncm', 'lrc', 'cover')]
|
||
if sub:
|
||
lines.append('')
|
||
lines.append('=== 移动明细(撤销日志) ===')
|
||
for m in sub:
|
||
if m['kind'] == 'cover':
|
||
lines.append(f' {m["dst"]} ← (解密提取的封面)')
|
||
else:
|
||
lines.append(f' {m["dst"]} ← {m["src"]}')
|
||
return '\n'.join(lines).strip() or '无需处理。'
|
||
|
||
|
||
def _src_mark(src):
|
||
"""导览中的来源标记。"""
|
||
return {'本地': '', '缓存': '[缓存] ', 'animethemes': '[网] ', 'anison': '[网] ',
|
||
'wikipedia': '[网] ', 'moegirl': '[网] ', 'musicbrainz': '[MB] ',
|
||
'手动': '[手动] '}.get(src, '')
|
||
|
||
|
||
def render_guide(result):
|
||
"""渲染导览——还原初代 预览.txt 的简洁格式(带领用户按时间顺序回顾):
|
||
|
||
共 N 个文件,按动漫系列(首播年份)排列:
|
||
|
||
【犬夜叉】(2000)
|
||
09_疾風の如く.flac ← 疾风の如く - 和田薫.flac
|
||
【日韩】
|
||
10_雪のツバサ.flac ← 雪のツバサ - redballoon.flac
|
||
...
|
||
"""
|
||
by_cat = result['by_cat']
|
||
cfg = result['ctx']['cfg']
|
||
lines = [f'共 {len(result["recs"])} 个文件,按动漫系列(首播年份)排列:']
|
||
current = None
|
||
# Anisong: by_cat 已按系列首播年份排序,系列头带年份
|
||
for rec in by_cat['anisong']:
|
||
sn = rec['group'] or '未命名系列'
|
||
if sn != current:
|
||
entry = next((s for s in result['ctx']['series_list'] if s['name'] == sn), None)
|
||
d = (entry.get('date') or '') if entry else ''
|
||
year = d[:4] if d and not d.startswith('9999') else ''
|
||
lines.append('\n【' + sn + '】' + (f'({year})' if year else ''))
|
||
current = sn
|
||
src_name = rec.get('orig_src') or rec['src'] # 应用后仍显示最初的来源名
|
||
lines.append(f' {_src_mark(rec["source"])}{rec["final_name"]} ← {src_name}')
|
||
# 其余分类: 分类头 + 歌曲行
|
||
for key in CATEGORY_KEYS[1:]:
|
||
if not by_cat[key]:
|
||
continue
|
||
lines.append('\n【' + cfg['categories'][key] + '】')
|
||
for rec in by_cat[key]:
|
||
src_name = rec.get('orig_src') or rec['src']
|
||
lines.append(f' {_src_mark(rec["source"])}{rec["final_name"]} ← {src_name}')
|
||
for w in result['ctx']['warnings']:
|
||
lines.append(f'⚠ {w}')
|
||
return '\n'.join(lines).strip() or '无需处理。'
|
||
|
||
|
||
def write_guide(tool_dir, text):
|
||
"""导览写入工具目录的 导览.txt,返回文件路径。"""
|
||
p = os.path.join(tool_dir, 'data', '导览.txt')
|
||
try:
|
||
with open(p, 'w', encoding='utf-8-sig') as fp:
|
||
fp.write(text + '\n')
|
||
except Exception:
|
||
pass
|
||
return p
|
||
|
||
|
||
def move_file(src, dst):
|
||
"""移动文件: 同盘 os.rename,跨盘回退 shutil.move(复制+删除)。"""
|
||
try:
|
||
os.rename(src, dst)
|
||
except OSError:
|
||
import shutil
|
||
shutil.move(src, dst)
|
||
|
||
|
||
def execute_moves(moves, progress_cb=None, stop_check=None):
|
||
"""执行移动清单,返回 {'moved','skipped','failed','stopped'}。
|
||
|
||
每条 move 会被打上结果标记: m['done']=True(成功) / False(跳过或失败),
|
||
m['error']=失败原因——供 Web 端把新位置回写记录、支持继续编辑再次应用。
|
||
"""
|
||
stats = {'moved': 0, 'skipped': 0, 'failed': 0, 'stopped': False}
|
||
for m in moves:
|
||
if stop_check and stop_check():
|
||
stats['stopped'] = True
|
||
break
|
||
dst = m['dst']
|
||
if m['kind'] == 'cover':
|
||
try:
|
||
os.makedirs(os.path.dirname(dst), exist_ok=True)
|
||
with open(dst, 'wb') as f:
|
||
f.write(m['data'])
|
||
stats['moved'] += 1
|
||
m['done'] = True
|
||
except OSError as e:
|
||
print(f'封面写入失败: {os.path.basename(dst)} ({e})')
|
||
stats['failed'] += 1
|
||
m['done'] = False
|
||
m['error'] = str(e)[:80]
|
||
continue
|
||
if os.path.exists(dst):
|
||
print(f'跳过(目标已存在): {dst}')
|
||
stats['skipped'] += 1
|
||
m['done'] = False
|
||
m['error'] = '目标已存在'
|
||
continue
|
||
try:
|
||
os.makedirs(os.path.dirname(dst), exist_ok=True)
|
||
move_file(m['src'], dst)
|
||
stats['moved'] += 1
|
||
m['done'] = True
|
||
if progress_cb:
|
||
progress_cb('apply', stats['moved'], len(moves), os.path.basename(dst))
|
||
except OSError as e:
|
||
print(f'移动失败: {os.path.basename(m["src"])} -> {os.path.basename(dst)} ({e})')
|
||
stats['failed'] += 1
|
||
m['done'] = False
|
||
m['error'] = str(e)[:80]
|
||
return stats
|
||
|
||
|
||
def print_source_summary(ctx):
|
||
"""运行结束的源状态汇总(动漫源 + MusicBrainz,含熔断提示)。"""
|
||
broken = ctx.get('broken', {})
|
||
for name, (_, needs_proxy) in SOURCES.items():
|
||
u = ctx['source_use'][name]
|
||
if ctx['enabled'].get(name) and u['fail']:
|
||
hint = '该源需要加速器/代理,请确认已开启后重试' if needs_proxy else '网络异常或服务不可用'
|
||
if broken.get(name, 0) >= SOURCE_MAX_FAILS:
|
||
hint += f'(连续失败 {SOURCE_MAX_FAILS} 次,本次运行已停用)'
|
||
print(f' ✗ 数据源 {name} 连接失败 {u["fail"]} 次 —— {hint}')
|
||
elif ctx['enabled'].get(name) and u['ok']:
|
||
tag = '(需代理)' if needs_proxy else ''
|
||
print(f' ✓ 数据源 {name} 正常 {tag}')
|
||
mb = ctx['source_use']['musicbrainz']
|
||
if mb['fail'] and not ctx['mb_state']['enabled']:
|
||
print(f' ✗ MusicBrainz 连接失败 {mb["fail"]} 次 —— 该源需要加速器/代理,请确认已开启后重试'
|
||
f'(连续失败 {ctx["cfg"]["musicbrainz"]["max_failures"]} 次,本次运行已停用)')
|
||
elif mb['fail']:
|
||
print(f' ⚠ MusicBrainz 部分失败 {mb["fail"]} 次,对应歌曲已归入'
|
||
f'「{ctx["cfg"]["categories"]["other"]}」')
|
||
elif mb['ok'] or mb['cache']:
|
||
print(f' ✓ MusicBrainz 正常(联网 {mb["ok"]},缓存 {mb["cache"]})')
|
||
|
||
|
||
def run_pipeline(folder, opts, progress_cb=None, stop_check=None):
|
||
"""完整流水线(不执行移动),返回 RESULT。CLI 与 Web 共用。
|
||
|
||
opts: {'offline': bool, 'recursive': bool, 'output': 输出根目录覆盖设置.json}
|
||
progress_cb(phase, i, n, text): phase='identify' 逐文件;'decrypt'/'heal' 单行消息。
|
||
"""
|
||
tool_dir = tool_root()
|
||
cfg = load_settings(tool_dir)
|
||
if opts.get('offline'):
|
||
cfg['sources'] = {k: False for k in cfg['sources']}
|
||
cfg['musicbrainz']['enabled'] = False
|
||
recursive = bool(opts.get('recursive'))
|
||
if opts.get('output'):
|
||
root = os.path.abspath(opts['output'])
|
||
elif cfg['output']:
|
||
root = os.path.abspath(cfg['output'])
|
||
else:
|
||
root = folder
|
||
|
||
map_path = os.path.join(tool_dir, 'data', 'anime_map.json')
|
||
if not os.path.exists(map_path):
|
||
raise FileNotFoundError(f'未找到映射文件: {map_path}')
|
||
lookup, series_list = load_anime_map(tool_dir)
|
||
artist_cache = load_artist_map(tool_dir)
|
||
source_use = {n: {'ok': 0, 'fail': 0, 'err': None} for n in SOURCE_ORDER}
|
||
source_use['musicbrainz'] = {'ok': 0, 'fail': 0, 'cache': 0, 'err': None}
|
||
ctx = {'tool_dir': tool_dir, 'folder': folder, 'root': root, 'cfg': cfg,
|
||
'lookup': lookup, 'series_list': series_list, 'artist_cache': artist_cache,
|
||
'enabled': dict(cfg['sources']), 'source_use': source_use,
|
||
'mb_state': {'enabled': cfg['musicbrainz']['enabled'], 'consecutive': 0,
|
||
'last_ts': 0.0, 'err': None},
|
||
'warnings': []}
|
||
|
||
def emit(text, phase='decrypt'):
|
||
if progress_cb:
|
||
progress_cb(phase, 0, 0, text)
|
||
|
||
def emit_section(title):
|
||
if progress_cb:
|
||
progress_cb('section', 0, 0, title)
|
||
|
||
ncm_pairs, covers, ncm_metas, d_msgs = decrypt_ncm(folder, cfg, recursive)
|
||
h_msgs = self_heal(root, cfg)
|
||
if d_msgs:
|
||
emit_section('② ncm 解密')
|
||
for m in d_msgs:
|
||
emit(m, 'decrypt')
|
||
if h_msgs:
|
||
emit_section('③ 自愈检查')
|
||
for m in h_msgs:
|
||
emit(m, 'heal')
|
||
recs = collect_files(folder, cfg, recursive, lookup, ncm_metas)
|
||
for rec in recs:
|
||
if rec['src'] in ncm_pairs:
|
||
rec['ncm_pair'] = ncm_pairs[rec['src']]
|
||
if rec['src'] in covers:
|
||
rec['cover'] = covers[rec['src']]
|
||
if recs:
|
||
emit_section('④ 识别')
|
||
online_hits = identify(recs, ctx, progress_cb, stop_check)
|
||
if not opts.get('offline'): # 联网运行: 重新生成识别缓存(不累积历史)
|
||
write_online_cache(tool_dir, recs)
|
||
by_cat = group_and_order(recs, ctx)
|
||
assign_numbers(by_cat, root, cfg)
|
||
moves, skips, warnings = build_move_plan(by_cat, ctx)
|
||
ctx['warnings'].extend(warnings)
|
||
save_artist_map(tool_dir, artist_cache)
|
||
stats = {k: {'moves': 0, 'skips': 0} for k in CATEGORY_KEYS}
|
||
for m in moves:
|
||
stats[m['category']]['moves'] += 1
|
||
for s in skips:
|
||
stats[s['category']]['skips'] += 1
|
||
return {'ctx': ctx, 'recs': recs, 'by_cat': by_cat, 'moves': moves,
|
||
'skips': skips, 'online_hits': online_hits, 'stats': stats}
|
||
|
||
|
||
def main():
|
||
if hasattr(sys.stdout, 'reconfigure'):
|
||
sys.stdout.reconfigure(encoding='utf-8', errors='replace')
|
||
setup_console()
|
||
|
||
if '--web' in sys.argv:
|
||
try:
|
||
import web_server
|
||
except ImportError:
|
||
print('缺少 web_server.py 模块(Web 界面未随本版本提供)。')
|
||
return 1
|
||
folder = next((a for a in sys.argv[1:] if os.path.isdir(a)), None)
|
||
web_server.start(tool_root(), folder)
|
||
return 0
|
||
|
||
apply_now = '--apply' in sys.argv
|
||
offline = '--offline' in sys.argv
|
||
recursive = '--recursive' in sys.argv
|
||
tool_dir = tool_root()
|
||
folder = next((a for a in sys.argv[1:] if os.path.isdir(a)), None)
|
||
if folder is None:
|
||
folder = find_target_folder(tool_dir)
|
||
if not os.path.isdir(folder):
|
||
print(f'文件夹不存在: {folder}')
|
||
return 1
|
||
|
||
print_section('① 扫描')
|
||
if recursive:
|
||
cfg0 = load_settings(tool_dir)
|
||
media = _walk_media(folder, cfg0)
|
||
audio_n = sum(1 for p in media if p.lower().endswith(AUDIO_EXTS))
|
||
ncm_n = sum(1 for p in media if p.lower().endswith('.ncm'))
|
||
print(f' 递归发现 {audio_n} 个音乐文件、{ncm_n} 个 .ncm(处理文件夹: {folder},含子目录)')
|
||
else:
|
||
audio_n = sum(1 for f in os.listdir(folder) if f.lower().endswith(AUDIO_EXTS))
|
||
ncm_n = sum(1 for f in os.listdir(folder) if f.lower().endswith('.ncm'))
|
||
print(f' 发现 {audio_n} 个音乐文件、{ncm_n} 个 .ncm(处理文件夹: {folder})')
|
||
# 网易云 VIP 曲目默认存入 VipSongsDownload 子文件夹,提示用户开递归
|
||
vip = os.path.join(folder, 'VipSongsDownload')
|
||
if os.path.isdir(vip) and any(f.lower().endswith(PROC_EXTS) for f in os.listdir(vip)):
|
||
print(c('yellow', ' ⚠ 检测到 VipSongsDownload 文件夹(网易云 VIP 曲目存放处),'
|
||
'本次未包含——请加 --recursive 或勾选「递归整理子目录」。'))
|
||
|
||
def progress_bridge(phase, *args):
|
||
"""CLI 进度回调: 区块标题/识别进度/解密与自愈消息(识别进度附带的详情被忽略)。"""
|
||
if phase == 'section':
|
||
print_section(args[-1])
|
||
elif phase == 'identify' and len(args) >= 4:
|
||
i, n, label, result = args[0], args[1], args[2], args[3]
|
||
print(f' [{i}/{n}] {label} ... {c("cyan", "→")} {result}')
|
||
elif phase in ('decrypt', 'heal'):
|
||
print(f' {args[-1]}')
|
||
|
||
try:
|
||
result = run_pipeline(folder, {'offline': offline, 'recursive': recursive}, progress_bridge)
|
||
except FileNotFoundError as e:
|
||
print(e)
|
||
return 1
|
||
|
||
print_section('⑤ 导览')
|
||
plan_text = render_plan(result['by_cat'], result['moves'], result['ctx'])
|
||
print(plan_text)
|
||
for w in result['ctx']['warnings']:
|
||
print(c('yellow', f' ⚠ {w}'))
|
||
if result['online_hits']:
|
||
print(f'\n本次联网自动识别 {result["online_hits"]} 首(结果已写入 识别缓存.json,下次离线可用)。')
|
||
guide_path = write_guide(tool_dir, render_guide(result))
|
||
print(f'导览已保存到 {guide_path}')
|
||
|
||
if not result['moves'] and not result['skips']:
|
||
print('无需处理,未做任何修改。')
|
||
return 0
|
||
|
||
if apply_now:
|
||
answer = 'y'
|
||
else:
|
||
try:
|
||
answer = input('\n确认无误?输入 y 回车执行移动,其他键取消: ').strip().lower()
|
||
except EOFError:
|
||
answer = ''
|
||
if answer != 'y':
|
||
print('已取消,未做任何修改。')
|
||
print()
|
||
print_source_summary(result['ctx'])
|
||
return 0
|
||
|
||
print_section('⑥ 执行')
|
||
stats = execute_moves(result['moves'], progress_cb=progress_bridge)
|
||
write_anime_map(tool_dir, result['recs'])
|
||
if recursive:
|
||
n_removed = remove_empty_dirs(folder)
|
||
if n_removed:
|
||
print(f'已清理源文件夹中的 {n_removed} 个空目录。')
|
||
print()
|
||
for key in CATEGORY_KEYS:
|
||
cnt = result['stats'][key]
|
||
if not (cnt['moves'] or cnt['skips']):
|
||
continue
|
||
name = result['ctx']['cfg']['categories'][key]
|
||
print(f' {name:16s} 移动 {cnt["moves"]:3d} 跳过 {cnt["skips"]:3d}')
|
||
print(f'\n完成: 移动 {stats["moved"]} 个,跳过 {stats["skipped"]} 个,失败 {stats["failed"]} 个。')
|
||
if stats['stopped']:
|
||
print('(已中途停止)')
|
||
print_source_summary(result['ctx'])
|
||
return 0
|
||
|
||
|
||
if __name__ == '__main__':
|
||
sys.exit(main())
|