feat(music): 音乐订阅刷新与识别缓存

This commit is contained in:
jxxghp
2026-08-11 11:39:08 +08:00
parent 778d185c9d
commit d0edcfa7bb
12 changed files with 1364 additions and 37 deletions

View File

@@ -8,11 +8,14 @@ from app.chain.music import MusicChain
from app.schemas.types import MediaType
from app.core.context import MusicAlbumInfo, MusicArtistInfo, MusicInfo
from app.core.security import verify_token
from app.db.models.user import User
from app.db.user_oper import get_current_active_superuser_async
from app.modules.listenbrainz import (
LISTENBRAINZ_CHART_RANGES,
LISTENBRAINZ_FRESH_MAX_DAYS,
LISTENBRAINZ_FRESH_SORTS,
)
from app.modules.musicbrainz.music_cache import MusicBrainzCache
router = APIRouter()
@@ -67,6 +70,53 @@ async def recognize_music(
return _serialize_music(info)
@router.get(
"/cache", summary="查询音乐识别缓存", response_model=schemas.Response
)
async def music_recognition_cache(
_: User = Depends(get_current_active_superuser_async),
) -> schemas.Response:
"""查询可管理的 MusicBrainz 识别缓存。"""
cache_items = MusicBrainzCache().list_items()
recognized_count = sum(1 for item in cache_items if item["media_id"])
return schemas.Response(
success=True,
data={
"count": len(cache_items),
"recognized": recognized_count,
"unrecognized": len(cache_items) - recognized_count,
"data": cache_items,
},
)
@router.delete(
"/cache/{cache_key:path}",
summary="删除指定音乐识别缓存",
response_model=schemas.Response,
)
async def delete_music_recognition_cache(
cache_key: str,
_: User = Depends(get_current_active_superuser_async),
) -> schemas.Response:
"""按缓存键删除单条 MusicBrainz 识别缓存。"""
deleted_item = MusicBrainzCache().delete(cache_key)
if not deleted_item:
return schemas.Response(success=False, message="音乐识别缓存不存在")
return schemas.Response(success=True, message="音乐识别缓存删除成功")
@router.delete(
"/cache", summary="清空音乐识别缓存", response_model=schemas.Response
)
async def clear_music_recognition_cache(
_: User = Depends(get_current_active_superuser_async),
) -> schemas.Response:
"""清空全部 MusicBrainz 识别缓存。"""
MusicBrainzCache().clear()
return schemas.Response(success=True, message="音乐识别缓存清理完成")
@router.get(
"/explore",
summary="探索音乐",

View File

@@ -118,8 +118,11 @@ async def delete_cache(
if len(cache_data[domain]) == original_count:
return schemas.Response(success=False, message="未找到指定的种子")
# 保存更新后的缓存
await torrents_chain.async_save_cache(cache_data, torrents_chain.cache_file)
# 保存更新后的缓存:影视与音乐分别回写各自存储文件
video_cache, music_cache = torrents_chain.split_cache_contexts(cache_data)
video_file, music_file = torrents_chain.cache_files()
await torrents_chain.async_save_cache(video_cache, video_file)
await torrents_chain.async_save_cache(music_cache, music_file)
return schemas.Response(success=True, message="种子删除成功")
except Exception as e:
@@ -248,8 +251,11 @@ async def reidentify_cache(
# 更新上下文中的媒体信息
target_context.media_info = mediainfo
# 保存更新后的缓存
await torrents_chain.async_save_cache(cache_data, TorrentsChain().cache_file)
# 保存更新后的缓存:影视与音乐分别回写各自存储文件
video_cache, music_cache = torrents_chain.split_cache_contexts(cache_data)
video_file, music_file = torrents_chain.cache_files()
await torrents_chain.async_save_cache(video_cache, video_file)
await torrents_chain.async_save_cache(music_cache, music_file)
return schemas.Response(
success=True,

View File

@@ -2081,6 +2081,8 @@ class SubscribeChain(ChainBase):
torrents = TorrentsChain().refresh(
sites=sites,
progress_callback=_update_refresh_progress if progress_callback else None,
# 存在音乐订阅时额外抓取站点音乐专用入口,音乐不一定在默认种子首页
include_music=self.has_music_subscribe(),
)
self.match(
torrents,
@@ -2132,6 +2134,13 @@ class SubscribeChain(ChainBase):
return ret_sites
def has_music_subscribe(self) -> bool:
"""判断是否存在可搜索状态的音乐订阅,用于决定是否额外刷新站点音乐入口。"""
return any(
subscribe.type == MediaType.MUSIC.value
for subscribe in SubscribeOper().list(self.get_states_for_search('R')) or []
)
def match(
self,
torrents: Dict[str, List[Context]],

View File

@@ -30,6 +30,9 @@ class TorrentsChain(ChainBase):
_spider_file = "__torrents_cache__"
_rss_file = "__rss_cache__"
# 音乐资源独立缓存,与影视种子分开计算配额与存储,避免被影视资源挤出
_music_spider_file = "__torrents_music_cache__"
_music_rss_file = "__rss_music_cache__"
@property
def cache_file(self) -> str:
@@ -58,7 +61,7 @@ class TorrentsChain(ChainBase):
def get_torrents(self, stype: Optional[str] = None) -> Dict[str, List[Context]]:
"""
获取当前缓存的种子
获取当前缓存的种子,包含独立缓存的音乐资源
:param stype: 强制指定缓存类型spider:爬虫缓存rss:rss缓存
"""
@@ -74,11 +77,59 @@ class TorrentsChain(ChainBase):
# 兼容性处理为旧版本的Context对象补齐新增候选识别字段
self._ensure_context_compatibility(torrents_cache, stype=stype)
# 合并音乐独立缓存,供订阅匹配等消费方按站点读取完整候选
music_cache = self.get_music_torrents(stype=stype)
for domain, contexts in music_cache.items():
if contexts:
torrents_cache.setdefault(domain, []).extend(contexts)
return torrents_cache
def get_music_torrents(self, stype: Optional[str] = None) -> Dict[str, List[Context]]:
"""
获取音乐独立缓存的种子
:param stype: 强制指定缓存类型spider:爬虫缓存rss:rss缓存
"""
if not stype:
stype = settings.SUBSCRIBE_MODE
music_file = self._music_spider_file if stype == 'spider' else self._music_rss_file
music_cache = self.load_cache(music_file) or {}
# 兼容性处理为旧版本的Context对象补齐新增候选识别字段
self._ensure_context_compatibility(music_cache, stype=stype)
return music_cache
def cache_files(self, stype: Optional[str] = None) -> tuple:
"""
返回影视与音乐缓存文件名,供按当前订阅模式回写各自缓存
:param stype: 强制指定缓存类型spider:爬虫缓存rss:rss缓存
"""
if not stype:
stype = settings.SUBSCRIBE_MODE
if stype == 'spider':
return self._spider_file, self._music_spider_file
return self._rss_file, self._music_rss_file
@staticmethod
def split_cache_contexts(
torrents_cache: Dict[str, List[Context]],
) -> tuple:
"""
将合并读取的缓存按种子分类拆分为影视缓存与音乐缓存,用于分别回写各自存储文件。
"""
video_cache: Dict[str, List[Context]] = {}
music_cache: Dict[str, List[Context]] = {}
for domain, contexts in torrents_cache.items():
for context in contexts:
torrent = context.torrent_info
if torrent and torrent.category in (MediaType.MUSIC, MediaType.MUSIC.value):
music_cache.setdefault(domain, []).append(context)
else:
video_cache.setdefault(domain, []).append(context)
return video_cache, music_cache
async def async_get_torrents(self, stype: Optional[str] = None) -> Dict[str, List[Context]]:
"""
异步获取当前缓存的种子
异步获取当前缓存的种子,包含独立缓存的音乐资源
:param stype: 强制指定缓存类型spider:爬虫缓存rss:rss缓存
"""
@@ -88,11 +139,19 @@ class TorrentsChain(ChainBase):
# 异步读取缓存
if stype == 'spider':
torrents_cache = await self.async_load_cache(self._spider_file) or {}
music_cache = await self.async_load_cache(self._music_spider_file) or {}
else:
torrents_cache = await self.async_load_cache(self._rss_file) or {}
music_cache = await self.async_load_cache(self._music_rss_file) or {}
# 兼容性处理为旧版本的Context对象补齐新增候选识别字段
self._ensure_context_compatibility(torrents_cache, stype=stype)
self._ensure_context_compatibility(music_cache, stype=stype)
# 合并音乐独立缓存,供订阅匹配等消费方按站点读取完整候选
for domain, contexts in music_cache.items():
if contexts:
torrents_cache.setdefault(domain, []).extend(contexts)
return torrents_cache
@@ -385,20 +444,24 @@ class TorrentsChain(ChainBase):
def clear_torrents(self):
"""
清理种子缓存数据
清理种子缓存数据,包含音乐独立缓存
"""
logger.info(f'开始清理种子缓存数据 ...')
self.remove_cache(self._spider_file)
self.remove_cache(self._rss_file)
self.remove_cache(self._music_spider_file)
self.remove_cache(self._music_rss_file)
logger.info(f'种子缓存数据清理完成')
async def async_clear_torrents(self):
"""
异步清理种子缓存数据
异步清理种子缓存数据,包含音乐独立缓存
"""
logger.info(f'开始异步清理种子缓存数据 ...')
await self.async_remove_cache(self._spider_file)
await self.async_remove_cache(self._rss_file)
await self.async_remove_cache(self._music_spider_file)
await self.async_remove_cache(self._music_rss_file)
logger.info(f'异步种子缓存数据清理完成')
def browse(self, domain: str, keyword: Optional[str] = None, cat: Optional[str] = None,
@@ -465,6 +528,8 @@ class TorrentsChain(ChainBase):
if not rss_items:
logger.error(f'站点 {domain} 未获取到RSS数据')
return []
# 站点级媒体类型,用于给缺少分类信息的 RSS 种子补充分类
site_media_type = MediaType.from_agent(site.get("media_type"))
# 组装种子
ret_torrents: List[TorrentInfo] = []
try:
@@ -484,6 +549,8 @@ class TorrentsChain(ChainBase):
page_url=item.get("link"),
size=item.get("size"),
pubdate=item["pubdate"].strftime("%Y-%m-%d %H:%M:%S") if item.get("pubdate") else None,
# RSS 报文不带站点分类,按站点媒体类型补充,否则音乐资源无法进入音乐订阅匹配
category=site_media_type.value if site_media_type else None,
)
ret_torrents.append(torrentinfo)
finally:
@@ -491,17 +558,71 @@ class TorrentsChain(ChainBase):
del rss_items
return ret_torrents
@staticmethod
def _music_browse_paths(site: dict) -> List[str]:
"""
返回站点独立于默认浏览入口的音乐种子页面路径。
部分站点的默认种子列表只显示电影和电视剧,音乐需要单独的菜单页面进入;
这类站点在索引配置中用 type=music 的搜索路径声明音乐入口。
默认入口已覆盖音乐(音乐站点或未定义独立入口)时返回空列表。
"""
# 音乐站点全站都是音乐资源,默认浏览入口已经覆盖
if MediaType.from_agent(site.get("media_type")) == MediaType.MUSIC:
return []
paths = (site.get("search") or {}).get("paths") or []
if len(paths) <= 1:
return []
# 计算默认浏览使用的路径,与其相同的音乐入口无需重复抓取
browse_conf = site.get("browse") or {}
default_path = browse_conf.get("path")
if not default_path:
default_path = next(
(item.get("path") for item in paths if item.get("type") in (None, "all")),
paths[0].get("path"),
)
return [
item.get("path") for item in paths
if item.get("type") == "music"
and item.get("path")
and item.get("path") != default_path
]
def __append_music_browse_torrents(
self,
domain: str,
torrents: List[TorrentInfo],
) -> List[TorrentInfo]:
"""
追加抓取站点音乐专用入口的最新种子,并按种子链接去重后返回合并结果。
"""
seen = {torrent.enclosure for torrent in torrents if torrent.enclosure}
for page in range(2):
page_torrents = self.browse(domain=domain, page=page, mtype=MediaType.MUSIC)
if not page_torrents:
# 某一页没有数据,说明已经到最后一页,停止获取
break
for torrent in page_torrents:
if torrent.enclosure and torrent.enclosure in seen:
continue
if torrent.enclosure:
seen.add(torrent.enclosure)
torrents.append(torrent)
return torrents
def refresh(
self,
stype: Optional[str] = None,
sites: List[int] = None,
progress_callback: Optional[Callable[..., None]] = None,
include_music: bool = False,
) -> Dict[str, List[Context]]:
"""
刷新站点最新资源,识别并缓存起来
:param stype: 强制指定缓存类型spider:爬虫缓存rss:rss缓存
:param sites: 强制指定站点ID列表为空则读取设置的订阅站点
:param progress_callback: 资源刷新进度更新回调
:param include_music: 是否额外抓取站点的音乐专用浏览入口,服务音乐订阅
"""
def __is_no_cache_site(_domain: str) -> bool:
@@ -521,13 +642,21 @@ class TorrentsChain(ChainBase):
if not sites:
sites = SystemConfigOper().get(SystemConfigKey.RssSites) or []
# 读取缓存
torrents_cache = self.get_torrents()
# 读取缓存,影视与音乐分别独立存储
if stype == 'spider':
torrents_cache = self.load_cache(self._spider_file) or {}
music_cache = self.load_cache(self._music_spider_file) or {}
else:
torrents_cache = self.load_cache(self._rss_file) or {}
music_cache = self.load_cache(self._music_rss_file) or {}
self._ensure_context_compatibility(torrents_cache, stype=stype)
self._ensure_context_compatibility(music_cache, stype=stype)
# 缓存过滤掉无效种子
for _domain, _torrents in torrents_cache.items():
torrents_cache[_domain] = [_torrent for _torrent in _torrents
if not TorrentHelper().is_invalid(_torrent.torrent_info.enclosure)]
# 缓存过滤掉无效种子(影视与音乐缓存分别处理)
for _cache in (torrents_cache, music_cache):
for _domain, _torrents in _cache.items():
_cache[_domain] = [_torrent for _torrent in _torrents
if not TorrentHelper().is_invalid(_torrent.torrent_info.enclosure)]
# 需要刷新的站点domain
domains = []
@@ -572,31 +701,47 @@ class TorrentsChain(ChainBase):
else:
# 如果某一页没有数据,说明已经到最后一页,停止获取
break
# 存在音乐订阅时,默认首页可能不包含音乐资源,需要额外抓取音乐专用入口
if include_music and self._music_browse_paths(indexer):
torrents = self.__append_music_browse_torrents(
domain=domain, torrents=torrents
)
else:
# 刷新RSS种子
torrents: List[TorrentInfo] = self.rss(domain=domain)
# 按pubdate降序排列
torrents.sort(key=lambda x: x.pubdate or '', reverse=True)
# 取前N条
torrents = torrents[:settings.CONF.refresh]
if torrents:
# 音乐与影视按同一公共参数独立计算刷新配额,并分别写入各自缓存,音乐不会被影视资源挤出
music_torrents = [
t for t in torrents if t.category == MediaType.MUSIC.value
][:settings.CONF.refresh]
torrents = [
t for t in torrents if t.category != MediaType.MUSIC.value
][:settings.CONF.refresh]
if torrents or music_torrents:
if __is_no_cache_site(domain):
# 不需要缓存的站点,直接处理
logger.info(f'{indexer.get("name")}{len(torrents)} 个种子 (不缓存)')
logger.info(f'{indexer.get("name")}{len(torrents) + len(music_torrents)} 个种子 (不缓存)')
torrents_cache[domain] = []
music_cache[domain] = []
else:
# 过滤出没有处理过的种子 - 优化:使用集合查找,避免重复创建字符串列表
cached_signatures = {f'{t.torrent_info.title}{t.torrent_info.description}'
for t in torrents_cache.get(domain) or []}
torrents = [torrent for torrent in torrents
if f'{torrent.title}{torrent.description}' not in cached_signatures]
if torrents:
logger.info(f'{indexer.get("name")}{len(torrents)} 个新种子')
# 音乐种子对照音乐独立缓存去重
music_signatures = {f'{t.torrent_info.title}{t.torrent_info.description}'
for t in music_cache.get(domain) or []}
music_torrents = [torrent for torrent in music_torrents
if f'{torrent.title}{torrent.description}' not in music_signatures]
if torrents or music_torrents:
logger.info(f'{indexer.get("name")}{len(torrents) + len(music_torrents)} 个新种子')
else:
logger.info(f'{indexer.get("name")} 没有新种子')
continue
try:
for torrent in torrents:
for torrent in torrents + music_torrents:
if global_vars.is_system_stopped:
break
if not torrent.enclosure:
@@ -647,29 +792,39 @@ class TorrentsChain(ChainBase):
# 如果未识别到媒体信息设置初始失败次数为1
if not mediainfo or not all(resolve_media_identity(media=mediainfo)):
context.media_recognize_fail_count = 1
# 添加到缓存
if not torrents_cache.get(domain):
torrents_cache[domain] = [context]
# 添加到缓存:音乐进入独立缓存,与影视分开存储
if torrent.category == MediaType.MUSIC.value:
target_cache = music_cache
else:
torrents_cache[domain].append(context)
# 如果超过了限制条数则移除掉前面的
if len(torrents_cache[domain]) > settings.CONF.torrents:
torrents_cache[domain] = torrents_cache[domain][-settings.CONF.torrents:]
target_cache = torrents_cache
if not target_cache.get(domain):
target_cache[domain] = [context]
else:
target_cache[domain].append(context)
# 如果超过了限制条数则移除掉前面的,音乐与影视各自独立计算配额
if len(target_cache[domain]) > settings.CONF.torrents:
target_cache[domain] = target_cache[domain][-settings.CONF.torrents:]
finally:
torrents.clear()
music_torrents.clear()
del torrents
del music_torrents
else:
logger.info(f'{indexer.get("name")} 没有获取到种子')
# 保存缓存到本地
# 保存缓存到本地,影视与音乐分别存储
if stype == "spider":
self.save_cache(torrents_cache, self._spider_file)
self.save_cache(music_cache, self._music_spider_file)
else:
self.save_cache(torrents_cache, self._rss_file)
self.save_cache(music_cache, self._music_rss_file)
# 去除不在站点范围内的缓存种子
if sites and torrents_cache:
torrents_cache = {k: v for k, v in torrents_cache.items() if k in domains}
if sites and music_cache:
music_cache = {k: v for k, v in music_cache.items() if k in domains}
if progress_callback:
progress_callback(
@@ -678,6 +833,11 @@ class TorrentsChain(ChainBase):
data={"total": total_indexers, "finished": total_indexers},
)
# 订阅匹配需要完整候选,音乐独立缓存在返回值中按站点合并
for _domain, _contexts in music_cache.items():
if _contexts:
torrents_cache.setdefault(_domain, []).extend(_contexts)
return torrents_cache
@staticmethod

View File

@@ -165,6 +165,8 @@ class SiteSpider:
# 种子搜索相对路径
paths = self.search.get('paths', [])
torrentspath = ""
# 是否选中了媒体类型专用路径,浏览模式下专用路径优先于 browse 配置
typed_path_selected = False
if len(paths) == 1:
torrentspath = paths[0].get('path', '')
else:
@@ -183,6 +185,7 @@ class SiteSpider:
not expected_type and path_type == "all"
):
torrentspath = path.get('path', '')
typed_path_selected = bool(expected_type and path_type == expected_type)
break
if not torrentspath:
torrentspath = fallback_path
@@ -282,16 +285,18 @@ class SiteSpider:
"page": self.page or 0,
"keyword": ""
}
# 有单独浏览路径
if self.browse:
# 有单独浏览路径;指定了媒体类型专用路径时不覆盖,确保音乐等专用入口可达
if self.browse and not typed_path_selected:
torrentspath = self.browse.get("path")
if self.browse.get("start"):
start_page = int(self.browse.get("start")) + int(self.page or 0)
inputs_dict.update({
"page": start_page
})
elif self.page:
torrentspath = torrentspath + f"?page={self.page}"
elif self.page and "{page}" not in str(torrentspath):
# 按路径是否已带查询参数选择连接符,避免拼出两个问号的非法地址
separator = "&" if "?" in str(torrentspath) else "?"
torrentspath = torrentspath + f"{separator}page={self.page}"
# 搜索Url
searchurl = self.domain + str(torrentspath).format(**inputs_dict)

View File

@@ -19,6 +19,7 @@ from app.core.context import (
from app.core.meta import MetaBase, MetaMusic
from app.log import logger
from app.modules import _ModuleBase
from app.modules.musicbrainz.music_cache import MusicBrainzCache
from app.schemas.types import MediaRecognizeType, MediaType, ModuleType
from app.utils.http import RequestUtils
from app.utils.zhconv import convert as zhconv_convert
@@ -36,6 +37,8 @@ class MusicBrainzModule(_ModuleBase):
_request_interval = 1.0
_request_lock = threading.Lock()
_last_request_at = 0.0
# 本地识别缓存,由模块管理器初始化时挂载
cache: MusicBrainzCache = None
# 全局复用 HTTP 会话keep-alive 省去每次请求的 DNS+TLS 握手(约 6s → 0.4s
_session: Optional[Session] = None
_session_lock = threading.Lock()
@@ -72,14 +75,32 @@ class MusicBrainzModule(_ModuleBase):
)
def init_module(self) -> None:
"""初始化无状态的 MusicBrainz 模块。"""
"""初始化 MusicBrainz 模块并挂载本地识别缓存"""
self.cache = MusicBrainzCache()
def init_setting(self) -> Optional[Tuple[str, Union[str, bool]]]:
"""MusicBrainz 无需独立密钥或启用开关。"""
return None
def stop(self) -> None:
"""停止模块;当前实现没有需要释放的持久资源"""
"""停止模块,退出前持久化识别缓存"""
if self.cache:
try:
self.cache.save()
except Exception as err:
logger.error(f"保存音乐识别缓存失败:{str(err)}")
def scheduler_job(self) -> None:
"""定时任务每10分钟持久化一次音乐识别缓存。"""
if self.cache:
self.cache.save()
def clear_cache(self) -> None:
"""响应全局缓存清理事件,清空音乐识别缓存。"""
logger.info("开始清除音乐识别缓存 ...")
if self.cache:
self.cache.clear()
logger.info("音乐识别缓存清除完成")
def test(self) -> Tuple[bool, str]:
"""测试 MusicBrainz 搜索接口连通性。"""
@@ -610,11 +631,23 @@ class MusicBrainzModule(_ModuleBase):
if source == self._source and mediaid:
return self.recognize_music(source, str(mediaid))
return None
# 识别缓存命中直接响应,避免重复搜索占用 MusicBrainz 限流配额
cache_enabled = bool(kwargs.get("cache", True))
if cache_enabled and self.cache:
cached_info = self.cache.get(meta)
if cached_info:
if cached_info.media_id:
logger.info(f"{meta.title} 使用音乐识别缓存:{cached_info.title}")
else:
logger.info(f"{meta.title} 使用音乐识别缓存:无法识别")
cached_info.recognize_cache_hit = True
return cached_info
# 携带数据源与原生 ID 的请求优先按详情识别
resolved_source = source or meta.media_source
if resolved_source and (mediaid or meta.media_id):
info = self.recognize_music(resolved_source, str(mediaid or meta.media_id))
if info:
self._update_recognize_cache(meta, info)
return info
# 无身份时按标题搜索并挑选可信候选,检索不到时返回元数据兑底
# 文件识别只能从 Recording 中挑选,专辑或艺术家同名结果不能成为音轨身份。
@@ -626,7 +659,38 @@ class MusicBrainzModule(_ModuleBase):
if not matched and meta.artists:
albums = self._search_albums(meta, limit=10)
matched = self._select_album_candidate(meta, albums)
return matched or self._info_from_meta(meta)
result = matched or self._info_from_meta(meta)
# 无远端身份的兑底结果同样入缓存,避免批量识别时反复搜索同一文件
self._update_recognize_cache(meta, result)
return result
def _update_recognize_cache(self, meta: MetaMusic, info: Optional[MusicInfo]) -> None:
"""识别完成后把结果写入本地识别缓存,未挂载缓存时静默跳过。"""
if self.cache:
self.cache.update(meta, info)
def update_recognize_cache(
self,
meta: MetaBase,
mediainfo: MusicInfo,
) -> Optional[bool]:
"""回填音乐本地识别缓存,共享识别成功后避免重复回查。"""
if not meta or not mediainfo:
return None
if not isinstance(meta, MetaMusic) or not isinstance(mediainfo, MusicInfo):
return None
if mediainfo.source != self._source:
return None
self._update_recognize_cache(meta, mediainfo)
return True
async def async_update_recognize_cache(
self,
meta: MetaBase,
mediainfo: MusicInfo,
) -> Optional[bool]:
"""异步回填音乐本地识别缓存。"""
return self.update_recognize_cache(meta=meta, mediainfo=mediainfo)
async def async_recognize_media(
self,

View File

@@ -0,0 +1,234 @@
import pickle
import traceback
from math import ceil
from threading import RLock
from time import time
from typing import Optional
from app.core.cache import FileCache, TTLCache
from app.core.config import settings
from app.core.context import MusicInfo
from app.core.meta import MetaMusic
from app.log import logger
from app.utils.singleton import WeakSingleton
lock = RLock()
PERSISTENCE_VERSION = 1
PERSISTENCE_REGION = "recognize"
PERSISTENCE_KEY = "musicbrainz"
class MusicBrainzCache(metaclass=WeakSingleton):
"""
MusicBrainz识别缓存数据
{
"source": '',
"media_id": '',
"title": '',
"artists": [],
"album": '',
"year": '',
"music_type": ''
}
"""
def __init__(self):
"""初始化音乐识别缓存并恢复未过期的持久化数据。"""
self.maxsize = settings.CONF.musicbrainz
self.ttl = settings.CONF.meta
self.region = "__musicbrainz_cache__"
self._cache = TTLCache(region=self.region, maxsize=self.maxsize, ttl=self.ttl)
self._expires_at: dict[str, float] = {}
self._dirty = False
self._file_cache = None
if not self._cache.is_redis():
self._file_cache = FileCache(base=settings.CACHE_PATH, ttl=self.ttl)
self._restore()
def _restore(self) -> None:
"""从统一文件缓存恢复仍在有效期内的音乐识别数据。"""
try:
content = self._file_cache.get(PERSISTENCE_KEY, region=PERSISTENCE_REGION)
if not content:
return
payload = pickle.loads(content)
now = time()
if (
not isinstance(payload, dict)
or payload.get("version") != PERSISTENCE_VERSION
or not isinstance(payload.get("items"), dict)
):
return
for key, item in payload["items"].items():
if not isinstance(item, dict):
self._dirty = True
continue
value = item.get("value")
expires_at = item.get("expires_at")
if not isinstance(value, dict) or not isinstance(expires_at, (int, float)):
self._dirty = True
continue
remaining_ttl = expires_at - now
if remaining_ttl <= 0:
self._dirty = True
continue
self._cache.set(key, value, ttl=ceil(remaining_ttl))
self._expires_at[key] = expires_at
except Exception as err:
logger.error(f"加载音乐识别缓存失败:{str(err)} - {traceback.format_exc()}")
def _set(self, key: str, value: dict) -> None:
"""写入单条音乐识别缓存并记录其独立过期时间。"""
self._cache.set(key, value)
if not self._cache.is_redis():
self._expires_at[key] = time() + self.ttl
self._dirty = True
def clear(self):
"""
清空所有音乐识别缓存
"""
with lock:
self._cache.clear()
self._expires_at.clear()
self._dirty = True
self.save(force=True)
def list_items(self) -> list[dict]:
"""
返回可供管理界面展示的音乐识别缓存列表。
"""
with lock:
cache_items = []
for key, value in self._cache.items():
if not isinstance(value, dict):
continue
cache_items.append({
"key": key,
"media_id": value.get("media_id") or "",
"title": value.get("title") or "",
"artists": value.get("artists") or [],
"album": value.get("album") or "",
"year": value.get("year") or "",
"music_type": value.get("music_type") or "recording",
"cover_url": value.get("cover_url") or "",
})
return sorted(cache_items, key=lambda item: item["key"])
@staticmethod
def __get_key(meta: MetaMusic) -> str:
"""
获取缓存KEY携带数据源原生 ID 时以 ID 为准身份
"""
artists = "/".join(meta.artists or [])
return f"[音乐]{meta.media_id or meta.title}-{artists}-{meta.album}-{meta.year}"
def get(self, meta: MetaMusic) -> Optional[MusicInfo]:
"""
根据元数据获取缓存的音乐识别结果
@param meta: 音乐元数据
@return: 缓存命中的音乐信息,未命中返回 None
"""
key = self.__get_key(meta)
with lock:
cache_data = self._cache.get(key)
if not cache_data and self._expires_at.pop(key, None) is not None:
self._dirty = True
if not cache_data:
return None
try:
return MusicInfo.from_dict(cache_data)
except Exception as err:
logger.error(f"解析音乐识别缓存失败:{str(err)}")
return None
def delete(self, key: str) -> dict:
"""
删除缓存信息
@param key: 缓存key
@return: 被删除的缓存内容
"""
with lock:
cache_data = self._cache.get(key)
if cache_data:
self._cache.delete(key)
self._expires_at.pop(key, None)
self._dirty = True
self.save(force=True)
return cache_data
return {}
def update(self, meta: MetaMusic, info: Optional[MusicInfo]) -> None:
"""
新增或更新缓存条目,无远端身份的兜底结果也写入内存负缓存,
避免批量识别时反复请求 MusicBrainz 触发限流
"""
if not meta or not info:
return
key = self.__get_key(meta)
cache_data = info.to_dict()
# 上游原始响应体积大且不参与身份恢复,不入缓存
cache_data.pop("raw_data", None)
with lock:
self._set(key, cache_data)
def save(self, force: bool = False) -> None:
"""
使用统一文件缓存保存未过期的音乐识别数据。
"""
if self._cache.is_redis():
return
if not self._file_cache:
return
with lock:
now = time()
cache_items = dict(self._cache.items())
active_keys = set(cache_items)
stale_keys = set(self._expires_at) - active_keys
if stale_keys:
for key in stale_keys:
self._expires_at.pop(key, None)
self._dirty = True
persisted_items = {}
for key, value in cache_items.items():
expires_at = self._expires_at.get(key)
if expires_at is None:
expires_at = now + self.ttl
self._expires_at[key] = expires_at
self._dirty = True
# 负缓存只留在内存,重启后允许重新尝试识别
if expires_at <= now or not value.get("media_id"):
continue
persisted_items[key] = {
"value": value,
"expires_at": expires_at,
}
if not force and not self._dirty:
return
try:
if persisted_items:
payload = {
"version": PERSISTENCE_VERSION,
"items": persisted_items,
}
self._file_cache.set(
PERSISTENCE_KEY,
pickle.dumps(payload, pickle.HIGHEST_PROTOCOL),
region=PERSISTENCE_REGION,
)
else:
self._file_cache.delete(PERSISTENCE_KEY, region=PERSISTENCE_REGION)
self._dirty = False
except Exception as err:
logger.error(f"保存音乐识别缓存失败:{str(err)} - {traceback.format_exc()}")
def __del__(self):
"""实例释放前保存非 Redis 缓存。"""
try:
self.save()
except Exception:
pass