mirror of
https://github.com/jxxghp/MoviePilot.git
synced 2026-08-14 10:14:36 +08:00
feat(music): 音乐订阅刷新与识别缓存
This commit is contained in:
@@ -8,11 +8,14 @@ from app.chain.music import MusicChain
|
||||
from app.schemas.types import MediaType
|
||||
from app.core.context import MusicAlbumInfo, MusicArtistInfo, MusicInfo
|
||||
from app.core.security import verify_token
|
||||
from app.db.models.user import User
|
||||
from app.db.user_oper import get_current_active_superuser_async
|
||||
from app.modules.listenbrainz import (
|
||||
LISTENBRAINZ_CHART_RANGES,
|
||||
LISTENBRAINZ_FRESH_MAX_DAYS,
|
||||
LISTENBRAINZ_FRESH_SORTS,
|
||||
)
|
||||
from app.modules.musicbrainz.music_cache import MusicBrainzCache
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
@@ -67,6 +70,53 @@ async def recognize_music(
|
||||
return _serialize_music(info)
|
||||
|
||||
|
||||
@router.get(
|
||||
"/cache", summary="查询音乐识别缓存", response_model=schemas.Response
|
||||
)
|
||||
async def music_recognition_cache(
|
||||
_: User = Depends(get_current_active_superuser_async),
|
||||
) -> schemas.Response:
|
||||
"""查询可管理的 MusicBrainz 识别缓存。"""
|
||||
cache_items = MusicBrainzCache().list_items()
|
||||
recognized_count = sum(1 for item in cache_items if item["media_id"])
|
||||
return schemas.Response(
|
||||
success=True,
|
||||
data={
|
||||
"count": len(cache_items),
|
||||
"recognized": recognized_count,
|
||||
"unrecognized": len(cache_items) - recognized_count,
|
||||
"data": cache_items,
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
@router.delete(
|
||||
"/cache/{cache_key:path}",
|
||||
summary="删除指定音乐识别缓存",
|
||||
response_model=schemas.Response,
|
||||
)
|
||||
async def delete_music_recognition_cache(
|
||||
cache_key: str,
|
||||
_: User = Depends(get_current_active_superuser_async),
|
||||
) -> schemas.Response:
|
||||
"""按缓存键删除单条 MusicBrainz 识别缓存。"""
|
||||
deleted_item = MusicBrainzCache().delete(cache_key)
|
||||
if not deleted_item:
|
||||
return schemas.Response(success=False, message="音乐识别缓存不存在")
|
||||
return schemas.Response(success=True, message="音乐识别缓存删除成功")
|
||||
|
||||
|
||||
@router.delete(
|
||||
"/cache", summary="清空音乐识别缓存", response_model=schemas.Response
|
||||
)
|
||||
async def clear_music_recognition_cache(
|
||||
_: User = Depends(get_current_active_superuser_async),
|
||||
) -> schemas.Response:
|
||||
"""清空全部 MusicBrainz 识别缓存。"""
|
||||
MusicBrainzCache().clear()
|
||||
return schemas.Response(success=True, message="音乐识别缓存清理完成")
|
||||
|
||||
|
||||
@router.get(
|
||||
"/explore",
|
||||
summary="探索音乐",
|
||||
|
||||
@@ -118,8 +118,11 @@ async def delete_cache(
|
||||
if len(cache_data[domain]) == original_count:
|
||||
return schemas.Response(success=False, message="未找到指定的种子")
|
||||
|
||||
# 保存更新后的缓存
|
||||
await torrents_chain.async_save_cache(cache_data, torrents_chain.cache_file)
|
||||
# 保存更新后的缓存:影视与音乐分别回写各自存储文件
|
||||
video_cache, music_cache = torrents_chain.split_cache_contexts(cache_data)
|
||||
video_file, music_file = torrents_chain.cache_files()
|
||||
await torrents_chain.async_save_cache(video_cache, video_file)
|
||||
await torrents_chain.async_save_cache(music_cache, music_file)
|
||||
|
||||
return schemas.Response(success=True, message="种子删除成功")
|
||||
except Exception as e:
|
||||
@@ -248,8 +251,11 @@ async def reidentify_cache(
|
||||
# 更新上下文中的媒体信息
|
||||
target_context.media_info = mediainfo
|
||||
|
||||
# 保存更新后的缓存
|
||||
await torrents_chain.async_save_cache(cache_data, TorrentsChain().cache_file)
|
||||
# 保存更新后的缓存:影视与音乐分别回写各自存储文件
|
||||
video_cache, music_cache = torrents_chain.split_cache_contexts(cache_data)
|
||||
video_file, music_file = torrents_chain.cache_files()
|
||||
await torrents_chain.async_save_cache(video_cache, video_file)
|
||||
await torrents_chain.async_save_cache(music_cache, music_file)
|
||||
|
||||
return schemas.Response(
|
||||
success=True,
|
||||
|
||||
@@ -2081,6 +2081,8 @@ class SubscribeChain(ChainBase):
|
||||
torrents = TorrentsChain().refresh(
|
||||
sites=sites,
|
||||
progress_callback=_update_refresh_progress if progress_callback else None,
|
||||
# 存在音乐订阅时额外抓取站点音乐专用入口,音乐不一定在默认种子首页
|
||||
include_music=self.has_music_subscribe(),
|
||||
)
|
||||
self.match(
|
||||
torrents,
|
||||
@@ -2132,6 +2134,13 @@ class SubscribeChain(ChainBase):
|
||||
|
||||
return ret_sites
|
||||
|
||||
def has_music_subscribe(self) -> bool:
|
||||
"""判断是否存在可搜索状态的音乐订阅,用于决定是否额外刷新站点音乐入口。"""
|
||||
return any(
|
||||
subscribe.type == MediaType.MUSIC.value
|
||||
for subscribe in SubscribeOper().list(self.get_states_for_search('R')) or []
|
||||
)
|
||||
|
||||
def match(
|
||||
self,
|
||||
torrents: Dict[str, List[Context]],
|
||||
|
||||
@@ -30,6 +30,9 @@ class TorrentsChain(ChainBase):
|
||||
|
||||
_spider_file = "__torrents_cache__"
|
||||
_rss_file = "__rss_cache__"
|
||||
# 音乐资源独立缓存,与影视种子分开计算配额与存储,避免被影视资源挤出
|
||||
_music_spider_file = "__torrents_music_cache__"
|
||||
_music_rss_file = "__rss_music_cache__"
|
||||
|
||||
@property
|
||||
def cache_file(self) -> str:
|
||||
@@ -58,7 +61,7 @@ class TorrentsChain(ChainBase):
|
||||
|
||||
def get_torrents(self, stype: Optional[str] = None) -> Dict[str, List[Context]]:
|
||||
"""
|
||||
获取当前缓存的种子
|
||||
获取当前缓存的种子,包含独立缓存的音乐资源
|
||||
:param stype: 强制指定缓存类型,spider:爬虫缓存,rss:rss缓存
|
||||
"""
|
||||
|
||||
@@ -74,11 +77,59 @@ class TorrentsChain(ChainBase):
|
||||
# 兼容性处理:为旧版本的Context对象补齐新增候选识别字段
|
||||
self._ensure_context_compatibility(torrents_cache, stype=stype)
|
||||
|
||||
# 合并音乐独立缓存,供订阅匹配等消费方按站点读取完整候选
|
||||
music_cache = self.get_music_torrents(stype=stype)
|
||||
for domain, contexts in music_cache.items():
|
||||
if contexts:
|
||||
torrents_cache.setdefault(domain, []).extend(contexts)
|
||||
|
||||
return torrents_cache
|
||||
|
||||
def get_music_torrents(self, stype: Optional[str] = None) -> Dict[str, List[Context]]:
|
||||
"""
|
||||
获取音乐独立缓存的种子
|
||||
:param stype: 强制指定缓存类型,spider:爬虫缓存,rss:rss缓存
|
||||
"""
|
||||
if not stype:
|
||||
stype = settings.SUBSCRIBE_MODE
|
||||
music_file = self._music_spider_file if stype == 'spider' else self._music_rss_file
|
||||
music_cache = self.load_cache(music_file) or {}
|
||||
# 兼容性处理:为旧版本的Context对象补齐新增候选识别字段
|
||||
self._ensure_context_compatibility(music_cache, stype=stype)
|
||||
return music_cache
|
||||
|
||||
def cache_files(self, stype: Optional[str] = None) -> tuple:
|
||||
"""
|
||||
返回影视与音乐缓存文件名,供按当前订阅模式回写各自缓存
|
||||
:param stype: 强制指定缓存类型,spider:爬虫缓存,rss:rss缓存
|
||||
"""
|
||||
if not stype:
|
||||
stype = settings.SUBSCRIBE_MODE
|
||||
if stype == 'spider':
|
||||
return self._spider_file, self._music_spider_file
|
||||
return self._rss_file, self._music_rss_file
|
||||
|
||||
@staticmethod
|
||||
def split_cache_contexts(
|
||||
torrents_cache: Dict[str, List[Context]],
|
||||
) -> tuple:
|
||||
"""
|
||||
将合并读取的缓存按种子分类拆分为影视缓存与音乐缓存,用于分别回写各自存储文件。
|
||||
"""
|
||||
video_cache: Dict[str, List[Context]] = {}
|
||||
music_cache: Dict[str, List[Context]] = {}
|
||||
for domain, contexts in torrents_cache.items():
|
||||
for context in contexts:
|
||||
torrent = context.torrent_info
|
||||
if torrent and torrent.category in (MediaType.MUSIC, MediaType.MUSIC.value):
|
||||
music_cache.setdefault(domain, []).append(context)
|
||||
else:
|
||||
video_cache.setdefault(domain, []).append(context)
|
||||
return video_cache, music_cache
|
||||
|
||||
async def async_get_torrents(self, stype: Optional[str] = None) -> Dict[str, List[Context]]:
|
||||
"""
|
||||
异步获取当前缓存的种子
|
||||
异步获取当前缓存的种子,包含独立缓存的音乐资源
|
||||
:param stype: 强制指定缓存类型,spider:爬虫缓存,rss:rss缓存
|
||||
"""
|
||||
|
||||
@@ -88,11 +139,19 @@ class TorrentsChain(ChainBase):
|
||||
# 异步读取缓存
|
||||
if stype == 'spider':
|
||||
torrents_cache = await self.async_load_cache(self._spider_file) or {}
|
||||
music_cache = await self.async_load_cache(self._music_spider_file) or {}
|
||||
else:
|
||||
torrents_cache = await self.async_load_cache(self._rss_file) or {}
|
||||
music_cache = await self.async_load_cache(self._music_rss_file) or {}
|
||||
|
||||
# 兼容性处理:为旧版本的Context对象补齐新增候选识别字段
|
||||
self._ensure_context_compatibility(torrents_cache, stype=stype)
|
||||
self._ensure_context_compatibility(music_cache, stype=stype)
|
||||
|
||||
# 合并音乐独立缓存,供订阅匹配等消费方按站点读取完整候选
|
||||
for domain, contexts in music_cache.items():
|
||||
if contexts:
|
||||
torrents_cache.setdefault(domain, []).extend(contexts)
|
||||
|
||||
return torrents_cache
|
||||
|
||||
@@ -385,20 +444,24 @@ class TorrentsChain(ChainBase):
|
||||
|
||||
def clear_torrents(self):
|
||||
"""
|
||||
清理种子缓存数据
|
||||
清理种子缓存数据,包含音乐独立缓存
|
||||
"""
|
||||
logger.info(f'开始清理种子缓存数据 ...')
|
||||
self.remove_cache(self._spider_file)
|
||||
self.remove_cache(self._rss_file)
|
||||
self.remove_cache(self._music_spider_file)
|
||||
self.remove_cache(self._music_rss_file)
|
||||
logger.info(f'种子缓存数据清理完成')
|
||||
|
||||
async def async_clear_torrents(self):
|
||||
"""
|
||||
异步清理种子缓存数据
|
||||
异步清理种子缓存数据,包含音乐独立缓存
|
||||
"""
|
||||
logger.info(f'开始异步清理种子缓存数据 ...')
|
||||
await self.async_remove_cache(self._spider_file)
|
||||
await self.async_remove_cache(self._rss_file)
|
||||
await self.async_remove_cache(self._music_spider_file)
|
||||
await self.async_remove_cache(self._music_rss_file)
|
||||
logger.info(f'异步种子缓存数据清理完成')
|
||||
|
||||
def browse(self, domain: str, keyword: Optional[str] = None, cat: Optional[str] = None,
|
||||
@@ -465,6 +528,8 @@ class TorrentsChain(ChainBase):
|
||||
if not rss_items:
|
||||
logger.error(f'站点 {domain} 未获取到RSS数据!')
|
||||
return []
|
||||
# 站点级媒体类型,用于给缺少分类信息的 RSS 种子补充分类
|
||||
site_media_type = MediaType.from_agent(site.get("media_type"))
|
||||
# 组装种子
|
||||
ret_torrents: List[TorrentInfo] = []
|
||||
try:
|
||||
@@ -484,6 +549,8 @@ class TorrentsChain(ChainBase):
|
||||
page_url=item.get("link"),
|
||||
size=item.get("size"),
|
||||
pubdate=item["pubdate"].strftime("%Y-%m-%d %H:%M:%S") if item.get("pubdate") else None,
|
||||
# RSS 报文不带站点分类,按站点媒体类型补充,否则音乐资源无法进入音乐订阅匹配
|
||||
category=site_media_type.value if site_media_type else None,
|
||||
)
|
||||
ret_torrents.append(torrentinfo)
|
||||
finally:
|
||||
@@ -491,17 +558,71 @@ class TorrentsChain(ChainBase):
|
||||
del rss_items
|
||||
return ret_torrents
|
||||
|
||||
@staticmethod
|
||||
def _music_browse_paths(site: dict) -> List[str]:
|
||||
"""
|
||||
返回站点独立于默认浏览入口的音乐种子页面路径。
|
||||
|
||||
部分站点的默认种子列表只显示电影和电视剧,音乐需要单独的菜单页面进入;
|
||||
这类站点在索引配置中用 type=music 的搜索路径声明音乐入口。
|
||||
默认入口已覆盖音乐(音乐站点或未定义独立入口)时返回空列表。
|
||||
"""
|
||||
# 音乐站点全站都是音乐资源,默认浏览入口已经覆盖
|
||||
if MediaType.from_agent(site.get("media_type")) == MediaType.MUSIC:
|
||||
return []
|
||||
paths = (site.get("search") or {}).get("paths") or []
|
||||
if len(paths) <= 1:
|
||||
return []
|
||||
# 计算默认浏览使用的路径,与其相同的音乐入口无需重复抓取
|
||||
browse_conf = site.get("browse") or {}
|
||||
default_path = browse_conf.get("path")
|
||||
if not default_path:
|
||||
default_path = next(
|
||||
(item.get("path") for item in paths if item.get("type") in (None, "all")),
|
||||
paths[0].get("path"),
|
||||
)
|
||||
return [
|
||||
item.get("path") for item in paths
|
||||
if item.get("type") == "music"
|
||||
and item.get("path")
|
||||
and item.get("path") != default_path
|
||||
]
|
||||
|
||||
def __append_music_browse_torrents(
|
||||
self,
|
||||
domain: str,
|
||||
torrents: List[TorrentInfo],
|
||||
) -> List[TorrentInfo]:
|
||||
"""
|
||||
追加抓取站点音乐专用入口的最新种子,并按种子链接去重后返回合并结果。
|
||||
"""
|
||||
seen = {torrent.enclosure for torrent in torrents if torrent.enclosure}
|
||||
for page in range(2):
|
||||
page_torrents = self.browse(domain=domain, page=page, mtype=MediaType.MUSIC)
|
||||
if not page_torrents:
|
||||
# 某一页没有数据,说明已经到最后一页,停止获取
|
||||
break
|
||||
for torrent in page_torrents:
|
||||
if torrent.enclosure and torrent.enclosure in seen:
|
||||
continue
|
||||
if torrent.enclosure:
|
||||
seen.add(torrent.enclosure)
|
||||
torrents.append(torrent)
|
||||
return torrents
|
||||
|
||||
def refresh(
|
||||
self,
|
||||
stype: Optional[str] = None,
|
||||
sites: List[int] = None,
|
||||
progress_callback: Optional[Callable[..., None]] = None,
|
||||
include_music: bool = False,
|
||||
) -> Dict[str, List[Context]]:
|
||||
"""
|
||||
刷新站点最新资源,识别并缓存起来
|
||||
:param stype: 强制指定缓存类型,spider:爬虫缓存,rss:rss缓存
|
||||
:param sites: 强制指定站点ID列表,为空则读取设置的订阅站点
|
||||
:param progress_callback: 资源刷新进度更新回调
|
||||
:param include_music: 是否额外抓取站点的音乐专用浏览入口,服务音乐订阅
|
||||
"""
|
||||
|
||||
def __is_no_cache_site(_domain: str) -> bool:
|
||||
@@ -521,13 +642,21 @@ class TorrentsChain(ChainBase):
|
||||
if not sites:
|
||||
sites = SystemConfigOper().get(SystemConfigKey.RssSites) or []
|
||||
|
||||
# 读取缓存
|
||||
torrents_cache = self.get_torrents()
|
||||
# 读取缓存,影视与音乐分别独立存储
|
||||
if stype == 'spider':
|
||||
torrents_cache = self.load_cache(self._spider_file) or {}
|
||||
music_cache = self.load_cache(self._music_spider_file) or {}
|
||||
else:
|
||||
torrents_cache = self.load_cache(self._rss_file) or {}
|
||||
music_cache = self.load_cache(self._music_rss_file) or {}
|
||||
self._ensure_context_compatibility(torrents_cache, stype=stype)
|
||||
self._ensure_context_compatibility(music_cache, stype=stype)
|
||||
|
||||
# 缓存过滤掉无效种子
|
||||
for _domain, _torrents in torrents_cache.items():
|
||||
torrents_cache[_domain] = [_torrent for _torrent in _torrents
|
||||
if not TorrentHelper().is_invalid(_torrent.torrent_info.enclosure)]
|
||||
# 缓存过滤掉无效种子(影视与音乐缓存分别处理)
|
||||
for _cache in (torrents_cache, music_cache):
|
||||
for _domain, _torrents in _cache.items():
|
||||
_cache[_domain] = [_torrent for _torrent in _torrents
|
||||
if not TorrentHelper().is_invalid(_torrent.torrent_info.enclosure)]
|
||||
|
||||
# 需要刷新的站点domain
|
||||
domains = []
|
||||
@@ -572,31 +701,47 @@ class TorrentsChain(ChainBase):
|
||||
else:
|
||||
# 如果某一页没有数据,说明已经到最后一页,停止获取
|
||||
break
|
||||
# 存在音乐订阅时,默认首页可能不包含音乐资源,需要额外抓取音乐专用入口
|
||||
if include_music and self._music_browse_paths(indexer):
|
||||
torrents = self.__append_music_browse_torrents(
|
||||
domain=domain, torrents=torrents
|
||||
)
|
||||
else:
|
||||
# 刷新RSS种子
|
||||
torrents: List[TorrentInfo] = self.rss(domain=domain)
|
||||
# 按pubdate降序排列
|
||||
torrents.sort(key=lambda x: x.pubdate or '', reverse=True)
|
||||
# 取前N条
|
||||
torrents = torrents[:settings.CONF.refresh]
|
||||
if torrents:
|
||||
# 音乐与影视按同一公共参数独立计算刷新配额,并分别写入各自缓存,音乐不会被影视资源挤出
|
||||
music_torrents = [
|
||||
t for t in torrents if t.category == MediaType.MUSIC.value
|
||||
][:settings.CONF.refresh]
|
||||
torrents = [
|
||||
t for t in torrents if t.category != MediaType.MUSIC.value
|
||||
][:settings.CONF.refresh]
|
||||
if torrents or music_torrents:
|
||||
if __is_no_cache_site(domain):
|
||||
# 不需要缓存的站点,直接处理
|
||||
logger.info(f'{indexer.get("name")} 有 {len(torrents)} 个种子 (不缓存)')
|
||||
logger.info(f'{indexer.get("name")} 有 {len(torrents) + len(music_torrents)} 个种子 (不缓存)')
|
||||
torrents_cache[domain] = []
|
||||
music_cache[domain] = []
|
||||
else:
|
||||
# 过滤出没有处理过的种子 - 优化:使用集合查找,避免重复创建字符串列表
|
||||
cached_signatures = {f'{t.torrent_info.title}{t.torrent_info.description}'
|
||||
for t in torrents_cache.get(domain) or []}
|
||||
torrents = [torrent for torrent in torrents
|
||||
if f'{torrent.title}{torrent.description}' not in cached_signatures]
|
||||
if torrents:
|
||||
logger.info(f'{indexer.get("name")} 有 {len(torrents)} 个新种子')
|
||||
# 音乐种子对照音乐独立缓存去重
|
||||
music_signatures = {f'{t.torrent_info.title}{t.torrent_info.description}'
|
||||
for t in music_cache.get(domain) or []}
|
||||
music_torrents = [torrent for torrent in music_torrents
|
||||
if f'{torrent.title}{torrent.description}' not in music_signatures]
|
||||
if torrents or music_torrents:
|
||||
logger.info(f'{indexer.get("name")} 有 {len(torrents) + len(music_torrents)} 个新种子')
|
||||
else:
|
||||
logger.info(f'{indexer.get("name")} 没有新种子')
|
||||
continue
|
||||
try:
|
||||
for torrent in torrents:
|
||||
for torrent in torrents + music_torrents:
|
||||
if global_vars.is_system_stopped:
|
||||
break
|
||||
if not torrent.enclosure:
|
||||
@@ -647,29 +792,39 @@ class TorrentsChain(ChainBase):
|
||||
# 如果未识别到媒体信息,设置初始失败次数为1
|
||||
if not mediainfo or not all(resolve_media_identity(media=mediainfo)):
|
||||
context.media_recognize_fail_count = 1
|
||||
# 添加到缓存
|
||||
if not torrents_cache.get(domain):
|
||||
torrents_cache[domain] = [context]
|
||||
# 添加到缓存:音乐进入独立缓存,与影视分开存储
|
||||
if torrent.category == MediaType.MUSIC.value:
|
||||
target_cache = music_cache
|
||||
else:
|
||||
torrents_cache[domain].append(context)
|
||||
# 如果超过了限制条数则移除掉前面的
|
||||
if len(torrents_cache[domain]) > settings.CONF.torrents:
|
||||
torrents_cache[domain] = torrents_cache[domain][-settings.CONF.torrents:]
|
||||
target_cache = torrents_cache
|
||||
if not target_cache.get(domain):
|
||||
target_cache[domain] = [context]
|
||||
else:
|
||||
target_cache[domain].append(context)
|
||||
# 如果超过了限制条数则移除掉前面的,音乐与影视各自独立计算配额
|
||||
if len(target_cache[domain]) > settings.CONF.torrents:
|
||||
target_cache[domain] = target_cache[domain][-settings.CONF.torrents:]
|
||||
finally:
|
||||
torrents.clear()
|
||||
music_torrents.clear()
|
||||
del torrents
|
||||
del music_torrents
|
||||
else:
|
||||
logger.info(f'{indexer.get("name")} 没有获取到种子')
|
||||
|
||||
# 保存缓存到本地
|
||||
# 保存缓存到本地,影视与音乐分别存储
|
||||
if stype == "spider":
|
||||
self.save_cache(torrents_cache, self._spider_file)
|
||||
self.save_cache(music_cache, self._music_spider_file)
|
||||
else:
|
||||
self.save_cache(torrents_cache, self._rss_file)
|
||||
self.save_cache(music_cache, self._music_rss_file)
|
||||
|
||||
# 去除不在站点范围内的缓存种子
|
||||
if sites and torrents_cache:
|
||||
torrents_cache = {k: v for k, v in torrents_cache.items() if k in domains}
|
||||
if sites and music_cache:
|
||||
music_cache = {k: v for k, v in music_cache.items() if k in domains}
|
||||
|
||||
if progress_callback:
|
||||
progress_callback(
|
||||
@@ -678,6 +833,11 @@ class TorrentsChain(ChainBase):
|
||||
data={"total": total_indexers, "finished": total_indexers},
|
||||
)
|
||||
|
||||
# 订阅匹配需要完整候选,音乐独立缓存在返回值中按站点合并
|
||||
for _domain, _contexts in music_cache.items():
|
||||
if _contexts:
|
||||
torrents_cache.setdefault(_domain, []).extend(_contexts)
|
||||
|
||||
return torrents_cache
|
||||
|
||||
@staticmethod
|
||||
|
||||
@@ -165,6 +165,8 @@ class SiteSpider:
|
||||
# 种子搜索相对路径
|
||||
paths = self.search.get('paths', [])
|
||||
torrentspath = ""
|
||||
# 是否选中了媒体类型专用路径,浏览模式下专用路径优先于 browse 配置
|
||||
typed_path_selected = False
|
||||
if len(paths) == 1:
|
||||
torrentspath = paths[0].get('path', '')
|
||||
else:
|
||||
@@ -183,6 +185,7 @@ class SiteSpider:
|
||||
not expected_type and path_type == "all"
|
||||
):
|
||||
torrentspath = path.get('path', '')
|
||||
typed_path_selected = bool(expected_type and path_type == expected_type)
|
||||
break
|
||||
if not torrentspath:
|
||||
torrentspath = fallback_path
|
||||
@@ -282,16 +285,18 @@ class SiteSpider:
|
||||
"page": self.page or 0,
|
||||
"keyword": ""
|
||||
}
|
||||
# 有单独浏览路径
|
||||
if self.browse:
|
||||
# 有单独浏览路径;指定了媒体类型专用路径时不覆盖,确保音乐等专用入口可达
|
||||
if self.browse and not typed_path_selected:
|
||||
torrentspath = self.browse.get("path")
|
||||
if self.browse.get("start"):
|
||||
start_page = int(self.browse.get("start")) + int(self.page or 0)
|
||||
inputs_dict.update({
|
||||
"page": start_page
|
||||
})
|
||||
elif self.page:
|
||||
torrentspath = torrentspath + f"?page={self.page}"
|
||||
elif self.page and "{page}" not in str(torrentspath):
|
||||
# 按路径是否已带查询参数选择连接符,避免拼出两个问号的非法地址
|
||||
separator = "&" if "?" in str(torrentspath) else "?"
|
||||
torrentspath = torrentspath + f"{separator}page={self.page}"
|
||||
# 搜索Url
|
||||
searchurl = self.domain + str(torrentspath).format(**inputs_dict)
|
||||
|
||||
|
||||
@@ -19,6 +19,7 @@ from app.core.context import (
|
||||
from app.core.meta import MetaBase, MetaMusic
|
||||
from app.log import logger
|
||||
from app.modules import _ModuleBase
|
||||
from app.modules.musicbrainz.music_cache import MusicBrainzCache
|
||||
from app.schemas.types import MediaRecognizeType, MediaType, ModuleType
|
||||
from app.utils.http import RequestUtils
|
||||
from app.utils.zhconv import convert as zhconv_convert
|
||||
@@ -36,6 +37,8 @@ class MusicBrainzModule(_ModuleBase):
|
||||
_request_interval = 1.0
|
||||
_request_lock = threading.Lock()
|
||||
_last_request_at = 0.0
|
||||
# 本地识别缓存,由模块管理器初始化时挂载
|
||||
cache: MusicBrainzCache = None
|
||||
# 全局复用 HTTP 会话:keep-alive 省去每次请求的 DNS+TLS 握手(约 6s → 0.4s)
|
||||
_session: Optional[Session] = None
|
||||
_session_lock = threading.Lock()
|
||||
@@ -72,14 +75,32 @@ class MusicBrainzModule(_ModuleBase):
|
||||
)
|
||||
|
||||
def init_module(self) -> None:
|
||||
"""初始化无状态的 MusicBrainz 模块。"""
|
||||
"""初始化 MusicBrainz 模块并挂载本地识别缓存。"""
|
||||
self.cache = MusicBrainzCache()
|
||||
|
||||
def init_setting(self) -> Optional[Tuple[str, Union[str, bool]]]:
|
||||
"""MusicBrainz 无需独立密钥或启用开关。"""
|
||||
return None
|
||||
|
||||
def stop(self) -> None:
|
||||
"""停止模块;当前实现没有需要释放的持久资源。"""
|
||||
"""停止模块,退出前持久化识别缓存。"""
|
||||
if self.cache:
|
||||
try:
|
||||
self.cache.save()
|
||||
except Exception as err:
|
||||
logger.error(f"保存音乐识别缓存失败:{str(err)}")
|
||||
|
||||
def scheduler_job(self) -> None:
|
||||
"""定时任务,每10分钟持久化一次音乐识别缓存。"""
|
||||
if self.cache:
|
||||
self.cache.save()
|
||||
|
||||
def clear_cache(self) -> None:
|
||||
"""响应全局缓存清理事件,清空音乐识别缓存。"""
|
||||
logger.info("开始清除音乐识别缓存 ...")
|
||||
if self.cache:
|
||||
self.cache.clear()
|
||||
logger.info("音乐识别缓存清除完成")
|
||||
|
||||
def test(self) -> Tuple[bool, str]:
|
||||
"""测试 MusicBrainz 搜索接口连通性。"""
|
||||
@@ -610,11 +631,23 @@ class MusicBrainzModule(_ModuleBase):
|
||||
if source == self._source and mediaid:
|
||||
return self.recognize_music(source, str(mediaid))
|
||||
return None
|
||||
# 识别缓存命中直接响应,避免重复搜索占用 MusicBrainz 限流配额
|
||||
cache_enabled = bool(kwargs.get("cache", True))
|
||||
if cache_enabled and self.cache:
|
||||
cached_info = self.cache.get(meta)
|
||||
if cached_info:
|
||||
if cached_info.media_id:
|
||||
logger.info(f"{meta.title} 使用音乐识别缓存:{cached_info.title}")
|
||||
else:
|
||||
logger.info(f"{meta.title} 使用音乐识别缓存:无法识别")
|
||||
cached_info.recognize_cache_hit = True
|
||||
return cached_info
|
||||
# 携带数据源与原生 ID 的请求优先按详情识别
|
||||
resolved_source = source or meta.media_source
|
||||
if resolved_source and (mediaid or meta.media_id):
|
||||
info = self.recognize_music(resolved_source, str(mediaid or meta.media_id))
|
||||
if info:
|
||||
self._update_recognize_cache(meta, info)
|
||||
return info
|
||||
# 无身份时按标题搜索并挑选可信候选,检索不到时返回元数据兑底
|
||||
# 文件识别只能从 Recording 中挑选,专辑或艺术家同名结果不能成为音轨身份。
|
||||
@@ -626,7 +659,38 @@ class MusicBrainzModule(_ModuleBase):
|
||||
if not matched and meta.artists:
|
||||
albums = self._search_albums(meta, limit=10)
|
||||
matched = self._select_album_candidate(meta, albums)
|
||||
return matched or self._info_from_meta(meta)
|
||||
result = matched or self._info_from_meta(meta)
|
||||
# 无远端身份的兑底结果同样入缓存,避免批量识别时反复搜索同一文件
|
||||
self._update_recognize_cache(meta, result)
|
||||
return result
|
||||
|
||||
def _update_recognize_cache(self, meta: MetaMusic, info: Optional[MusicInfo]) -> None:
|
||||
"""识别完成后把结果写入本地识别缓存,未挂载缓存时静默跳过。"""
|
||||
if self.cache:
|
||||
self.cache.update(meta, info)
|
||||
|
||||
def update_recognize_cache(
|
||||
self,
|
||||
meta: MetaBase,
|
||||
mediainfo: MusicInfo,
|
||||
) -> Optional[bool]:
|
||||
"""回填音乐本地识别缓存,共享识别成功后避免重复回查。"""
|
||||
if not meta or not mediainfo:
|
||||
return None
|
||||
if not isinstance(meta, MetaMusic) or not isinstance(mediainfo, MusicInfo):
|
||||
return None
|
||||
if mediainfo.source != self._source:
|
||||
return None
|
||||
self._update_recognize_cache(meta, mediainfo)
|
||||
return True
|
||||
|
||||
async def async_update_recognize_cache(
|
||||
self,
|
||||
meta: MetaBase,
|
||||
mediainfo: MusicInfo,
|
||||
) -> Optional[bool]:
|
||||
"""异步回填音乐本地识别缓存。"""
|
||||
return self.update_recognize_cache(meta=meta, mediainfo=mediainfo)
|
||||
|
||||
async def async_recognize_media(
|
||||
self,
|
||||
|
||||
234
app/modules/musicbrainz/music_cache.py
Normal file
234
app/modules/musicbrainz/music_cache.py
Normal file
@@ -0,0 +1,234 @@
|
||||
import pickle
|
||||
import traceback
|
||||
from math import ceil
|
||||
from threading import RLock
|
||||
from time import time
|
||||
from typing import Optional
|
||||
|
||||
from app.core.cache import FileCache, TTLCache
|
||||
from app.core.config import settings
|
||||
from app.core.context import MusicInfo
|
||||
from app.core.meta import MetaMusic
|
||||
from app.log import logger
|
||||
from app.utils.singleton import WeakSingleton
|
||||
|
||||
lock = RLock()
|
||||
PERSISTENCE_VERSION = 1
|
||||
PERSISTENCE_REGION = "recognize"
|
||||
PERSISTENCE_KEY = "musicbrainz"
|
||||
|
||||
|
||||
class MusicBrainzCache(metaclass=WeakSingleton):
|
||||
"""
|
||||
MusicBrainz识别缓存数据
|
||||
{
|
||||
"source": '',
|
||||
"media_id": '',
|
||||
"title": '',
|
||||
"artists": [],
|
||||
"album": '',
|
||||
"year": '',
|
||||
"music_type": ''
|
||||
}
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
"""初始化音乐识别缓存并恢复未过期的持久化数据。"""
|
||||
self.maxsize = settings.CONF.musicbrainz
|
||||
self.ttl = settings.CONF.meta
|
||||
self.region = "__musicbrainz_cache__"
|
||||
self._cache = TTLCache(region=self.region, maxsize=self.maxsize, ttl=self.ttl)
|
||||
self._expires_at: dict[str, float] = {}
|
||||
self._dirty = False
|
||||
self._file_cache = None
|
||||
if not self._cache.is_redis():
|
||||
self._file_cache = FileCache(base=settings.CACHE_PATH, ttl=self.ttl)
|
||||
self._restore()
|
||||
|
||||
def _restore(self) -> None:
|
||||
"""从统一文件缓存恢复仍在有效期内的音乐识别数据。"""
|
||||
try:
|
||||
content = self._file_cache.get(PERSISTENCE_KEY, region=PERSISTENCE_REGION)
|
||||
if not content:
|
||||
return
|
||||
payload = pickle.loads(content)
|
||||
now = time()
|
||||
if (
|
||||
not isinstance(payload, dict)
|
||||
or payload.get("version") != PERSISTENCE_VERSION
|
||||
or not isinstance(payload.get("items"), dict)
|
||||
):
|
||||
return
|
||||
|
||||
for key, item in payload["items"].items():
|
||||
if not isinstance(item, dict):
|
||||
self._dirty = True
|
||||
continue
|
||||
value = item.get("value")
|
||||
expires_at = item.get("expires_at")
|
||||
if not isinstance(value, dict) or not isinstance(expires_at, (int, float)):
|
||||
self._dirty = True
|
||||
continue
|
||||
remaining_ttl = expires_at - now
|
||||
if remaining_ttl <= 0:
|
||||
self._dirty = True
|
||||
continue
|
||||
self._cache.set(key, value, ttl=ceil(remaining_ttl))
|
||||
self._expires_at[key] = expires_at
|
||||
except Exception as err:
|
||||
logger.error(f"加载音乐识别缓存失败:{str(err)} - {traceback.format_exc()}")
|
||||
|
||||
def _set(self, key: str, value: dict) -> None:
|
||||
"""写入单条音乐识别缓存并记录其独立过期时间。"""
|
||||
self._cache.set(key, value)
|
||||
if not self._cache.is_redis():
|
||||
self._expires_at[key] = time() + self.ttl
|
||||
self._dirty = True
|
||||
|
||||
def clear(self):
|
||||
"""
|
||||
清空所有音乐识别缓存
|
||||
"""
|
||||
with lock:
|
||||
self._cache.clear()
|
||||
self._expires_at.clear()
|
||||
self._dirty = True
|
||||
self.save(force=True)
|
||||
|
||||
def list_items(self) -> list[dict]:
|
||||
"""
|
||||
返回可供管理界面展示的音乐识别缓存列表。
|
||||
"""
|
||||
with lock:
|
||||
cache_items = []
|
||||
for key, value in self._cache.items():
|
||||
if not isinstance(value, dict):
|
||||
continue
|
||||
cache_items.append({
|
||||
"key": key,
|
||||
"media_id": value.get("media_id") or "",
|
||||
"title": value.get("title") or "",
|
||||
"artists": value.get("artists") or [],
|
||||
"album": value.get("album") or "",
|
||||
"year": value.get("year") or "",
|
||||
"music_type": value.get("music_type") or "recording",
|
||||
"cover_url": value.get("cover_url") or "",
|
||||
})
|
||||
return sorted(cache_items, key=lambda item: item["key"])
|
||||
|
||||
@staticmethod
|
||||
def __get_key(meta: MetaMusic) -> str:
|
||||
"""
|
||||
获取缓存KEY,携带数据源原生 ID 时以 ID 为准身份
|
||||
"""
|
||||
artists = "/".join(meta.artists or [])
|
||||
return f"[音乐]{meta.media_id or meta.title}-{artists}-{meta.album}-{meta.year}"
|
||||
|
||||
def get(self, meta: MetaMusic) -> Optional[MusicInfo]:
|
||||
"""
|
||||
根据元数据获取缓存的音乐识别结果
|
||||
@param meta: 音乐元数据
|
||||
@return: 缓存命中的音乐信息,未命中返回 None
|
||||
"""
|
||||
key = self.__get_key(meta)
|
||||
with lock:
|
||||
cache_data = self._cache.get(key)
|
||||
if not cache_data and self._expires_at.pop(key, None) is not None:
|
||||
self._dirty = True
|
||||
if not cache_data:
|
||||
return None
|
||||
try:
|
||||
return MusicInfo.from_dict(cache_data)
|
||||
except Exception as err:
|
||||
logger.error(f"解析音乐识别缓存失败:{str(err)}")
|
||||
return None
|
||||
|
||||
def delete(self, key: str) -> dict:
|
||||
"""
|
||||
删除缓存信息
|
||||
@param key: 缓存key
|
||||
@return: 被删除的缓存内容
|
||||
"""
|
||||
with lock:
|
||||
cache_data = self._cache.get(key)
|
||||
if cache_data:
|
||||
self._cache.delete(key)
|
||||
self._expires_at.pop(key, None)
|
||||
self._dirty = True
|
||||
self.save(force=True)
|
||||
return cache_data
|
||||
return {}
|
||||
|
||||
def update(self, meta: MetaMusic, info: Optional[MusicInfo]) -> None:
|
||||
"""
|
||||
新增或更新缓存条目,无远端身份的兜底结果也写入内存负缓存,
|
||||
避免批量识别时反复请求 MusicBrainz 触发限流
|
||||
"""
|
||||
if not meta or not info:
|
||||
return
|
||||
key = self.__get_key(meta)
|
||||
cache_data = info.to_dict()
|
||||
# 上游原始响应体积大且不参与身份恢复,不入缓存
|
||||
cache_data.pop("raw_data", None)
|
||||
with lock:
|
||||
self._set(key, cache_data)
|
||||
|
||||
def save(self, force: bool = False) -> None:
|
||||
"""
|
||||
使用统一文件缓存保存未过期的音乐识别数据。
|
||||
"""
|
||||
if self._cache.is_redis():
|
||||
return
|
||||
if not self._file_cache:
|
||||
return
|
||||
with lock:
|
||||
now = time()
|
||||
cache_items = dict(self._cache.items())
|
||||
active_keys = set(cache_items)
|
||||
stale_keys = set(self._expires_at) - active_keys
|
||||
if stale_keys:
|
||||
for key in stale_keys:
|
||||
self._expires_at.pop(key, None)
|
||||
self._dirty = True
|
||||
|
||||
persisted_items = {}
|
||||
for key, value in cache_items.items():
|
||||
expires_at = self._expires_at.get(key)
|
||||
if expires_at is None:
|
||||
expires_at = now + self.ttl
|
||||
self._expires_at[key] = expires_at
|
||||
self._dirty = True
|
||||
# 负缓存只留在内存,重启后允许重新尝试识别
|
||||
if expires_at <= now or not value.get("media_id"):
|
||||
continue
|
||||
persisted_items[key] = {
|
||||
"value": value,
|
||||
"expires_at": expires_at,
|
||||
}
|
||||
|
||||
if not force and not self._dirty:
|
||||
return
|
||||
|
||||
try:
|
||||
if persisted_items:
|
||||
payload = {
|
||||
"version": PERSISTENCE_VERSION,
|
||||
"items": persisted_items,
|
||||
}
|
||||
self._file_cache.set(
|
||||
PERSISTENCE_KEY,
|
||||
pickle.dumps(payload, pickle.HIGHEST_PROTOCOL),
|
||||
region=PERSISTENCE_REGION,
|
||||
)
|
||||
else:
|
||||
self._file_cache.delete(PERSISTENCE_KEY, region=PERSISTENCE_REGION)
|
||||
self._dirty = False
|
||||
except Exception as err:
|
||||
logger.error(f"保存音乐识别缓存失败:{str(err)} - {traceback.format_exc()}")
|
||||
|
||||
def __del__(self):
|
||||
"""实例释放前保存非 Redis 缓存。"""
|
||||
try:
|
||||
self.save()
|
||||
except Exception:
|
||||
pass
|
||||
Reference in New Issue
Block a user