From 96118e73e749aa952f8f5c926a0525ee7306d973 Mon Sep 17 00:00:00 2001 From: jxxghp Date: Sat, 15 Aug 2026 08:03:55 +0800 Subject: [PATCH] refactor(string): split utilities by responsibility --- app/adapters/external/cookiecloud.py | 9 +- app/adapters/external/market.py | 2 +- app/agent/tools/impl/_torrent_search_utils.py | 4 +- app/agent/tools/impl/list_directory.py | 7 +- app/agent/tools/impl/update_site.py | 4 +- app/api/endpoints/site.py | 17 +- app/api/endpoints/storage.py | 4 +- app/application/messaging/message.py | 11 +- app/application/security/cookie.py | 4 +- app/application/site/__init__.py | 24 - app/application/torrent.py | 26 +- app/chain/download.py | 18 +- app/chain/media.py | 6 +- app/chain/message.py | 10 +- app/chain/scraping.py | 2 +- app/chain/search.py | 4 +- app/chain/site.py | 39 +- app/chain/torrents.py | 7 +- app/chain/transfer.py | 19 +- app/domain/context.py | 4 +- app/domain/episode.py | 39 + app/domain/meta/infopath.py | 4 +- app/domain/meta/metaanime.py | 19 +- app/domain/meta/metabase.py | 6 +- app/domain/meta/metavideo.py | 8 +- app/domain/site.py | 42 +- app/domain/string.py | 903 ------------------ app/domain/title.py | 88 ++ app/domain/torrent.py | 14 + app/foundation/dom.py | 7 +- app/foundation/size.py | 79 ++ app/foundation/temporal.py | 111 +++ app/foundation/text.py | 235 ++++- app/foundation/url.py | 76 ++ app/helper/.resource-compat | 2 - app/modules/discord/discord.py | 4 +- app/modules/filemanager/__init__.py | 6 +- app/modules/filemanager/storages/alipan.py | 6 +- app/modules/filemanager/storages/rclone.py | 6 +- app/modules/filemanager/storages/u115.py | 6 +- app/modules/filter/__init__.py | 4 +- app/modules/indexer/__init__.py | 15 +- app/modules/indexer/parser/__init__.py | 4 +- app/modules/indexer/parser/bitpt.py | 13 +- app/modules/indexer/parser/discuz.py | 23 +- app/modules/indexer/parser/file_list.py | 25 +- app/modules/indexer/parser/gazelle.py | 27 +- app/modules/indexer/parser/hddolby.py | 4 +- app/modules/indexer/parser/ipt_project.py | 25 +- app/modules/indexer/parser/mtorrent.py | 4 +- app/modules/indexer/parser/nexus_audiences.py | 34 +- app/modules/indexer/parser/nexus_hhanclub.py | 15 +- app/modules/indexer/parser/nexus_php.py | 48 +- app/modules/indexer/parser/nexus_rabbit.py | 21 +- app/modules/indexer/parser/rousi.py | 9 +- app/modules/indexer/parser/small_horse.py | 25 +- app/modules/indexer/parser/sunnypt.py | 6 +- app/modules/indexer/parser/tnode.py | 4 +- app/modules/indexer/parser/torrent_leech.py | 25 +- app/modules/indexer/parser/unit3d.py | 23 +- app/modules/indexer/parser/yema.py | 4 +- app/modules/indexer/parser/zhixing.py | 4 +- app/modules/indexer/spider/__init__.py | 10 +- app/modules/indexer/spider/haidan.py | 7 +- app/modules/indexer/spider/hddolby.py | 4 +- app/modules/indexer/spider/mtorrent.py | 13 +- app/modules/indexer/spider/rousi.py | 9 +- app/modules/indexer/spider/sunnypt.py | 6 +- app/modules/indexer/spider/tnode.py | 4 +- app/modules/indexer/spider/torrentleech.py | 9 +- app/modules/indexer/spider/yema.py | 6 +- app/modules/qbittorrent/__init__.py | 15 +- app/modules/qbittorrent/qbittorrent.py | 7 +- app/modules/qqbot/qqbot.py | 4 +- app/modules/rtorrent/__init__.py | 15 +- app/modules/slack/slack.py | 6 +- app/modules/synologychat/synologychat.py | 7 +- app/modules/telegram/telegram.py | 4 +- app/modules/themoviedb/tmdbapi.py | 12 +- app/modules/transmission/__init__.py | 12 +- app/modules/vocechat/vocechat.py | 4 +- app/modules/wechat/wechat.py | 4 +- app/modules/wechat/wechatbot.py | 4 +- app/modules/wechatclawbot/wechatclawbot.py | 4 +- app/runtime/compat/manifest.py | 10 +- app/sdk/string.py | 165 ++++ app/sdk/utilities.py | 2 +- docs/backend-module-refactor-compatibility.md | 5 +- docs/rules/05-architecture.md | 26 +- tests/test_architecture_dependencies.py | 36 + tests/test_downloader_path_mapping.py | 51 +- tests/test_nexus_audiences_parser.py | 6 +- tests/test_plugin_sdk.py | 8 +- tests/test_qbittorrent_compat.py | 71 +- tests/test_resource_v3.py | 37 - tests/test_site_domain.py | 42 + tests/test_string.py | 37 +- tests/test_string_compat.py | 120 +++ tests/test_torrent_leech_parser.py | 6 +- 99 files changed, 1608 insertions(+), 1424 deletions(-) create mode 100644 app/domain/episode.py delete mode 100644 app/domain/string.py create mode 100644 app/domain/title.py create mode 100644 app/domain/torrent.py create mode 100644 app/foundation/size.py create mode 100644 app/foundation/temporal.py delete mode 100644 app/helper/.resource-compat create mode 100644 app/sdk/string.py create mode 100644 tests/test_site_domain.py create mode 100644 tests/test_string_compat.py diff --git a/app/adapters/external/cookiecloud.py b/app/adapters/external/cookiecloud.py index 2267af140..07061ffac 100644 --- a/app/adapters/external/cookiecloud.py +++ b/app/adapters/external/cookiecloud.py @@ -5,7 +5,8 @@ from app.runtime.config import settings from app.runtime.log import logger from app.foundation.crypto import CryptoJsUtils, HashUtils from app.adapters.network.http import RequestUtils -from app.domain.string import StringUtils +from app.domain import site as site_rules +from app.foundation import text as text_tools from app.foundation.url import UrlUtils @@ -23,8 +24,8 @@ class CookieCloudHelper: 同步CookieCloud配置项 """ self._server = UrlUtils.standardize_base_url(settings.COOKIECLOUD_HOST) - self._key = StringUtils.safe_strip(settings.COOKIECLOUD_KEY) - self._password = StringUtils.safe_strip(settings.COOKIECLOUD_PASSWORD) + self._key = text_tools.strip_optional(settings.COOKIECLOUD_KEY) + self._password = text_tools.strip_optional(settings.COOKIECLOUD_PASSWORD) self._enable_local = settings.COOKIECLOUD_ENABLE_LOCAL self._local_path = settings.COOKIE_PATH @@ -83,7 +84,7 @@ class CookieCloudHelper: domain_groups = {} for site, cookies in contents.items(): for cookie in cookies: - domain_key = StringUtils.get_url_domain(cookie.get("domain")) + domain_key = site_rules.extract_domain(cookie.get("domain")) if not domain_groups.get(domain_key): domain_groups[domain_key] = [cookie] else: diff --git a/app/adapters/external/market.py b/app/adapters/external/market.py index 7c87eaf14..0dffd8f31 100644 --- a/app/adapters/external/market.py +++ b/app/adapters/external/market.py @@ -36,7 +36,7 @@ from app.runtime.log import logger from app.schemas.types import SystemConfigKey from app.adapters.network.http import RequestUtils, AsyncRequestUtils from app.foundation.singleton import WeakSingleton -from app.domain.string import StringUtils + from app.foundation.version import compare_version from app.adapters.system.host import SystemUtils from app.foundation.url import UrlUtils diff --git a/app/agent/tools/impl/_torrent_search_utils.py b/app/agent/tools/impl/_torrent_search_utils.py index f25c98155..f12cff5b0 100644 --- a/app/agent/tools/impl/_torrent_search_utils.py +++ b/app/agent/tools/impl/_torrent_search_utils.py @@ -6,7 +6,7 @@ from typing import List, Optional from app.domain.context import Context from app.schemas.types import MediaType, media_type_to_agent from app.foundation.crypto import HashUtils -from app.domain.string import StringUtils +from app.foundation import size as size_tools from ._music_utils import simplify_music_info SEARCH_RESULT_CACHE_FILE = "__search_result__" @@ -150,7 +150,7 @@ def simplify_search_result( if torrent_info: simplified["torrent_info"] = { "title": torrent_info.title, - "size": StringUtils.format_size(torrent_info.size), + "size": size_tools.format_size(torrent_info.size), "seeders": torrent_info.seeders, "peers": torrent_info.peers, "site_name": torrent_info.site_name, diff --git a/app/agent/tools/impl/list_directory.py b/app/agent/tools/impl/list_directory.py index aa3c68722..bb2e23d13 100644 --- a/app/agent/tools/impl/list_directory.py +++ b/app/agent/tools/impl/list_directory.py @@ -12,7 +12,8 @@ from app.agent.tools.tags import ToolTag from app.chain.storage import StorageChain from app.runtime.log import logger from app.schemas.file import FileItem -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation import text as text_tools DEFAULT_DIRECTORY_PAGE_SIZE = 50 @@ -100,7 +101,7 @@ class ListDirectoryTool(MoviePilotTool): file_list.sort( key=lambda x: ( 0 if x.type == "dir" else 1, - StringUtils.natural_sort_key(x.name or ""), + text_tools.natural_sort_key(x.name or ""), ) ) @@ -115,7 +116,7 @@ class ListDirectoryTool(MoviePilotTool): ] simplified_items = [] for item in limited_list: - size_str = StringUtils.str_filesize(item.size) if item.size else None + size_str = size_tools.format_compact_size(item.size) if item.size else None modify_time_str = None if item.modify_time: try: diff --git a/app/agent/tools/impl/update_site.py b/app/agent/tools/impl/update_site.py index df2552151..6a1ad9ed9 100644 --- a/app/agent/tools/impl/update_site.py +++ b/app/agent/tools/impl/update_site.py @@ -11,7 +11,7 @@ from app.runtime.events import eventmanager from app.db.oper.site import SiteOper from app.runtime.log import logger from app.schemas.types import EventType -from app.domain.string import StringUtils +from app.foundation import url as url_tools class UpdateSiteInput(BaseModel): @@ -141,7 +141,7 @@ class UpdateSiteTool(MoviePilotTool): # URL处理(需要校正格式) if url is not None: - _scheme, _netloc = StringUtils.get_url_netloc(url) + _scheme, _netloc = url_tools.split_netloc(url) site_dict["url"] = f"{_scheme}://{_netloc}/" if pri is not None: diff --git a/app/api/endpoints/site.py b/app/api/endpoints/site.py index 237dbb0d0..4987587c4 100644 --- a/app/api/endpoints/site.py +++ b/app/api/endpoints/site.py @@ -32,7 +32,8 @@ from app.application.site.sites import SitesHelper # pylint: disable=no-name-in from app.runtime.log import logger from app.scheduler import Scheduler from app.schemas.types import SystemConfigKey, EventType, MediaType -from app.domain.string import StringUtils +from app.domain import site as site_rules +from app.foundation import url as url_tools router = ResponseAPIRouter() @@ -117,7 +118,7 @@ async def read_sites_by_media_type( continue if indexer.get("id") is not None: supported_ids.add(str(indexer.get("id"))) - domain = StringUtils.get_url_domain(indexer.get("domain")) + domain = site_rules.extract_domain(indexer.get("domain")) if domain: supported_domains.add(domain) @@ -146,7 +147,7 @@ async def add_site( return schemas.Response( success=False, message="用户未通过认证,无法使用站点功能!" ) - domain = StringUtils.get_url_domain(site_in.url) + domain = site_rules.extract_domain(site_in.url) site_info = await SitesHelper().async_get_indexer(domain) if not site_info: return schemas.Response( @@ -157,7 +158,7 @@ async def add_site( # 保存站点信息 site_in.domain = domain # 校正地址格式 - _scheme, _netloc = StringUtils.get_url_netloc(site_in.url) + _scheme, _netloc = url_tools.split_netloc(site_in.url) site_in.url = f"{_scheme}://{_netloc}/" site_in.name = site_info.get("name") site_in.id = None @@ -183,9 +184,9 @@ async def update_site( if not site: return schemas.Response(success=False, message="站点不存在") # 校正地址格式 - _scheme, _netloc = StringUtils.get_url_netloc(site_in.url) + _scheme, _netloc = url_tools.split_netloc(site_in.url) site_in.url = f"{_scheme}://{_netloc}/" - site_in.domain = StringUtils.get_url_domain(site_in.url) + site_in.domain = site_rules.extract_domain(site_in.url) await site.async_update(db, site_in.model_dump()) # 通知站点更新 await eventmanager.async_send_event( @@ -521,7 +522,7 @@ async def read_site_by_domain( """ 通过域名获取站点信息 """ - domain = StringUtils.get_url_domain(site_url) + domain = site_rules.extract_domain(site_url) site = await Site.async_get_by_domain(db, domain) if not site: raise HTTPException( @@ -544,7 +545,7 @@ async def read_statistic_by_domain( """ 通过域名获取站点统计信息 """ - domain = StringUtils.get_url_domain(site_url) + domain = site_rules.extract_domain(site_url) sitestatistic = await SiteStatistic.async_get_by_domain(db, domain) if sitestatistic: return sitestatistic diff --git a/app/api/endpoints/storage.py b/app/api/endpoints/storage.py index 9a1eee683..9f53a1b8e 100644 --- a/app/api/endpoints/storage.py +++ b/app/api/endpoints/storage.py @@ -22,7 +22,7 @@ from app.api.deps import ( ) from app.runtime.progress import ProgressHelper from app.schemas.types import ProgressKey -from app.domain.string import StringUtils +from app.foundation import text as text_tools router = ResponseAPIRouter() @@ -119,7 +119,7 @@ def list_files( _pat = re.compile(fnmatch.translate(keyword), re.IGNORECASE) file_list = [f for f in file_list if _pat.match(f.name or "")] if sort == "name": - file_list.sort(key=lambda x: StringUtils.natural_sort_key(x.name or "")) + file_list.sort(key=lambda x: text_tools.natural_sort_key(x.name or "")) else: file_list.sort(key=lambda x: x.modify_time or -math.inf, reverse=True) return file_list diff --git a/app/application/messaging/message.py b/app/application/messaging/message.py index cb3667c62..428fc9762 100644 --- a/app/application/messaging/message.py +++ b/app/application/messaging/message.py @@ -25,7 +25,8 @@ from app.schemas.tmdb import TmdbEpisode from app.schemas.transfer import TransferInfo from app.schemas.types import MUSIC_ENTITY_ALBUM, SystemConfigKey from app.foundation.singleton import Singleton, SingletonClass -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation.crypto import HashUtils class TemplateContextBuilder: @@ -346,7 +347,7 @@ class TemplateContextBuilder: return if torrentinfo.size: if str(torrentinfo.size).replace(".", "").isdigit(): - size = StringUtils.str_filesize(torrentinfo.size) + size = size_tools.format_compact_size(torrentinfo.size) else: size = torrentinfo.size else: @@ -391,7 +392,7 @@ class TemplateContextBuilder: ctx = { "transfer_type": transferinfo.transfer_type, "file_count": transferinfo.file_count, - "total_size": StringUtils.str_filesize(transferinfo.total_size), + "total_size": size_tools.format_compact_size(transferinfo.total_size), "err_msg": transferinfo.message, } context.update(ctx) @@ -472,9 +473,9 @@ class TemplateHelper(metaclass=SingletonClass): """ if isinstance(cuntent, dict): base_str = cuntent.get("title", '') + cuntent.get("text", '') - return StringUtils.md5_hash(json.dumps(base_str, sort_keys=True, ensure_ascii=False)) + return HashUtils.md5(json.dumps(base_str, sort_keys=True, ensure_ascii=False)) - return StringUtils.md5_hash(cuntent) + return HashUtils.md5(cuntent) def get_cache_context(self, cuntent: Union[str, dict]) -> Optional[dict]: """ diff --git a/app/application/security/cookie.py b/app/application/security/cookie.py index 2d1ef5abd..f5048a516 100644 --- a/app/application/security/cookie.py +++ b/app/application/security/cookie.py @@ -10,7 +10,7 @@ from app.application.security.twofactor import TwoFactorAuth from app.runtime.log import logger from app.adapters.network.http import RequestUtils from app.domain.site import SiteUtils -from app.domain.string import StringUtils +from app.foundation import url as url_tools class CookieHelper: @@ -355,4 +355,4 @@ class CookieHelper: return "" if imageurl.startswith("/"): imageurl = imageurl[1:] - return "%s/%s" % (StringUtils.get_base_url(siteurl), imageurl) + return "%s/%s" % (url_tools.base_url(siteurl), imageurl) diff --git a/app/application/site/__init__.py b/app/application/site/__init__.py index 4691296ba..d3384d66e 100644 --- a/app/application/site/__init__.py +++ b/app/application/site/__init__.py @@ -1,25 +1 @@ """站点目录、认证与索引资源的应用能力包。""" - -from importlib.machinery import EXTENSION_SUFFIXES -from pathlib import Path - - -def _include_legacy_resource_directory( - package_paths: list[str], package_dir: Path -) -> None: - """canonical 扩展缺失时允许读取旧 Docker 更新器写入的资源目录。""" - extension_names = tuple(f"sites{suffix}" for suffix in EXTENSION_SUFFIXES) - if any((package_dir / name).is_file() for name in extension_names): - return - - legacy_dir = package_dir.parent.parent / "helper" - if ( - legacy_dir.is_dir() - and any((legacy_dir / name).is_file() for name in extension_names) - and str(legacy_dir) not in package_paths - ): - # 旧镜像内固化的 mp_update.sh 无法随源码热更新,只在过渡场景扩展包搜索路径。 - package_paths.append(str(legacy_dir)) - - -_include_legacy_resource_directory(__path__, Path(__file__).resolve().parent) diff --git a/app/application/torrent.py b/app/application/torrent.py index 2e211bd20..f7e7ddc55 100644 --- a/app/application/torrent.py +++ b/app/application/torrent.py @@ -19,7 +19,9 @@ from app.runtime.log import logger from app.schemas.types import MediaType, SystemConfigKey from app.adapters.network.http import RequestUtils from app.schemas.media import resolve_media_identity -from app.domain.string import StringUtils +from app.domain import torrent as torrent_rules +from app.foundation import text as text_tools +from app.foundation.crypto import HashUtils _SIZE_UNIT = 1024 * 1024 @@ -85,7 +87,7 @@ class TorrentHelper: if url.startswith("magnet:"): return None, url, "", [], f"磁力链接" # 构建 torrent 种子文件的缓存路径 - cache_path = Path(StringUtils.md5_hash(url)).with_suffix(".torrent") + cache_path = Path(HashUtils.md5(url)).with_suffix(".torrent") # 缓存处理器 cache_backend = FileCache() # 读取缓存的种子文件 @@ -249,7 +251,7 @@ class TorrentHelper: return "", [] # 检查是否为磁力链接 - if StringUtils.is_magnet_link(torrent_content): + if torrent_rules.is_magnet_link(torrent_content): return "", [] try: @@ -423,15 +425,15 @@ class TorrentHelper: return True # 要匹配的媒体标题、原标题 media_titles = { - StringUtils.clear_upper(mediainfo.title), - StringUtils.clear_upper(mediainfo.original_title) + text_tools.normalize_upper(mediainfo.title), + text_tools.normalize_upper(mediainfo.original_title) } - {""} # 要匹配的媒体别名、译名 - media_names = {StringUtils.clear_upper(name) for name in mediainfo.names if name} + media_names = {text_tools.normalize_upper(name) for name in mediainfo.names if name} # 识别的种子中英文名 meta_names = { - StringUtils.clear_upper(torrent_meta.cn_name), - StringUtils.clear_upper(torrent_meta.en_name) + text_tools.normalize_upper(torrent_meta.cn_name), + text_tools.normalize_upper(torrent_meta.en_name) } - {""} # 比对种子识别类型 if torrent_meta.type == MediaType.TV and mediainfo.type != MediaType.TV: @@ -470,19 +472,19 @@ class TorrentHelper: # 标题拆分 if torrent_meta.org_string: # 只拆分出标题中的非英文单词进行匹配,英文单词容易误匹配(带空格的多个单词组合除外) - titles = [StringUtils.clear_upper(t) for t in re.split( + titles = [text_tools.normalize_upper(t) for t in re.split( r'[\s/【】.\[\]\-]+', torrent_meta.org_string - ) if not StringUtils.is_english_word(t)] + ) if not text_tools.is_english_word(t)] # 在标题中判断是否存在标题、原语种标题 if media_titles.intersection(titles): logger.info(f'{mediainfo.title} 通过标题匹配到资源:{torrent.site_name} - {torrent.title}') return True # 在副标题中(非英文单词)判断是否存在标题、原语种标题、别名、译名 if torrent.description: - subtitles = {StringUtils.clear_upper(t) for t in re.split( + subtitles = {text_tools.normalize_upper(t) for t in re.split( r'[\s/【】|]+', - torrent.description) if not StringUtils.is_english_word(t)} + torrent.description) if not text_tools.is_english_word(t)} if media_titles.intersection(subtitles) or media_names.intersection(subtitles): logger.info(f'{mediainfo.title} 通过副标题匹配到资源:{torrent.site_name} - {torrent.title},' f'副标题:{torrent.description}') diff --git a/app/chain/download.py b/app/chain/download.py index 394a3eb9d..7987cc1e8 100644 --- a/app/chain/download.py +++ b/app/chain/download.py @@ -39,7 +39,9 @@ from app.schemas.types import MUSIC_ENTITY_ALBUM, MediaSource, MediaType, Torren ChainEventType from app.adapters.network.http import RequestUtils from app.schemas.media import build_media_key, resolve_media_identity -from app.domain.string import StringUtils +from app.domain import episode as episode_rules +from app.foundation import size as size_tools +from app.foundation import text as text_tools from app.adapters.system.host import SystemUtils @@ -138,7 +140,7 @@ class DownloadChain(ChainBase): ).apply_path_context(file_path) track_identity: Union[int, str, None] = file_meta.track_number if track_identity is None: - track_identity = StringUtils.clear_upper(file_meta.title or file_path.stem) + track_identity = text_tools.normalize_upper(file_meta.title or file_path.stem) if track_identity in (None, ""): return None return file_meta.disc_number or 1, track_identity @@ -554,7 +556,7 @@ class DownloadChain(ChainBase): return meta.episode episode_list = getattr(meta, "episode_list", None) if episode_list: - return StringUtils.format_ep(list(episode_list)) + return episode_rules.format_ranges(list(episode_list)) return None @staticmethod @@ -669,7 +671,7 @@ class DownloadChain(ChainBase): media_source=media_source, media_id=media_id, seasons=getattr(meta, "season", None), - episodes=StringUtils.format_ep(list(episodes)) if episodes else self._format_failure_episodes(meta), + episodes=episode_rules.format_ranges(list(episodes)) if episodes else self._format_failure_episodes(meta), site=site if isinstance(site, int) else None, site_name=getattr(torrent, "site_name", None), torrent_id=self._torrent_resource_key(torrent), @@ -916,7 +918,7 @@ class DownloadChain(ChainBase): return (None, str(err)) if return_detail else None # 实际下载的集数 - download_episodes = StringUtils.format_ep(list(episodes)) if episodes else None + download_episodes = episode_rules.format_ranges(list(episodes)) if episodes else None if episodes is not None: context.selected_episodes = sorted(set(episodes)) elif _meta and _meta.episode_list: @@ -1427,12 +1429,12 @@ class DownloadChain(ChainBase): if complete_coverage_matched: logger.info( f"{meta.org_string} 解析文件集数已完整覆盖目标范围:" - f"{StringUtils.format_ep(sorted(required_episodes))}") + f"{episode_rules.format_ranges(sorted(required_episodes))}") if required_episodes and not complete_coverage_matched: missing_episodes = sorted(required_episodes.difference(torrent_episodes_set)) logger.info( f"{meta.org_string} 解析文件集数未覆盖目标范围," - f"缺少 {StringUtils.format_ep(missing_episodes)},先放弃这个种子") + f"缺少 {episode_rules.format_ranges(missing_episodes)},先放弃这个种子") continue if not required_episodes and need_total and len(torrent_episodes) < need_total: logger.info( @@ -1858,7 +1860,7 @@ class DownloadChain(ChainBase): index = 1 for torrent in torrents: messages.append(f"{index}. {torrent.title} " - f"{StringUtils.str_filesize(torrent.size)} " + f"{size_tools.format_compact_size(torrent.size)} " f"{round(torrent.progress, 1)}%") index += 1 self.post_message(Notification( diff --git a/app/chain/media.py b/app/chain/media.py index fe2addf16..c8afc0c1e 100644 --- a/app/chain/media.py +++ b/app/chain/media.py @@ -41,7 +41,7 @@ from app.domain.media import is_music_media_source from app.schemas.media import normalize_media_source, resolve_media_identity from app.foundation.singleton import Singleton from app.foundation.text import convert as zhconv_convert -from app.domain.string import StringUtils +from app.domain import title as title_rules recognize_lock = Lock() @@ -1412,7 +1412,7 @@ class MediaChain(ChainBase, metaclass=Singleton): """ # 提取要素 mtype, key_word, season_num, episode_num, year, content = ( - StringUtils.get_keyword(title) + title_rules.parse_search_keyword(title) ) # 识别 meta = MetaInfo(content) @@ -1889,7 +1889,7 @@ class MediaChain(ChainBase, metaclass=Singleton): """ # 提取要素 mtype, key_word, season_num, episode_num, year, content = ( - StringUtils.get_keyword(title) + title_rules.parse_search_keyword(title) ) # 识别 meta = MetaInfo(content) diff --git a/app/chain/message.py b/app/chain/message.py index 0f50f8420..7d21432bf 100644 --- a/app/chain/message.py +++ b/app/chain/message.py @@ -43,7 +43,9 @@ from app.schemas.system import TransferDirectoryConf from app.schemas.types import EventType, MessageChannel, MediaType from app.adapters.network.http import RequestUtils from app.schemas.media import build_media_key, resolve_media_identity -from app.domain.string import StringUtils +from app.domain import episode as episode_rules +from app.domain import title as title_rules +from app.foundation import url as url_tools class MessageChain(ChainBase): @@ -2562,9 +2564,9 @@ class MediaInteractionChain(ChainBase): return "ReSubscribe", re.sub(r"洗版[::\s]*", "", text) if text.startswith("搜索") or text.startswith("下载"): return "ReSearch", re.sub(r"(搜索|下载)[::\s]*", "", text) - if StringUtils.is_link(text): + if url_tools.is_link(text): return None, text - if not StringUtils.is_media_title_like(text): + if not title_rules.is_media_title_like(text): return None, text return "Search", text @@ -3683,7 +3685,7 @@ class MediaInteractionChain(ChainBase): season_map = no_exists.get(mediakey) or {} if show_missing_only: return [ - f"第 {sea} 季缺失 {StringUtils.str_series(no_exist.episodes) if no_exist.episodes else no_exist.total_episode} 集" + f"第 {sea} 季缺失 {episode_rules.compact_numbers(no_exist.episodes) if no_exist.episodes else no_exist.total_episode} 集" for sea, no_exist in season_map.items() ] return [ diff --git a/app/chain/scraping.py b/app/chain/scraping.py index 3b43c6f32..d1d1f8f1a 100644 --- a/app/chain/scraping.py +++ b/app/chain/scraping.py @@ -47,7 +47,7 @@ from app.domain.media import is_music_media_source from app.schemas.media import normalize_media_source, resolve_media_identity from app.runtime.reload import ConfigReloadMixin from app.foundation.singleton import Singleton -from app.domain.string import StringUtils + from app.chain.media import MediaChain diff --git a/app/chain/search.py b/app/chain/search.py index afcdbb83c..53f7c2965 100644 --- a/app/chain/search.py +++ b/app/chain/search.py @@ -36,7 +36,7 @@ from app.schemas.types import ( SystemConfigKey, ) from app.schemas.media import build_media_key, parse_media_key, resolve_media_identity -from app.domain.string import StringUtils +from app.foundation import size as size_tools from app.foundation.text import convert as zhconv_convert @@ -466,7 +466,7 @@ class SearchChain(ChainBase): "index": index, "title": torrent.torrent_info.title or "未知", "size": ( - StringUtils.format_size(torrent.torrent_info.size) + size_tools.format_size(torrent.torrent_info.size) if torrent.torrent_info.size else "0 B" ), diff --git a/app/chain/site.py b/app/chain/site.py index ead8fa06b..b3efe4e48 100644 --- a/app/chain/site.py +++ b/app/chain/site.py @@ -32,7 +32,10 @@ from app.schemas import MessageChannel, Notification, SiteUserData from app.schemas.types import EventType, NotificationType from app.adapters.network.http import RequestUtils from app.domain.site import SiteUtils -from app.domain.string import StringUtils +from app.domain import site as site_rules +from app.foundation import size as size_tools +from app.foundation import url as url_tools +from app.foundation.dom import DomUtils site_interaction_manager = SlashInteractionManager() @@ -71,7 +74,7 @@ class SiteChain(ChainBase): """ userdata: SiteUserData = self.run_module("refresh_userdata", site=site) if userdata: - SiteOper().update_userdata(domain=StringUtils.get_url_domain(site.get("domain")), + SiteOper().update_userdata(domain=site_rules.extract_domain(site.get("domain")), name=site.get("name"), payload=userdata.model_dump()) # 发送事件 @@ -229,7 +232,7 @@ class SiteChain(ChainBase): 判断站点是否已经登陆:m-team """ user_agent = site.ua or settings.USER_AGENT - domain = StringUtils.get_url_domain(site.url) + domain = site_rules.extract_domain(site.url) url = f"https://api.{domain}/api/member/profile" headers = { "User-Agent": user_agent, @@ -352,7 +355,7 @@ class SiteChain(ChainBase): """ 判断站点是否已经登陆:rousi """ - url = f"https://{StringUtils.get_url_domain(site.url)}/api/v1/profile" + url = f"https://{site_rules.extract_domain(site.url)}/api/v1/profile" headers = { "Content-Type": "application/json", "Accept": "application/json", @@ -391,7 +394,7 @@ class SiteChain(ChainBase): return favicon_url, None html = etree.HTML(html_text) try: - if StringUtils.is_valid_html_element(html): + if DomUtils.has_child_elements(html): fav_link = html.xpath('//head/link[contains(@rel, "icon")]/@href') if fav_link: favicon_url = urljoin(url, fav_link[0]) @@ -422,10 +425,10 @@ class SiteChain(ChainBase): """ 根据主域名获取索引器地址 """ - if StringUtils.get_url_domain(inx.get("domain")) == sub_domain: + if site_rules.extract_domain(inx.get("domain")) == sub_domain: return inx.get("domain") for ext_d in inx.get("ext_domains", []): - if StringUtils.get_url_domain(ext_d) == sub_domain: + if site_rules.extract_domain(ext_d) == sub_domain: return ext_d return sub_domain @@ -496,7 +499,7 @@ class SiteChain(ChainBase): _update_count += 1 elif indexer: if settings.COOKIECLOUD_BLACKLIST and any( - StringUtils.get_url_domain(domain) == StringUtils.get_url_domain(black_domain) for black_domain + site_rules.extract_domain(domain) == site_rules.extract_domain(black_domain) for black_domain in str(settings.COOKIECLOUD_BLACKLIST).split(",")): logger.warn(f"站点 {domain} 已在黑名单中,不添加站点") continue @@ -600,7 +603,7 @@ class SiteChain(ChainBase): if not domain: return if str(domain).startswith("http"): - domain = StringUtils.get_url_domain(domain) + domain = site_rules.extract_domain(domain) # 站点信息 siteoper = SiteOper() siteshelper = SitesHelper() @@ -642,7 +645,7 @@ class SiteChain(ChainBase): if not domain: return # 获取主域名中间那段 - domain_host = StringUtils.get_url_host(domain) + domain_host = url_tools.host_label(domain) # 查询以"site.domain_host"开头的配置项,并清除 systemconfig = SystemConfigOper() site_keys = systemconfig.all().keys() @@ -664,7 +667,7 @@ class SiteChain(ChainBase): if not domain: return if str(domain).startswith("http"): - domain = StringUtils.get_url_domain(domain) + domain = site_rules.extract_domain(domain) indexer = SitesHelper().get_indexer(domain) if not indexer: return @@ -678,7 +681,7 @@ class SiteChain(ChainBase): :return: (是否可用, 错误信息) """ # 检查域名是否可用 - domain = StringUtils.get_url_domain(url) + domain = site_rules.extract_domain(url) siteoper = SiteOper() site_info = siteoper.get_by_domain(domain) if not site_info: @@ -1188,7 +1191,7 @@ class SiteChain(ChainBase): "启用" if site.is_active else "禁用", "已配置" if site.cookie else "未配置", "是" if site.render else "否", - site.domain or StringUtils.get_url_domain(site.url or ""), + site.domain or site_rules.extract_domain(site.url or ""), ] for site in site_list ] @@ -1203,7 +1206,7 @@ class SiteChain(ChainBase): f"{site.id}. {site.name} | 状态:{'启用' if site.is_active else '禁用'}" f" | Cookie:{'已配置' if site.cookie else '未配置'}" f" | 渲染:{'是' if site.render else '否'}" - f" | 域名:{site.domain or StringUtils.get_url_domain(site.url or '')}" + f" | 域名:{site.domain or site_rules.extract_domain(site.url or '')}" ) return "\n".join(lines) @@ -1506,15 +1509,15 @@ class SiteChain(ChainBase): incDownloads += download messages[upload + (rand / 1000)] = ( f"【{site}】{updated_date}\n" - + f"上传量:{StringUtils.str_filesize(upload)}\n" - + f"下载量:{StringUtils.str_filesize(download)}\n" + + f"上传量:{size_tools.format_compact_size(upload)}\n" + + f"下载量:{size_tools.format_compact_size(download)}\n" + "————————————" ) if incDownloads or incUploads: sorted_messages = [messages[key] for key in sorted(messages.keys(), reverse=True)] sorted_messages.insert(0, f"【汇总】\n" - f"总上传:{StringUtils.str_filesize(incUploads)}\n" - f"总下载:{StringUtils.str_filesize(incDownloads)}\n" + f"总上传:{size_tools.format_compact_size(incUploads)}\n" + f"总下载:{size_tools.format_compact_size(incDownloads)}\n" f"————————————") self.post_message(Notification( channel=channel, diff --git a/app/chain/torrents.py b/app/chain/torrents.py index e6eb9488b..30c7bfc9b 100644 --- a/app/chain/torrents.py +++ b/app/chain/torrents.py @@ -20,7 +20,8 @@ from app.runtime.log import logger from app.schemas import Notification from app.schemas.types import SystemConfigKey, MessageChannel, NotificationType, MediaType from app.schemas.media import resolve_media_identity -from app.domain.string import StringUtils +from app.domain import site as site_rules +from app.foundation import text as text_tools class TorrentsChain(ChainBase): @@ -355,7 +356,7 @@ class TorrentsChain(ChainBase): """ 归一标题用于低置信标题兜底匹配。 """ - return (StringUtils.clear_upper(value or "") or "").strip() + return (text_tools.normalize_upper(value or "") or "").strip() def clear_torrents(self): """ @@ -603,7 +604,7 @@ class TorrentsChain(ChainBase): "current": indexer.get("id"), }, ) - domain = StringUtils.get_url_domain(indexer.get("domain")) + domain = site_rules.extract_domain(indexer.get("domain")) domains.append(domain) if stype == "spider": # 刷新首页种子 diff --git a/app/chain/transfer.py b/app/chain/transfer.py index 147877810..c934b739a 100755 --- a/app/chain/transfer.py +++ b/app/chain/transfer.py @@ -68,7 +68,8 @@ from app.application.transfer import TransferQueue, TransferTask from app.domain.media import normalize_music_type from app.schemas.media import normalize_media_source, resolve_media_identity from app.foundation.singleton import Singleton -from app.domain.string import StringUtils +from app.domain import episode as episode_rules +from app.foundation import text as text_tools from app.adapters.system.host import SystemUtils # 下载器锁 @@ -169,16 +170,16 @@ class JobManager: return "music", source, media_id, music_type artists = tuple( - StringUtils.clear_upper(artist) + text_tools.normalize_upper(artist) for artist in (getattr(media, "artists", None) or []) - if StringUtils.clear_upper(artist) + if text_tools.normalize_upper(artist) ) if music_type == MUSIC_ENTITY_ALBUM: - album_artist = StringUtils.clear_upper( + album_artist = text_tools.normalize_upper( getattr(media, "album_artist", None) or (artists[0] if artists else "") ) - album = StringUtils.clear_upper( + album = text_tools.normalize_upper( getattr(media, "album", None) or getattr(media, "title", None) or "" ) return "music", "local", music_type, album_artist, album, getattr(media, "year", None) @@ -188,8 +189,8 @@ class JobManager: "local", music_type, artists, - StringUtils.clear_upper(getattr(media, "title", None) or ""), - StringUtils.clear_upper(getattr(media, "album", None) or ""), + text_tools.normalize_upper(getattr(media, "title", None) or ""), + text_tools.normalize_upper(getattr(media, "album", None) or ""), getattr(media, "disc_number", None), getattr(media, "track_number", None), ) @@ -1420,7 +1421,7 @@ class TransferChain(ChainBase, ConfigReloadMixin, metaclass=Singleton): task.mediainfo, task.meta.begin_season ) if season_episodes: - se_str = f"{task.meta.season} {StringUtils.format_ep(season_episodes)}" + se_str = f"{task.meta.season} {episode_rules.format_ranges(season_episodes)}" else: se_str = f"{task.meta.season}" # 发送入库成功消息 @@ -3242,7 +3243,7 @@ class TransferChain(ChainBase, ConfigReloadMixin, metaclass=Singleton): return False if source_meta.type != target_meta.type: return False - if StringUtils.clear_upper(source_meta.name) != StringUtils.clear_upper( + if text_tools.normalize_upper(source_meta.name) != text_tools.normalize_upper( target_meta.name ): return False diff --git a/app/domain/context.py b/app/domain/context.py index f7b8e2334..ebf51474e 100644 --- a/app/domain/context.py +++ b/app/domain/context.py @@ -21,7 +21,7 @@ from app.schemas.types import ( MediaType, ) from app.schemas.media import normalize_media_source, resolve_media_identity -from app.domain.string import StringUtils +from app.foundation import temporal as time_tools BANGUMI_MOVIE_PLATFORMS = frozenset({"movie", "电影", "剧场版"}) ANILIST_MOVIE_FORMATS = frozenset({"MOVIE"}) @@ -779,7 +779,7 @@ class TorrentInfo: """ if not self.freedate: return "" - return StringUtils.diff_time_str(self.freedate) + return time_tools.format_remaining(self.freedate) def pub_minutes(self) -> float: """ diff --git a/app/domain/episode.py b/app/domain/episode.py new file mode 100644 index 000000000..616319cb5 --- /dev/null +++ b/app/domain/episode.py @@ -0,0 +1,39 @@ +"""剧集编号列表的业务显示规则。""" + +from typing import List + + +def compact_numbers(numbers: List[int]) -> str: + """把连续剧集编号压缩为逗号分隔的数字区间。""" + numbers.sort() + result = [] + start = numbers[0] + end = numbers[0] + for number in numbers[1:]: + if number == end + 1: + end = number + continue + result.append(str(start) if start == end else f"{start}-{end}") + start = end = number + result.append(str(start) if start == end else f"{start}-{end}") + return ",".join(result) + + +def format_ranges(numbers: List[int]) -> str: + """把剧集编号格式化为带 E 前缀和中文顿号的连续区间。""" + if not numbers: + return "" + if len(numbers) == 1: + return f"E{numbers[0]:02d}" + numbers.sort() + ranges = [] + start = numbers[0] + end = numbers[0] + for number in numbers[1:]: + if number == end + 1: + end = number + continue + ranges.append(f"E{start:02d}" if start == end else f"E{start:02d}-E{end:02d}") + start = end = number + ranges.append(f"E{start:02d}" if start == end else f"E{start:02d}-E{end:02d}") + return "、".join(ranges) diff --git a/app/domain/meta/infopath.py b/app/domain/meta/infopath.py index d204684f8..571d8db10 100644 --- a/app/domain/meta/infopath.py +++ b/app/domain/meta/infopath.py @@ -1,7 +1,7 @@ import regex as re from app.domain.meta.metabase import MetaBase -from app.domain.string import StringUtils +from app.foundation import text as text_tools AUXILIARY_CN_STEM_FULLMATCH_RE = re.compile( r"^(双语|字幕|特效|内封|外挂|官译|简体|繁体|繁中|简中|中英|简英|多语|" @@ -27,7 +27,7 @@ def should_use_parent_title_for_file_stem( return False if not PARENT_LATIN_TITLE_RE.search(parent_dir_name): return False - if not StringUtils.is_all_chinese(stem): + if not text_tools.is_all_chinese(stem): return False if len(stem) > 16: return False diff --git a/app/domain/meta/metaanime.py b/app/domain/meta/metaanime.py index ff806860f..db2baef83 100644 --- a/app/domain/meta/metaanime.py +++ b/app/domain/meta/metaanime.py @@ -6,7 +6,8 @@ import anitopy from app.domain.meta.customization import CustomizationMatcher from app.domain.meta.metabase import MetaBase from app.domain.meta.releasegroup import ReleaseGroupsMatcher -from app.domain.string import StringUtils +from app.domain import title as title_rules +from app.foundation import text as text_tools from app.foundation.text import convert as zhconv_convert from app.schemas.types import MediaType @@ -64,11 +65,11 @@ class MetaAnime(MetaBase): if anitopy_info: # 名称 name = anitopy_info.get("anime_title") - if not name or name in self._anime_no_words or (len(name) < 5 and not StringUtils.is_chinese(name)): + if not name or name in self._anime_no_words or (len(name) < 5 and not text_tools.contains_chinese(name)): anitopy_info = anitopy.parse("[ANIME]" + title) if anitopy_info: name = anitopy_info.get("anime_title") - if not name or name in self._anime_no_words or (len(name) < 5 and not StringUtils.is_chinese(name)): + if not name or name in self._anime_no_words or (len(name) < 5 and not text_tools.contains_chinese(name)): name_match = BRACKET_TITLE_RE.search(title) if name_match and name_match.group(1): name = name_match.group(1).strip() @@ -78,12 +79,12 @@ class MetaAnime(MetaBase): # 按/拆分中英文 if name.find("/") != -1: names = name.split("/") - if StringUtils.is_chinese(names[0]): + if text_tools.contains_chinese(names[0]): self.cn_name = names[0] if len(names) > 1: self.en_name = names[1] _split_flag = False - elif StringUtils.is_chinese(names[-1]): + elif text_tools.contains_chinese(names[-1]): self.cn_name = names[-1] if len(names) > 1: self.en_name = names[0] @@ -103,19 +104,19 @@ class MetaAnime(MetaBase): self.cn_name = "%s %s" % (self.cn_name or "", word) elif lastword_type == "en": self.en_name = "%s %s" % (self.en_name or "", word) - elif StringUtils.is_chinese(word): + elif text_tools.contains_chinese(word): self.cn_name = "%s %s" % (self.cn_name or "", word) lastword_type = "cn" else: self.en_name = "%s %s" % (self.en_name or "", word) lastword_type = "en" if self.cn_name: - _, self.cn_name, _, _, _, _ = StringUtils.get_keyword(self.cn_name) + _, self.cn_name, _, _, _, _ = title_rules.parse_search_keyword(self.cn_name) if self.cn_name: self.cn_name = self._name_nostring_pattern.sub('', self.cn_name).strip() if self.en_name: self.en_name = self._name_nostring_pattern.sub('', self.en_name).strip().title() - self._name = StringUtils.str_title(self.en_name) + self._name = text_tools.title_case(self.en_name) # 年份 year = anitopy_info.get("anime_year") if str(year).isdigit(): @@ -271,7 +272,7 @@ class MetaAnime(MetaBase): else: titles.append("%s%s" % (left_char, name.split("/")[0].strip())) elif name: - if StringUtils.is_chinese(name) and not StringUtils.is_all_chinese(name): + if text_tools.contains_chinese(name) and not text_tools.is_all_chinese(name): if not NUMERIC_BRACKET_RE.search(name): name = MIXED_CHINESE_TOKEN_RE.sub('', name).strip() if not name or name.strip().isdigit(): diff --git a/app/domain/meta/metabase.py b/app/domain/meta/metabase.py index 663766808..a3417dea7 100644 --- a/app/domain/meta/metabase.py +++ b/app/domain/meta/metabase.py @@ -8,7 +8,7 @@ import regex as re from app.schemas.types import MediaSource, MediaType from app.schemas.media import resolve_media_identity -from app.domain.string import StringUtils +from app.foundation import text as text_tools logger = logging.getLogger(__name__) @@ -128,7 +128,7 @@ class MetaBase(object): """ 返回名称 """ - if self.cn_name and StringUtils.is_all_chinese(self.cn_name): + if self.cn_name and text_tools.is_all_chinese(self.cn_name): return self.cn_name elif self.en_name: return self.en_name @@ -141,7 +141,7 @@ class MetaBase(object): """ 设置名称 """ - if StringUtils.is_all_chinese(name): + if text_tools.is_all_chinese(name): self.cn_name = name else: self.en_name = name diff --git a/app/domain/meta/metavideo.py b/app/domain/meta/metavideo.py index 2167b0e40..ee4c020b1 100644 --- a/app/domain/meta/metavideo.py +++ b/app/domain/meta/metavideo.py @@ -7,7 +7,7 @@ from app.domain.meta.customization import CustomizationMatcher from app.domain.meta.metabase import MetaBase from app.domain.meta.releasegroup import ReleaseGroupsMatcher from app.schemas.types import MediaType -from app.domain.string import StringUtils +from app.foundation import text as text_tools from app.domain.tokens import Tokens from app.domain.meta.streamingplatform import StreamingPlatforms from app.domain.meta.runtime import get_media_extensions @@ -218,7 +218,7 @@ class MetaVideo(MetaBase): self.init_subtitle(self.subtitle) # 去掉名字中不需要的干扰字符,过短的纯数字不要 self.cn_name = self.__fix_name(self.cn_name) - self.en_name = StringUtils.str_title(self.__fix_name(self.en_name)) + self.en_name = text_tools.title_case(self.__fix_name(self.en_name)) # 处理part if self.part and self.part.upper() == "PART": self.part = None @@ -245,7 +245,7 @@ class MetaVideo(MetaBase): if not description: return None titles = DESCRIPTION_SPLIT_RE.split(description) - if StringUtils.is_chinese(titles[0]): + if text_tools.contains_chinese(titles[0]): return titles[0] return None @@ -308,7 +308,7 @@ class MetaVideo(MetaBase): if token in self._name_se_words: self._last_token_type = 'name_se_words' return - if StringUtils.is_chinese(token): + if text_tools.contains_chinese(token): # 含有中文,直接做为标题(连着的数字或者英文会保留),且不再取用后面出现的中文 self._last_token_type = "cnname" if not self.cn_name: diff --git a/app/domain/site.py b/app/domain/site.py index e97ba372b..70d0169f2 100644 --- a/app/domain/site.py +++ b/app/domain/site.py @@ -1,6 +1,42 @@ from lxml import etree -from app.domain.string import StringUtils +from app.foundation.dom import DomUtils +from app.foundation.url import split_netloc + + +_SPECIAL_SITE_DOMAINS = ( + "u2.dmhy.org", + "pt.ecust.pp.ua", + "pt.gtkpw.xyz", + "pt.gtk.pw", +) + + +def urls_match(first: str, second: str) -> bool: + """判断两个地址是否指向忽略 www 前缀后的同一站点。""" + if not first or not second: + return False + if first.startswith("http"): + _scheme, first = split_netloc(first) + if second.startswith("http"): + _scheme, second = split_netloc(second) + return first.replace("www.", "") == second.replace("www.", "") + + +def extract_domain(url: str) -> str: + """按 MoviePilot 站点规则提取用于匹配的注册域名。""" + if not url: + return "" + for domain in _SPECIAL_SITE_DOMAINS: + if domain in url: + return domain + _scheme, netloc = split_netloc(url) + if not netloc: + return "" + labels = netloc.split(".") + if len(labels) > 3: + return netloc + return ".".join(labels[-2:]) class SiteUtils: @@ -16,7 +52,7 @@ class SiteUtils: """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return False # 存在明显的密码输入框,说明未登录 if html.xpath("//input[@type='password']"): @@ -49,7 +85,7 @@ class SiteUtils: """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return False # 站点签到支持的识别XPATH xpaths = [ diff --git a/app/domain/string.py b/app/domain/string.py deleted file mode 100644 index 86273a924..000000000 --- a/app/domain/string.py +++ /dev/null @@ -1,903 +0,0 @@ -import bisect -import datetime -import hashlib -import random -import re -from typing import Union, Tuple, Optional, Any, List, Generator -from urllib import parse - -import cn2an -import dateparser -import dateutil.parser - -from app.foundation.version import compare_version as compare_versions -from app.schemas.types import MediaType - -_special_domains = [ - 'u2.dmhy.org', - 'pt.ecust.pp.ua', - 'pt.gtkpw.xyz', - 'pt.gtk.pw' -] - -_max_media_title_words = 10 -_min_media_title_length = 2 -_non_media_title_pattern = re.compile(r"^#|^请[问帮你]|[??]$|^继续$") -_chat_intent_pattern = re.compile(r"帮我|请问|怎么|如何|为什么|可以|能否|推荐|介绍|谢谢|想看|找一下|搜一下") -_media_feature_pattern = re.compile( - r"第\s*[0-9一二三四五六七八九十百零]+\s*[季集]|S\d{1,2}(?:E\d{1,4})?|E\d{1,4}|(?:19|20)\d{2}", - re.IGNORECASE -) -_media_separator_pattern = re.compile(r"[\s\-_.::·'\"()\[\]【】]+") -_media_sentence_punctuation_pattern = re.compile(r"[,。!?!?,;;]") -_media_title_char_pattern = re.compile(r"[\u4e00-\u9fffA-Za-z]") - - -class StringUtils: - """提供媒体命名场景需要的字符串解析和格式化能力。""" - - - @staticmethod - def num_filesize(text: Union[str, int, float]) -> int: - """ - 将文件大小文本转化为字节 - """ - if not text: - return 0 - if not isinstance(text, str): - text = str(text) - if text.isdigit(): - return int(text) - text = text.replace(",", "").replace(" ", "").upper() - size = re.sub(r"[KMGTPI]*B?", "", text, flags=re.IGNORECASE) - try: - size = float(size) - except ValueError: - return 0 - if text.find("PB") != -1 or text.find("PIB") != -1: - size *= 1024 ** 5 - elif text.find("TB") != -1 or text.find("TIB") != -1: - size *= 1024 ** 4 - elif text.find("GB") != -1 or text.find("GIB") != -1: - size *= 1024 ** 3 - elif text.find("MB") != -1 or text.find("MIB") != -1: - size *= 1024 ** 2 - elif text.find("KB") != -1 or text.find("KIB") != -1: - size *= 1024 - return round(size) - - @staticmethod - def str_timelong(time_sec: Union[str, int, float]) -> str: - """ - 将数字转换为时间描述 - """ - if not isinstance(time_sec, int) or not isinstance(time_sec, float): - try: - time_sec = float(time_sec) - except ValueError: - return "" - d = [(0, '秒'), (60 - 1, '分'), (3600 - 1, '小时'), (86400 - 1, '天')] - s = [x[0] for x in d] - index = bisect.bisect_left(s, time_sec) - 1 - if index == -1: - return str(time_sec) - else: - b, u = d[index] - return str(round(time_sec / (b + 1))) + u - - @staticmethod - def str_secends(time_sec: Union[str, int, float]) -> str: - """ - 将秒转为时分秒字符串 - """ - hours = time_sec // 3600 - remainder_seconds = time_sec % 3600 - minutes = remainder_seconds // 60 - seconds = remainder_seconds % 60 - - time: str = str(int(seconds)) + '秒' - if minutes: - time = str(int(minutes)) + '分' + time - if hours: - time = str(int(hours)) + '时' + time - return time - - @staticmethod - def is_chinese(word: Union[str, list]) -> bool: - """ - 判断是否含有中文 - """ - if not word: - return False - if isinstance(word, list): - word = " ".join(word) - chn = re.compile(r'[\u4e00-\u9fff]') - if chn.search(word): - return True - else: - return False - - @staticmethod - def is_japanese(word: str) -> bool: - """ - 判断是否含有日文 - """ - jap = re.compile(r'[\u3040-\u309F\u30A0-\u30FF]') - if jap.search(word): - return True - else: - return False - - @staticmethod - def is_korean(word: str) -> bool: - """ - 判断是否包含韩文 - """ - kor = re.compile(r'[\uAC00-\uD7FF]') - if kor.search(word): - return True - else: - return False - - @staticmethod - def is_all_chinese(word: str) -> bool: - """ - 判断是否全是中文 - """ - for ch in word: - if ch == ' ': - continue - if '\u4e00' <= ch <= '\u9fff': - continue - else: - return False - return True - - @staticmethod - def is_english_word(word: str) -> bool: - """ - 判断是否为英文单词,有空格时返回False - """ - return word.encode().isalpha() - - @staticmethod - def str_int(text: str) -> int: - """ - web字符串转int - :param text: - :return: - """ - if text: - text = text.strip() - if not text: - return 0 - try: - return int(text.replace(',', '')) - except ValueError: - return 0 - - @staticmethod - def str_float(text: str) -> float: - """ - web字符串转float - :param text: - :return: - """ - if text: - text = text.strip() - if not text: - return 0.0 - try: - text = text.replace(',', '') - if text: - return float(text) - except ValueError: - pass - return 0.0 - - @staticmethod - def clear(text: Union[list, str], replace_word: str = "", - allow_space: bool = False) -> Union[list, str]: - """ - 忽略特殊字符 - """ - # 需要忽略的特殊字符 - CONVERT_EMPTY_CHARS = r"[、.。,,·::;;!!??'’\"“”()()\[\]【】「」\-—―\+\|\\_/&#~~]" - if not text: - return text - if not isinstance(text, list): - text = re.sub(r"[\u200B-\u200D\uFEFF]", - "", - re.sub(r"%s" % CONVERT_EMPTY_CHARS, replace_word, text), - flags=re.IGNORECASE) - if not allow_space: - return re.sub(r"\s+", "", text) - else: - return re.sub(r"\s+", " ", text).strip() - else: - return [StringUtils.clear(x) for x in text] - - @staticmethod - def clear_upper(text: Optional[str]) -> str: - """ - 去除特殊字符,同时大写 - """ - if not text: - return "" - return StringUtils.clear(text).upper().strip() - - @staticmethod - def str_filesize(size: Union[str, float, int], pre: int = 2) -> str: - """ - 将字节计算为文件大小描述(带单位的格式化后返回) - """ - if size is None: - return "" - size = re.sub(r"\s|B|iB", "", str(size), re.I) - if size.replace(".", "").isdigit(): - try: - size = float(size) - d = [(1024 - 1, 'K'), (1024 ** 2 - 1, 'M'), (1024 ** 3 - 1, 'G'), (1024 ** 4 - 1, 'T')] - s = [x[0] for x in d] - index = bisect.bisect_left(s, size) - 1 # noqa - if index == -1: - return str(size) + "B" - else: - b, u = d[index] - return str(round(size / (b + 1), pre)) + u - except ValueError: - return "" - if re.findall(r"[KMGTP]", size, re.I): - return size - else: - return size + "B" - - @staticmethod - def format_size(size_bytes: int) -> str: - """ - 将字节转换为人类可读格式 - """ - if not size_bytes or size_bytes == 0: - return "0 B" - - units = ["B", "KB", "MB", "GB", "TB", "PB"] - size = float(size_bytes) - unit_index = 0 - - while size >= 1024 and unit_index < len(units) - 1: - size /= 1024 - unit_index += 1 - - # 保留两位小数 - if unit_index == 0: - return f"{int(size)} {units[unit_index]}" - return f"{size:.2f} {units[unit_index]}" - - @staticmethod - def url_equal(url1: str, url2: str) -> bool: - """ - 比较两个地址是否为同一个网站 - """ - if not url1 or not url2: - return False - if url1.startswith("http"): - url1 = parse.urlparse(url1).netloc - if url2.startswith("http"): - url2 = parse.urlparse(url2).netloc - if url1.replace("www.", "") == url2.replace("www.", ""): - return True - return False - - @staticmethod - def get_url_netloc(url: str) -> Tuple[str, str]: - """ - 获取URL的协议和域名部分 - """ - if not url: - return "", "" - if not url.startswith("http"): - return "http", url - addr = parse.urlparse(url) - return addr.scheme, addr.netloc - - @staticmethod - def get_url_domain(url: str) -> str: - """ - 获取URL的域名部分,只保留最后两级 - """ - if not url: - return "" - for domain in _special_domains: - if domain in url: - return domain - _, netloc = StringUtils.get_url_netloc(url) - if netloc: - locs = netloc.split(".") - if len(locs) > 3: - return netloc - return ".".join(locs[-2:]) - return "" - - @staticmethod - def get_url_sld(url: str) -> str: - """ - 获取URL的二级域名部分,不含端口,若为IP则返回IP - """ - if not url: - return "" - _, netloc = StringUtils.get_url_netloc(url) - if not netloc: - return "" - netloc = netloc.split(":")[0].split(".") - if len(netloc) >= 2: - return netloc[-2] - return netloc[0] - - @staticmethod - def get_url_host(url: str) -> str: - """ - 获取URL的一级域名 - """ - if not url: - return "" - _, netloc = StringUtils.get_url_netloc(url) - if not netloc: - return "" - return netloc.split(".")[-2] - - @staticmethod - def get_base_url(url: str) -> str: - """ - 获取URL根地址 - """ - if not url: - return "" - scheme, netloc = StringUtils.get_url_netloc(url) - return f"{scheme}://{netloc}" - - @staticmethod - def clear_file_name(name: str) -> Optional[str]: - """移除文件名中不允许使用的字符。""" - if not name: - return None - return re.sub(r"[*?\\/\"<>~|]", "", name, flags=re.IGNORECASE).replace(":", ":") - - @staticmethod - def generate_random_str(randomlength: int = 16) -> str: - """ - 生成一个指定长度的随机字符串 - """ - random_str = '' - base_str = 'ABCDEFGHIGKLMNOPQRSTUVWXYZabcdefghigklmnopqrstuvwxyz0123456789' - length = len(base_str) - 1 - for i in range(randomlength): - random_str += base_str[random.randint(0, length)] - return random_str - - @staticmethod - def get_time(date: Any) -> Optional[datetime.datetime]: - """将常见日期表达解析为 datetime,无法解析时返回 None。""" - try: - return dateutil.parser.parse(date) - except dateutil.parser.ParserError: - return None - - @staticmethod - def unify_datetime_str(datetime_str: str) -> str: - """ - 日期时间格式化 统一转成 2020-10-14 07:48:04 这种格式 - # 场景1: 带有时区的日期字符串 eg: Sat, 15 Oct 2022 14:02:54 +0800 - # 场景2: 中间带T的日期字符串 eg: 2020-10-14T07:48:04 - # 场景3: 中间带T的日期字符串 eg: 2020-10-14T07:48:04.208 - # 场景4: 日期字符串以GMT结尾 eg: Fri, 14 Oct 2022 07:48:04 GMT - # 场景5: 日期字符串以UTC结尾 eg: Fri, 14 Oct 2022 07:48:04 UTC - # 场景6: 日期字符串以Z结尾 eg: Fri, 14 Oct 2022 07:48:04Z - # 场景7: 日期字符串为相对时间 eg: 1 month, 2 days ago - :param datetime_str: - :return: - """ - # 传入的参数如果是None 或者空字符串 直接返回 - if not datetime_str: - return datetime_str - - try: - return dateparser.parse(datetime_str).strftime('%Y-%m-%d %H:%M:%S') - except Exception as e: - print(str(e)) - return datetime_str - - @staticmethod - def format_timestamp(timestamp: str, date_format: str = '%Y-%m-%d %H:%M:%S') -> str: - """ - 时间戳转日期 - :param timestamp: - :param date_format: - :return: - """ - if isinstance(timestamp, str) and not timestamp.isdigit(): - return timestamp - try: - return datetime.datetime.fromtimestamp(int(timestamp)).strftime(date_format) - except Exception as e: - print(str(e)) - return timestamp - - @staticmethod - def str_to_timestamp(date_str: str) -> float: - """ - 日期转时间戳 - :param date_str: - :return: - """ - if not date_str: - return 0 - try: - return dateparser.parse(date_str).timestamp() - except Exception as e: - print(str(e)) - return 0 - - @staticmethod - def to_bool(text: str, default_val: bool = False) -> bool: - """ - 字符串转bool - :param text: 要转换的值 - :param default_val: 默认值 - :return: - """ - if isinstance(text, str) and not text: - return default_val - if isinstance(text, bool): - return text - if isinstance(text, int) or isinstance(text, float): - return True if text > 0 else False - if isinstance(text, str) and text.lower() in ['y', 'true', '1', 'yes', 'on']: - return True - return False - - @staticmethod - def str_from_cookiejar(cj: dict) -> str: - """ - 将cookiejar转换为字符串 - :param cj: - :return: - """ - return '; '.join(['='.join(item) for item in cj.items()]) - - @staticmethod - def get_idlist(content: str, dicts: List[dict]): - """ - 从字符串中提取id列表 - :param content: 字符串 - :param dicts: 字典 - :return: - """ - if not content: - return [] - id_list = [] - content_list = content.split() - for dic in dicts: - if dic.get('name') in content_list and dic.get('id') not in id_list: - id_list.append(dic.get('id')) - content = content.replace(dic.get('name'), '') - return id_list, re.sub(r'\s+', ' ', content).strip() - - @staticmethod - def md5_hash(data: Any) -> str: - """ - MD5 HASH - """ - if not data: - return "" - return hashlib.md5(str(data).encode()).hexdigest() - - @staticmethod - def str_timehours(minutes: int) -> str: - """ - 将分钟转换成小时和分钟 - :param minutes: - :return: - """ - if not minutes: - return "" - hours = minutes // 60 - minutes = minutes % 60 - if hours: - return "%s小时%s分" % (hours, minutes) - else: - return "%s分钟" % minutes - - @staticmethod - def str_amount(amount: object, curr="$") -> str: - """ - 格式化显示金额 - """ - if not amount: - return "0" - return curr + format(amount, ",") - - @staticmethod - def count_words(text: str) -> int: - """ - 计算字符串中包含的单词或汉字的数量,需要兼容中英文混合的情况 - :param text: 要计算的字符串 - :return: 字符串中包含的词数量 - """ - if not text: - return 0 - # 使用正则表达式匹配汉字和英文单词 - chinese_pattern = '[\u4e00-\u9fa5]' - english_pattern = '[a-zA-Z]+' - - # 匹配汉字和英文单词 - chinese_matches = re.findall(chinese_pattern, text) - english_matches = re.findall(english_pattern, text) - - # 过滤掉空格和数字 - chinese_words = [word for word in chinese_matches if word.isalpha()] - english_words = [word for word in english_matches if word.isalpha()] - - # 计算汉字和英文单词的数量 - chinese_count = len(chinese_words) - english_count = len(english_words) - - return chinese_count + english_count - - @staticmethod - def is_media_title_like(text: str) -> bool: - """ - 判断文本是否像影视剧名称 - """ - if not text: - return False - text = re.sub(r'\s+', ' ', text).strip() - if not text: - return False - if _non_media_title_pattern.search(text) \ - or StringUtils.count_words(text) > _max_media_title_words: - return False - if "://" in text or text.startswith("magnet:?"): - return False - if _chat_intent_pattern.search(text): - return False - if _media_sentence_punctuation_pattern.search(text): - return False - - # 先移除季/集/年份等媒体特征,再移除分隔符,只保留核心名称用于最终判定 - candidate = _media_feature_pattern.sub("", text) - candidate = _media_separator_pattern.sub("", candidate) - return len(candidate) >= _min_media_title_length and _media_title_char_pattern.search(candidate) is not None - - @staticmethod - def split_text(text: str, max_length: int) -> Generator: - """ - 把文本拆分为固定字节长度的数组,优先按换行拆分,避免单词内拆分 - """ - if not text: - yield '' - # 分行 - lines = re.split('\n', text) - buf = '' - for line in lines: - if len(line.encode('utf-8')) > max_length: - # 超长行继续拆分 - blank = "" - if re.match(r'^[A-Za-z0-9.\s]+', line): - # 英文行按空格拆分 - parts = line.split() - blank = " " - else: - # 中文行按字符拆分 - parts = line - part = '' - for p in parts: - if len((part + p).encode('utf-8')) > max_length: - # 超长则Yield - yield (buf + part).strip() - buf = '' - part = f"{blank}{p}" - else: - part = f"{part}{blank}{p}" - if part: - # 将最后的部分追加到buf - buf += part - else: - if len((buf + "\n" + line).encode('utf-8')) > max_length: - # buf超长则Yield - yield buf.strip() - buf = line - else: - # 短行直接追加到buf - if buf: - buf = f"{buf}\n{line}" - else: - buf = line - if buf: - # 处理文本末尾剩余部分 - yield buf.strip() - - @staticmethod - def get_keyword(content: str) \ - -> Tuple[Optional[MediaType], Optional[str], Optional[int], Optional[int], Optional[str], Optional[str]]: - """ - 从搜索关键字中拆分中年份、季、集、类型 - """ - if not content: - return None, None, None, None, None, None - - # 去掉查询中的电影或电视剧关键字 - mtype = MediaType.TV if re.search(r'^(电视剧|动漫|\s+电视剧|\s+动漫)', content) else None - content = re.sub(r'^(电影|电视剧|动漫|\s+电影|\s+电视剧|\s+动漫)', '', content).strip() - - # 稍微切一下剧集吧 - season_num = None - episode_num = None - season_re = re.search(r'第\s*([0-9一二三四五六七八九十]+)\s*季', content, re.IGNORECASE) - if season_re: - mtype = MediaType.TV - season_num = int(cn2an.cn2an(season_re.group(1), mode='smart')) - - episode_re = re.search(r'第\s*([0-9一二三四五六七八九十百零]+)\s*集', content, re.IGNORECASE) - if episode_re: - mtype = MediaType.TV - episode_num = int(cn2an.cn2an(episode_re.group(1), mode='smart')) - if episode_num and not season_num: - season_num = 1 - - year_re = re.search(r'[\s(]+(\d{4})[\s)]*', content) - year = year_re.group(1) if year_re else None - - key_word = re.sub( - r'第\s*[0-9一二三四五六七八九十]+\s*季|第\s*[0-9一二三四五六七八九十百零]+\s*集|[\s(]+(\d{4})[\s)]*', '', - content, flags=re.IGNORECASE).strip() - key_word = re.sub(r'\s+', ' ', key_word) if key_word else year - - return mtype, key_word, season_num, episode_num, year, content - - @staticmethod - def str_title(s: Optional[str]) -> str: - """ - 大写首字母兼容None - """ - return s.title() if s else s - - @staticmethod - def escape_markdown(content: str) -> str: - """ - Escapes Markdown characters in a string of Markdown. - - Credits to: simonsmh - - :param content: The string of Markdown to escape. - :type content: :obj:`str` - - :return: The escaped string. - :rtype: :obj:`str` - """ - - parses = re.sub(r"([_*\[\]()~`>#+\-=|.!{}])", r"\\\1", content) - reparse = re.sub(r"\\\\([_*\[\]()~`>#+\-=|.!{}])", r"\1", parses) - return reparse - - @staticmethod - def get_domain_address(address: str, prefix: bool = True) -> Tuple[Optional[str], Optional[int]]: - """ - 从地址中获取域名和端口号 - :param address: 地址 - :param prefix:返回域名是否要包含协议前缀 - """ - if not address: - return None, None - # 去掉末尾的/ - address = address.rstrip("/") - if prefix and not address.startswith("http"): - # 如果需要包含协议前缀,但地址不包含协议前缀,则添加 - address = "http://" + address - elif not prefix and address.startswith("http"): - # 如果不需要包含协议前缀,但地址包含协议前缀,则去掉 - address = address.split("://")[-1] - # 拆分域名和端口号 - parts = address.split(":") - if len(parts) > 3: - # 处理不希望包含多个冒号的情况(除了协议后的冒号) - return None, None - elif len(parts) == 3: - port = int(parts[-1]) - # 不含端口地址 - domain = ":".join(parts[:-1]).rstrip('/') - elif len(parts) == 2: - port = 443 if address.startswith("https") else 80 - domain = address - else: - return None, None - return domain, port - - @staticmethod - def str_series(array: List[int]) -> str: - """ - 将季集列表转化为字符串简写 - """ - - # 确保数组按照升序排列 - array.sort() - - result = [] - start = array[0] - end = array[0] - - for i in range(1, len(array)): - if array[i] == end + 1: - end = array[i] - else: - if start == end: - result.append(str(start)) - else: - result.append(f"{start}-{end}") - start = array[i] - end = array[i] - - # 处理最后一个序列 - if start == end: - result.append(str(start)) - else: - result.append(f"{start}-{end}") - - return ",".join(result) - - @staticmethod - def format_ep(nums: List[int]) -> str: - """ - 将剧集列表格式化为连续区间 - """ - if not nums: - return "" - if len(nums) == 1: - return f"E{nums[0]:02d}" - # 将数组升序排序 - nums.sort() - formatted_ranges = [] - start = nums[0] - end = nums[0] - - for i in range(1, len(nums)): - if nums[i] == end + 1: - end = nums[i] - else: - if start == end: - formatted_ranges.append(f"E{start:02d}") - else: - formatted_ranges.append(f"E{start:02d}-E{end:02d}") - start = end = nums[i] - - if start == end: - formatted_ranges.append(f"E{start:02d}") - else: - formatted_ranges.append(f"E{start:02d}-E{end:02d}") - - formatted_string = "、".join(formatted_ranges) - return formatted_string - - @staticmethod - def is_number(text: str) -> bool: - """ - 判断字符是否为可以转换为整数或者浮点数 - """ - if not text: - return False - try: - float(text) - return True - except ValueError: - return False - - @staticmethod - def find_common_prefix(str1: str, str2: str) -> str: - """返回两个字符串从首字符开始的公共前缀。""" - if not str1 or not str2: - return '' - common_prefix = [] - min_len = min(len(str1), len(str2)) - - for i in range(min_len): - if str1[i] == str2[i]: - common_prefix.append(str1[i]) - else: - break - - return ''.join(common_prefix) - - @staticmethod - def compare_version(v1: str, compare_type: str, v2: str, verbose: bool = False) \ - -> Tuple[Optional[bool], str | Exception] | Optional[bool]: - """兼容旧 StringUtils API,并转交基础版本比较能力。""" - return compare_versions(v1, compare_type, v2, verbose) - - @staticmethod - def diff_time_str(time_str: str): - """ - 输入YYYY-MM-DD HH24:MI:SS 格式的时间字符串,返回距离现在的剩余时间:xx天xx小时xx分钟 - """ - if not time_str: - return '' - try: - time_obj = datetime.datetime.strptime(time_str, '%Y-%m-%d %H:%M:%S') - except ValueError: - return time_str - now = datetime.datetime.now() - diff = time_obj - now - diff_seconds = diff.seconds - diff_days = diff.days - diff_hours = diff_seconds // 3600 - diff_minutes = (diff_seconds % 3600) // 60 - if diff_days > 0: - return f'{diff_days}天{diff_hours}小时{diff_minutes}分钟' - elif diff_hours > 0: - return f'{diff_hours}小时{diff_minutes}分钟' - elif diff_minutes > 0: - return f'{diff_minutes}分钟' - else: - return '' - - @staticmethod - def safe_strip(value) -> Optional[str]: - """ - 去除字符串两端的空白字符 - :return: 如果输入值不是 None,返回去除空白字符后的字符串,否则返回 None - """ - return value.strip() if value is not None else None - - @staticmethod - def is_valid_html_element(elem) -> bool: - """ - 检查elem是否为有效的HTML元素。元素必须为非None并且具有非零长度。 - - :param elem: 要检查的HTML元素 - :return: 如果elem有效(非None且长度大于0),返回True;否则返回False - """ - return elem is not None and len(elem) > 0 - - @staticmethod - def is_link(text: str) -> bool: - """ - 检查文件是否为链接地址,支持各类协议 - :param text: 要检查的文本 - :return: 如果URL有效,返回True;否则返回False - """ - if not text: - return False - # 检查是否以http、https、ftp等协议开头 - if re.match(r'^(http|https|ftp|ftps|sftp|ws|wss)://', text): - return True - # 检查是否为IP地址或域名 - if re.match(r'^[a-zA-Z0-9.-]+(\.[a-zA-Z]{2,})?$', text): - return True - return False - - @staticmethod - def is_magnet_link(content: Union[str, bytes]) -> bool: - """ - 判断内容是否为磁力链接 - """ - if not content: - return False - if isinstance(content, str) and content.startswith("magnet:"): - return True - if isinstance(content, bytes) and content.startswith(b"magnet:"): - return True - return False - - @staticmethod - def natural_sort_key(text: str) -> List[Union[int, str]]: - """ - 自然排序 - 将字符串拆分为数字和非数字部分,数字部分转换为整数,非数字部分转换为小写字母 - :param text: 要处理的字符串 - :return 用于排序的数字和字符串列表 - """ - if text is None: - return [] - - if not isinstance(text, str): - text = str(text) - - return [int(part) if part.isdigit() else part.lower() for part in re.split(r'(\d+)', text)] diff --git a/app/domain/title.py b/app/domain/title.py new file mode 100644 index 000000000..9a3471533 --- /dev/null +++ b/app/domain/title.py @@ -0,0 +1,88 @@ +"""媒体标题候选判断和搜索关键字解析规则。""" + +import re +from typing import Optional, Tuple + +import cn2an + +from app.foundation.text import count_words +from app.schemas.types import MediaType + + +_MAX_TITLE_WORDS = 10 +_MIN_TITLE_LENGTH = 2 +_NON_TITLE_PATTERN = re.compile(r"^#|^请[问帮你]|[??]$|^继续$") +_CHAT_INTENT_PATTERN = re.compile(r"帮我|请问|怎么|如何|为什么|可以|能否|推荐|介绍|谢谢|想看|找一下|搜一下") +_MEDIA_FEATURE_PATTERN = re.compile( + r"第\s*[0-9一二三四五六七八九十百零]+\s*[季集]|S\d{1,2}(?:E\d{1,4})?|E\d{1,4}|(?:19|20)\d{2}", + re.IGNORECASE, +) +_MEDIA_SEPARATOR_PATTERN = re.compile(r"[\s\-_.::·'\"()\[\]【】]+") +_SENTENCE_PUNCTUATION_PATTERN = re.compile(r"[,。!?!?,;;]") +_TITLE_CHARACTER_PATTERN = re.compile(r"[\u4e00-\u9fffA-Za-z]") + + +def is_media_title_like(value: str) -> bool: + """判断短文本是否具备影视标题特征而不是对话或链接。""" + if not value: + return False + normalized = re.sub(r"\s+", " ", value).strip() + if not normalized: + return False + if _NON_TITLE_PATTERN.search(normalized) or count_words(normalized) > _MAX_TITLE_WORDS: + return False + if "://" in normalized or normalized.startswith("magnet:?"): + return False + if _CHAT_INTENT_PATTERN.search(normalized): + return False + if _SENTENCE_PUNCTUATION_PATTERN.search(normalized): + return False + + candidate = _MEDIA_FEATURE_PATTERN.sub("", normalized) + candidate = _MEDIA_SEPARATOR_PATTERN.sub("", candidate) + return ( + len(candidate) >= _MIN_TITLE_LENGTH + and _TITLE_CHARACTER_PATTERN.search(candidate) is not None + ) + + +def parse_search_keyword( + content: str, +) -> Tuple[Optional[MediaType], Optional[str], Optional[int], Optional[int], Optional[str], Optional[str]]: + """从搜索文本中提取媒体类型、标题、季、集和年份。""" + if not content: + return None, None, None, None, None, None + + media_type = MediaType.TV if re.search(r"^(电视剧|动漫|\s+电视剧|\s+动漫)", content) else None + content = re.sub(r"^(电影|电视剧|动漫|\s+电影|\s+电视剧|\s+动漫)", "", content).strip() + + season = None + episode = None + season_match = re.search(r"第\s*([0-9一二三四五六七八九十]+)\s*季", content, re.IGNORECASE) + if season_match: + media_type = MediaType.TV + season = int(cn2an.cn2an(season_match.group(1), mode="smart")) + + episode_match = re.search( + r"第\s*([0-9一二三四五六七八九十百零]+)\s*集", + content, + re.IGNORECASE, + ) + if episode_match: + media_type = MediaType.TV + episode = int(cn2an.cn2an(episode_match.group(1), mode="smart")) + if episode and not season: + season = 1 + + year_match = re.search(r"[\s(]+(\d{4})[\s)]*", content) + year = year_match.group(1) if year_match else None + keyword = re.sub( + r"第\s*[0-9一二三四五六七八九十]+\s*季|" + r"第\s*[0-9一二三四五六七八九十百零]+\s*集|" + r"[\s(]+(\d{4})[\s)]*", + "", + content, + flags=re.IGNORECASE, + ).strip() + keyword = re.sub(r"\s+", " ", keyword) if keyword else year + return media_type, keyword, season, episode, year, content diff --git a/app/domain/torrent.py b/app/domain/torrent.py new file mode 100644 index 000000000..5719bef07 --- /dev/null +++ b/app/domain/torrent.py @@ -0,0 +1,14 @@ +"""种子链接内容的纯领域判断规则。""" + +from typing import Union + + +def is_magnet_link(content: Union[str, bytes]) -> bool: + """判断字符串或字节内容是否为磁力链接。""" + if not content: + return False + if isinstance(content, str): + return content.startswith("magnet:") + if isinstance(content, bytes): + return content.startswith(b"magnet:") + return False diff --git a/app/foundation/dom.py b/app/foundation/dom.py index 8699dd4b3..b7119e407 100644 --- a/app/foundation/dom.py +++ b/app/foundation/dom.py @@ -2,7 +2,12 @@ from typing import Union class DomUtils: - """提供 XML DOM 节点读取和创建辅助能力。""" + """提供不含业务状态的 XML/HTML DOM 基础能力。""" + + @staticmethod + def has_child_elements(element) -> bool: + """判断 DOM 元素是否存在且至少包含一个子元素。""" + return element is not None and len(element) > 0 @staticmethod def tag_value(tag_item, tag_name: str, attname: str = "", default: Union[str, int] = None): diff --git a/app/foundation/size.py b/app/foundation/size.py new file mode 100644 index 000000000..2589d98fe --- /dev/null +++ b/app/foundation/size.py @@ -0,0 +1,79 @@ +"""字节容量的解析与显示基础能力。""" + +import bisect +import re +from typing import Union + + +def parse_size(text: Union[str, int, float]) -> int: + """将带二进制容量单位的文本转换为字节数。""" + if not text: + return 0 + if not isinstance(text, str): + text = str(text) + if text.isdigit(): + return int(text) + normalized = text.replace(",", "").replace(" ", "").upper() + size_text = re.sub(r"[KMGTPI]*B?", "", normalized, flags=re.IGNORECASE) + try: + size = float(size_text) + except ValueError: + return 0 + if "PB" in normalized or "PIB" in normalized: + size *= 1024 ** 5 + elif "TB" in normalized or "TIB" in normalized: + size *= 1024 ** 4 + elif "GB" in normalized or "GIB" in normalized: + size *= 1024 ** 3 + elif "MB" in normalized or "MIB" in normalized: + size *= 1024 ** 2 + elif "KB" in normalized or "KIB" in normalized: + size *= 1024 + return round(size) + + +def format_compact_size(size: Union[str, float, int], precision: int = 2) -> str: + """将字节数格式化为不带尾部 B 的紧凑容量描述。""" + if size is None: + return "" + # 历史实现把 re.IGNORECASE 作为 count 位置参数传入;这里保留其最多替换两次、 + # 且仅匹配大写单位的实际行为,避免旧插件在边缘输入上发生变化。 + normalized = re.sub(r"\s|B|iB", "", str(size), count=re.IGNORECASE) + if normalized.replace(".", "").isdigit(): + try: + numeric_size = float(normalized) + thresholds = [ + (1024 - 1, "K"), + (1024 ** 2 - 1, "M"), + (1024 ** 3 - 1, "G"), + (1024 ** 4 - 1, "T"), + ] + index = bisect.bisect_left( + [threshold for threshold, _unit in thresholds], numeric_size + ) - 1 + if index == -1: + return f"{numeric_size}B" + threshold, unit = thresholds[index] + return f"{round(numeric_size / (threshold + 1), precision)}{unit}" + except ValueError: + return "" + if re.findall(r"[KMGTP]", normalized, re.IGNORECASE): + return normalized + return f"{normalized}B" + + +def format_size(size_bytes: int) -> str: + """将字节数转换为带空格和完整单位的人类可读格式。""" + if not size_bytes: + return "0 B" + + units = ["B", "KB", "MB", "GB", "TB", "PB"] + size = float(size_bytes) + unit_index = 0 + while size >= 1024 and unit_index < len(units) - 1: + size /= 1024 + unit_index += 1 + + if unit_index == 0: + return f"{int(size)} {units[unit_index]}" + return f"{size:.2f} {units[unit_index]}" diff --git a/app/foundation/temporal.py b/app/foundation/temporal.py new file mode 100644 index 000000000..3bcc73fa7 --- /dev/null +++ b/app/foundation/temporal.py @@ -0,0 +1,111 @@ +"""日期、时间戳和时长的无状态转换能力。""" + +import bisect +import datetime +from typing import Any, Optional, Union + +import dateparser +import dateutil.parser + + +def format_approx_duration(seconds: Union[str, int, float]) -> str: + """把秒数格式化为单一最大单位的近似时长。""" + try: + seconds = float(seconds) + except (TypeError, ValueError): + return "" + thresholds = [(0, "秒"), (60 - 1, "分"), (3600 - 1, "小时"), (86400 - 1, "天")] + index = bisect.bisect_left( + [threshold for threshold, _unit in thresholds], seconds + ) - 1 + if index == -1: + return str(seconds) + threshold, unit = thresholds[index] + return f"{round(seconds / (threshold + 1))}{unit}" + + +def format_duration(seconds: Union[str, int, float]) -> str: + """把秒数格式化为时分秒组合文本。""" + hours = seconds // 3600 + remainder_seconds = seconds % 3600 + minutes = remainder_seconds // 60 + seconds = remainder_seconds % 60 + + result = f"{int(seconds)}秒" + if minutes: + result = f"{int(minutes)}分{result}" + if hours: + result = f"{int(hours)}时{result}" + return result + + +def parse_datetime(value: Any) -> Optional[datetime.datetime]: + """将常见日期表达解析为 datetime,无法解析时返回 None。""" + try: + return dateutil.parser.parse(value) + except (TypeError, ValueError, dateutil.parser.ParserError): + return None + + +def normalize_datetime(value: str) -> str: + """把常见绝对或相对日期文本统一为本地日期时间格式。""" + if not value: + return value + try: + parsed = dateparser.parse(value) + return parsed.strftime("%Y-%m-%d %H:%M:%S") if parsed else value + except (TypeError, ValueError, OverflowError): + return value + + +def format_timestamp(timestamp: str, date_format: str = "%Y-%m-%d %H:%M:%S") -> str: + """把 Unix 时间戳转换为指定格式的本地日期文本。""" + if isinstance(timestamp, str) and not timestamp.isdigit(): + return timestamp + try: + return datetime.datetime.fromtimestamp(int(timestamp)).strftime(date_format) + except (TypeError, ValueError, OverflowError, OSError): + return timestamp + + +def parse_timestamp(value: str) -> float: + """把日期表达转换为 Unix 时间戳,无法解析时返回零。""" + if not value: + return 0 + try: + parsed = dateparser.parse(value) + return parsed.timestamp() if parsed else 0 + except (TypeError, ValueError, OverflowError): + return 0 + + +def format_minutes(minutes: int) -> str: + """把分钟数格式化为小时和分钟组合文本。""" + if not minutes: + return "" + hours, remaining_minutes = divmod(minutes, 60) + if hours: + return f"{hours}小时{remaining_minutes}分" + return f"{remaining_minutes}分钟" + + +def format_remaining(value: str) -> str: + """把本地日期时间文本格式化为距当前时间的剩余时长。""" + if not value: + return "" + try: + target = datetime.datetime.strptime(value, "%Y-%m-%d %H:%M:%S") + except ValueError: + return value + difference = target - datetime.datetime.now() + seconds = difference.seconds + days = difference.days + hours = seconds // 3600 + minutes = (seconds % 3600) // 60 + if days > 0: + return f"{days}天{hours}小时{minutes}分钟" + if hours > 0: + return f"{hours}小时{minutes}分钟" + if minutes > 0: + return f"{minutes}分钟" + return "" diff --git a/app/foundation/text.py b/app/foundation/text.py index c6d5560e9..f438914ef 100644 --- a/app/foundation/text.py +++ b/app/foundation/text.py @@ -1,4 +1,8 @@ -"""无业务状态的中文分词与简繁转换工具。""" +"""无业务状态的文本识别、清理、转换和分段能力。""" + +import random +import re +from typing import Generator, List, Optional, Union from jieba_next import cut as jieba_next_cut from zhconv_rs import zhconv as _zhconv # pylint: disable=no-name-in-module @@ -14,3 +18,232 @@ def cut(text: str, HMM: bool = True, cut_all: bool = False) -> list[str]: def convert(text: str, target: str) -> str: """使用 zhconv-rs 执行中文简繁转换,并隔离第三方包的函数名差异。""" return _zhconv(text, target) + + +def contains_chinese(value: Union[str, list]) -> bool: + """判断文本或文本列表中是否包含中文字符。""" + if not value: + return False + if isinstance(value, list): + value = " ".join(value) + return re.search(r"[\u4e00-\u9fff]", value) is not None + + +def contains_japanese(value: str) -> bool: + """判断文本中是否包含平假名或片假名。""" + return re.search(r"[\u3040-\u309F\u30A0-\u30FF]", value) is not None + + +def contains_korean(value: str) -> bool: + """判断文本中是否包含韩文字符。""" + return re.search(r"[\uAC00-\uD7FF]", value) is not None + + +def is_all_chinese(value: str) -> bool: + """判断除空格外的全部字符是否都是中文。""" + return all(character == " " or "\u4e00" <= character <= "\u9fff" for character in value) + + +def is_english_word(value: str) -> bool: + """判断文本是否为不含空格的英文字母单词。""" + return value.encode().isalpha() + + +def parse_int(value: str) -> int: + """解析可能带千位分隔符的整数,无法解析时返回零。""" + if value: + value = value.strip() + if not value: + return 0 + try: + return int(value.replace(",", "")) + except ValueError: + return 0 + + +def parse_float(value: str) -> float: + """解析可能带千位分隔符的浮点数,无法解析时返回零。""" + if value: + value = value.strip() + if not value: + return 0.0 + try: + return float(value.replace(",", "")) + except ValueError: + return 0.0 + + +def remove_punctuation( + value: Union[list, str], + replacement: str = "", + allow_space: bool = False, +) -> Union[list, str]: + """移除历史匹配规则使用的标点和零宽字符。""" + punctuation = r"[、.。,,·::;;!!??'’\"“”()()\[\]【】「」\-—―\+\|\\_/&#~~]" + if not value: + return value + if isinstance(value, list): + return [remove_punctuation(item) for item in value] + normalized = re.sub( + r"[\u200B-\u200D\uFEFF]", + "", + re.sub(punctuation, replacement, value, flags=re.IGNORECASE), + flags=re.IGNORECASE, + ) + if not allow_space: + return re.sub(r"\s+", "", normalized) + return re.sub(r"\s+", " ", normalized).strip() + + +def normalize_upper(value: Optional[str]) -> str: + """移除历史匹配标点、空白并转换为大写。""" + if not value: + return "" + return remove_punctuation(value).upper().strip() + + +def sanitize_filename(value: str) -> Optional[str]: + """移除文件名中不允许使用的字符并替换英文冒号。""" + if not value: + return None + return re.sub(r"[*?\\/\"<>~|]", "", value, flags=re.IGNORECASE).replace(":", ":") + + +def random_string(length: int = 16) -> str: + """生成兼容历史字符集的指定长度随机字符串。""" + alphabet = "ABCDEFGHIGKLMNOPQRSTUVWXYZabcdefghigklmnopqrstuvwxyz0123456789" + last_index = len(alphabet) - 1 + return "".join(alphabet[random.randint(0, last_index)] for _index in range(length)) + + +def parse_bool(value, default: bool = False) -> bool: + """按历史配置规则把字符串或数值转换为布尔值。""" + if isinstance(value, str) and not value: + return default + if isinstance(value, bool): + return value + if isinstance(value, (int, float)): + return value > 0 + return isinstance(value, str) and value.lower() in {"y", "true", "1", "yes", "on"} + + +def cookiejar_to_string(cookiejar: dict) -> str: + """把键值形式的 CookieJar 序列化为 Cookie 请求头文本。""" + return "; ".join("=".join(item) for item in cookiejar.items()) + + +def extract_named_ids(content: str, entries: List[dict]): + """从空格分隔文本中提取命名条目 ID,并返回剩余内容。""" + if not content: + return [] + identifiers = [] + content_parts = content.split() + for entry in entries: + if entry.get("name") in content_parts and entry.get("id") not in identifiers: + identifiers.append(entry.get("id")) + content = content.replace(entry.get("name"), "") + return identifiers, re.sub(r"\s+", " ", content).strip() + + +def format_amount(amount: object, currency: str = "$") -> str: + """使用千位分隔符和货币前缀格式化金额。""" + if not amount: + return "0" + return currency + format(amount, ",") + + +def count_words(value: str) -> int: + """统计中英文混合文本中的汉字数和英文单词数。""" + if not value: + return 0 + chinese_words = [ + word for word in re.findall(r"[\u4e00-\u9fa5]", value) if word.isalpha() + ] + english_words = [word for word in re.findall(r"[a-zA-Z]+", value) if word.isalpha()] + return len(chinese_words) + len(english_words) + + +def split_by_bytes(value: str, max_length: int) -> Generator[str, None, None]: + """按 UTF-8 字节上限分段,优先保持换行和英文单词完整。""" + if not value: + yield "" + lines = re.split("\n", value) + buffer = "" + for line in lines: + if len(line.encode("utf-8")) > max_length: + separator = "" + if re.match(r"^[A-Za-z0-9.\s]+", line): + parts = line.split() + separator = " " + else: + parts = line + part = "" + for item in parts: + if len((part + item).encode("utf-8")) > max_length: + yield (buffer + part).strip() + buffer = "" + part = f"{separator}{item}" + else: + part = f"{part}{separator}{item}" + if part: + buffer += part + elif len((buffer + "\n" + line).encode("utf-8")) > max_length: + yield buffer.strip() + buffer = line + elif buffer: + buffer = f"{buffer}\n{line}" + else: + buffer = line + if buffer: + yield buffer.strip() + + +def title_case(value: Optional[str]) -> str: + """转换为标题大小写,并兼容空值。""" + return value.title() if value else value + + +def escape_markdown(value: str) -> str: + """转义 Markdown 保留字符,并保持历史二次转义语义。""" + escaped = re.sub(r"([_*\[\]()~`>#+\-=|.!{}])", r"\\\1", value) + return re.sub(r"\\\\([_*\[\]()~`>#+\-=|.!{}])", r"\1", escaped) + + +def is_number(value: str) -> bool: + """判断文本能否转换为整数或浮点数。""" + if not value: + return False + try: + float(value) + return True + except ValueError: + return False + + +def common_prefix(first: str, second: str) -> str: + """返回两个字符串从首字符开始的公共前缀。""" + if not first or not second: + return "" + prefix = [] + for first_character, second_character in zip(first, second): + if first_character != second_character: + break + prefix.append(first_character) + return "".join(prefix) + + +def strip_optional(value) -> Optional[str]: + """去除可空值两端空白,并保持 None。""" + return value.strip() if value is not None else None + + +def natural_sort_key(value: str) -> List[Union[int, str]]: + """把文本拆成数字和小写文本片段,供自然排序使用。""" + if value is None: + return [] + if not isinstance(value, str): + value = str(value) + return [ + int(part) if part.isdigit() else part.lower() + for part in re.split(r"(\d+)", value) + ] diff --git a/app/foundation/url.py b/app/foundation/url.py index 92064e1f8..5aa218b92 100644 --- a/app/foundation/url.py +++ b/app/foundation/url.py @@ -1,4 +1,5 @@ import mimetypes +import re from pathlib import Path from typing import Optional, Union, Tuple from urllib import parse @@ -70,6 +71,7 @@ class UrlUtils: except Exception: return None + @staticmethod def get_mime_type(path_or_url: Union[str, Path], default_type: str = "application/octet-stream") -> str: """ @@ -135,3 +137,77 @@ class UrlUtils: return protocol, hostname, port, path except Exception: return None + + +def split_netloc(url: str) -> Tuple[str, str]: + """返回 URL 的协议与网络位置,并兼容未带协议的历史输入。""" + if not url: + return "", "" + if not url.startswith("http"): + return "http", url + address = urlparse(url) + return address.scheme, address.netloc + + +def second_level_label(url: str) -> str: + """返回不含端口的倒数第二级域名标签,IP 则保持原值。""" + if not url: + return "" + _scheme, netloc = split_netloc(url) + if not netloc: + return "" + labels = netloc.split(":")[0].split(".") + return labels[-2] if len(labels) >= 2 else labels[0] + + +def host_label(url: str) -> str: + """返回兼容历史语义的一级主机标签。""" + if not url: + return "" + _scheme, netloc = split_netloc(url) + if not netloc: + return "" + return netloc.split(".")[-2] + + +def base_url(url: str) -> str: + """返回由协议和网络位置组成的根地址。""" + if not url: + return "" + scheme, netloc = split_netloc(url) + return f"{scheme}://{netloc}" + + +def parse_address( + address: str, + include_scheme: bool = True, +) -> Tuple[Optional[str], Optional[int]]: + """按历史规则从服务地址中提取域名文本和端口。""" + if not address: + return None, None + address = address.rstrip("/") + if include_scheme and not address.startswith("http"): + address = f"http://{address}" + elif not include_scheme and address.startswith("http"): + address = address.split("://")[-1] + parts = address.split(":") + if len(parts) > 3: + return None, None + if len(parts) == 3: + port = int(parts[-1]) + domain = ":".join(parts[:-1]).rstrip("/") + elif len(parts) == 2: + port = 443 if address.startswith("https") else 80 + domain = address + else: + return None, None + return domain, port + + +def is_link(value: str) -> bool: + """判断文本是否为受支持协议链接、IP 或域名形式。""" + if not value: + return False + if re.match(r"^(http|https|ftp|ftps|sftp|ws|wss)://", value): + return True + return re.match(r"^[a-zA-Z0-9.-]+(\.[a-zA-Z]{2,})?$", value) is not None diff --git a/app/helper/.resource-compat b/app/helper/.resource-compat deleted file mode 100644 index 3360fbab2..000000000 --- a/app/helper/.resource-compat +++ /dev/null @@ -1,2 +0,0 @@ -该目录仅用于兼容旧 Docker 镜像中固化的资源更新脚本,不承载 Python 源码。 -新更新器和新镜像仍只把站点资源安装到 app/application/site。 diff --git a/app/modules/discord/discord.py b/app/modules/discord/discord.py index d28d5cb87..f060d1c46 100644 --- a/app/modules/discord/discord.py +++ b/app/modules/discord/discord.py @@ -14,7 +14,7 @@ from app.domain.context import MediaInfo, Context from app.domain.metainfo import MetaInfo from app.runtime.log import logger from app.schemas.types import NotificationType -from app.domain.string import StringUtils +from app.foundation import size as size_tools # Discord embed 字段解析白名单 # 只有这些消息类型会使用复杂的字段解析逻辑 @@ -1032,7 +1032,7 @@ class Discord: title_text = f"{meta.season_episode} {meta.resource_term} {meta.video_term} {meta.release_group}" title_text = re.sub(r"\s+", " ", title_text).strip() detail = [ - f"{torrent.site_name} | {StringUtils.str_filesize(torrent.size)} | {torrent.volume_factor} | {torrent.seeders}↑", + f"{torrent.site_name} | {size_tools.format_compact_size(torrent.size)} | {torrent.volume_factor} | {torrent.seeders}↑", meta.resource_term, meta.video_term, ] diff --git a/app/modules/filemanager/__init__.py b/app/modules/filemanager/__init__.py index e0d0443d9..53f9b394e 100644 --- a/app/modules/filemanager/__init__.py +++ b/app/modules/filemanager/__init__.py @@ -17,7 +17,7 @@ from app.modules.filemanager.transhandler import TransHandler from app.schemas import TransferInfo, ExistMediaInfo, TmdbEpisode, TransferDirectoryConf, FileItem, StorageUsage from app.schemas.types import MUSIC_ENTITY_ALBUM, MediaType, ModuleType, OtherModulesType from app.adapters.system.host import SystemUtils -from app.domain.string import StringUtils +from app.foundation import text as text_tools class FileManagerModule(_ModuleBase): @@ -596,7 +596,7 @@ class FileManagerModule(_ModuleBase): return ( file_meta.disc_number, file_meta.track_number, - StringUtils.clear_upper(file_meta.title or file_path.stem), + text_tools.normalize_upper(file_meta.title or file_path.stem), ) @classmethod @@ -606,7 +606,7 @@ class FileManagerModule(_ModuleBase): mediainfo: MusicInfo, ) -> bool: """按曲名和可用曲序判断单曲是否存在,避免专辑内任一文件造成误判。""" - target_title = StringUtils.clear_upper(mediainfo.title or "") + target_title = text_tools.normalize_upper(mediainfo.title or "") target_track = getattr(mediainfo, "track_number", None) target_disc = getattr(mediainfo, "disc_number", None) if not target_title: diff --git a/app/modules/filemanager/storages/alipan.py b/app/modules/filemanager/storages/alipan.py index c8e922c73..a55a9c562 100644 --- a/app/modules/filemanager/storages/alipan.py +++ b/app/modules/filemanager/storages/alipan.py @@ -17,7 +17,7 @@ from app.schemas.exception import StorageQueryError from app.schemas.types import StorageSchema from app.adapters.network.http import RequestUtils from app.foundation.singleton import WeakSingleton -from app.domain.string import StringUtils +from app.foundation import temporal as time_tools lock = threading.Lock() @@ -305,7 +305,7 @@ class AliPan(StorageBase, metaclass=WeakSingleton): name=fileinfo.get("name"), basename=fileinfo.get("name"), size=fileinfo.get("size"), - modify_time=StringUtils.str_to_timestamp(fileinfo.get("updated_at")), + modify_time=time_tools.parse_timestamp(fileinfo.get("updated_at")), drive_id=fileinfo.get("drive_id"), ) else: @@ -319,7 +319,7 @@ class AliPan(StorageBase, metaclass=WeakSingleton): basename=Path(fileinfo.get("name")).stem, size=fileinfo.get("size"), extension=fileinfo.get("file_extension"), - modify_time=StringUtils.str_to_timestamp(fileinfo.get("updated_at")), + modify_time=time_tools.parse_timestamp(fileinfo.get("updated_at")), thumbnail=fileinfo.get("thumbnail"), drive_id=fileinfo.get("drive_id"), ) diff --git a/app/modules/filemanager/storages/rclone.py b/app/modules/filemanager/storages/rclone.py index ae2442279..c5cc884cd 100644 --- a/app/modules/filemanager/storages/rclone.py +++ b/app/modules/filemanager/storages/rclone.py @@ -12,7 +12,7 @@ from app.runtime.log import logger from app.modules.filemanager.storages import StorageBase, transfer_process from app.schemas.exception import StorageQueryError from app.schemas.types import StorageSchema -from app.domain.string import StringUtils +from app.foundation import temporal as time_tools from app.adapters.system.host import SystemUtils _MAX_FOLDER_LOCKS = 4096 @@ -127,7 +127,7 @@ class Rclone(StorageBase): path=f"{parent}{item.get('Name')}" + "/", name=item.get("Name"), basename=item.get("Name"), - modify_time=StringUtils.str_to_timestamp(item.get("ModTime")) + modify_time=time_tools.parse_timestamp(item.get("ModTime")) ) else: return schemas.FileItem( @@ -138,7 +138,7 @@ class Rclone(StorageBase): basename=Path(item.get("Name")).stem, extension=Path(item.get("Name")).suffix[1:], size=item.get("Size"), - modify_time=StringUtils.str_to_timestamp(item.get("ModTime")) + modify_time=time_tools.parse_timestamp(item.get("ModTime")) ) @staticmethod diff --git a/app/modules/filemanager/storages/u115.py b/app/modules/filemanager/storages/u115.py index 53b5962db..f057d29ec 100644 --- a/app/modules/filemanager/storages/u115.py +++ b/app/modules/filemanager/storages/u115.py @@ -20,7 +20,7 @@ from app.modules.filemanager.storages import transfer_process from app.schemas.exception import StorageQueryError from app.schemas.types import StorageSchema from app.foundation.singleton import WeakSingleton -from app.domain.string import StringUtils +from app.foundation import size as size_tools from app.runtime.rate import QpsRateLimiter, RateStats @@ -690,7 +690,7 @@ class U115Pan(StorageBase, metaclass=WeakSingleton): if info_resp["file_category"] == "1" else None, pickcode=info_resp["pick_code"], - size=StringUtils.num_filesize(info_resp["size"]) + size=size_tools.parse_size(info_resp["size"]) if info_resp["file_category"] == "1" else None, modify_time=info_resp["utime"], @@ -742,7 +742,7 @@ class U115Pan(StorageBase, metaclass=WeakSingleton): # 初始化进度条 logger.info( - f"【115】开始上传: {local_path} -> {target_path},分片大小:{StringUtils.str_filesize(part_size)}" + f"【115】开始上传: {local_path} -> {target_path},分片大小:{size_tools.format_compact_size(part_size)}" ) progress_callback = transfer_process(local_path.as_posix()) diff --git a/app/modules/filter/__init__.py b/app/modules/filter/__init__.py index abce1bfb9..12fcb9df0 100644 --- a/app/modules/filter/__init__.py +++ b/app/modules/filter/__init__.py @@ -12,7 +12,7 @@ from app.modules.filter.RuleParser import RuleParser from app.modules.filter.builtin_rules import BUILTIN_RULE_SET from app.schemas.types import ModuleType, OtherModulesType, SystemConfigKey from app.adapters.system import rust as rust_accel -from app.domain.string import StringUtils +from app.foundation import size as size_tools _SIZE_UNIT = 1024 * 1024 @@ -388,7 +388,7 @@ class FilterModule(_ModuleBase): if not self.__match_size(torrent, size_range): # 大小范围不匹配 logger.debug(f"种子 {torrent.site_name} - {torrent.title} 大小 " - f"{StringUtils.str_filesize(torrent.size)} 不在范围 {size_range}MB") + f"{size_tools.format_compact_size(torrent.size)} 不在范围 {size_range}MB") return False if seeders: if torrent.seeders < int(seeders): diff --git a/app/modules/indexer/__init__.py b/app/modules/indexer/__init__.py index 4f617ccf8..1e9d2297c 100644 --- a/app/modules/indexer/__init__.py +++ b/app/modules/indexer/__init__.py @@ -21,7 +21,8 @@ from app.schemas.media import resolve_media_identity from app.modules.indexer.spider.yema import YemaSpider from app.schemas import SiteUserData from app.schemas.types import MediaType, ModuleType, OtherModulesType -from app.domain.string import StringUtils +from app.domain import site as site_rules +from app.foundation import text as text_tools SPIDER_PARSER_CLASSES = { "TNodeSpider": TNodeSpider, @@ -101,13 +102,13 @@ class IndexerModule(_ModuleBase): # 可能为关键字或ttxxxx if search_word \ and site.get('language') == "en" \ - and StringUtils.is_chinese(search_word): + and text_tools.contains_chinese(search_word): # 不支持中文 logger.warn(f"{site.get('name')} 不支持中文搜索") return False # 站点流控 - state, msg = SitesHelper().check(StringUtils.get_url_domain(site.get("domain"))) + state, msg = SitesHelper().check(site_rules.extract_domain(site.get("domain"))) if state: logger.warn(msg) return False @@ -124,14 +125,14 @@ class IndexerModule(_ModuleBase): if not text: return text # 去除特殊字符和多余空格 - return StringUtils.clear(text, replace_word=" ", allow_space=True) + return text_tools.remove_punctuation(text, replacement=" ", allow_space=True) @staticmethod def __indexer_statistic(site: dict, error_flag: bool = False, seconds: int = 0) -> None: """ 索引器统计 """ - domain = StringUtils.get_url_domain(site.get("domain")) + domain = site_rules.extract_domain(site.get("domain")) if error_flag: SiteOper().fail(domain) else: @@ -142,7 +143,7 @@ class IndexerModule(_ModuleBase): """ 异步索引器统计 """ - domain = StringUtils.get_url_domain(site.get("domain")) + domain = site_rules.extract_domain(site.get("domain")) if error_flag: await SiteOper().async_fail(domain) else: @@ -633,7 +634,7 @@ class IndexerModule(_ModuleBase): site_obj.parse() logger.debug(f"站点 {site.get('name')} 数据解析完成") return SiteUserData( - domain=StringUtils.get_url_domain(site.get("url")), + domain=site_rules.extract_domain(site.get("url")), userid=site_obj.userid, username=site_obj.username, user_level=site_obj.user_level, diff --git a/app/modules/indexer/parser/__init__.py b/app/modules/indexer/parser/__init__.py index af5a95e30..2dad3d202 100644 --- a/app/modules/indexer/parser/__init__.py +++ b/app/modules/indexer/parser/__init__.py @@ -13,7 +13,7 @@ from app.adapters.network.cloudflare import under_challenge from app.runtime.log import logger from app.adapters.network.http import RequestUtils from app.domain.site import SiteUtils -from app.domain.string import StringUtils +from app.foundation import size as size_tools # 站点框架 @@ -183,7 +183,7 @@ class SiteParserBase(metaclass=ABCMeta): """ 将站点页面中的文件大小文本转换为字节。 """ - return StringUtils.num_filesize(text) + return size_tools.parse_size(text) def parse(self): """ diff --git a/app/modules/indexer/parser/bitpt.py b/app/modules/indexer/parser/bitpt.py index d42a912eb..1e08700d1 100644 --- a/app/modules/indexer/parser/bitpt.py +++ b/app/modules/indexer/parser/bitpt.py @@ -10,7 +10,8 @@ from urllib.parse import urljoin, urlencode from bs4 import BeautifulSoup from app.modules.indexer.parser import SiteParserBase, SiteSchema -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation import temporal as time_tools class BitptSiteUserInfo(SiteParserBase): schema = SiteSchema.Bitpt @@ -52,10 +53,10 @@ class BitptSiteUserInfo(SiteParserBase): self.userid = info_dict.get('UID') self.username = info_dict.get('用户名').split('\xa0')[0] if '用户名' in info_dict else None self.user_level = info_dict.get('用户级别') if '用户级别' in info_dict else None - self.join_at = StringUtils.unify_datetime_str(info_dict.get('注册时间')) if '注册时间' in info_dict else None + self.join_at = time_tools.normalize_datetime(info_dict.get('注册时间')) if '注册时间' in info_dict else None - self.upload = StringUtils.num_filesize(info_dict.get('上传流量')) if '上传流量' in info_dict else 0 - self.download = StringUtils.num_filesize(info_dict.get('下载流量')) if '下载流量' in info_dict else 0 + self.upload = size_tools.parse_size(info_dict.get('上传流量')) if '上传流量' in info_dict else 0 + self.download = size_tools.parse_size(info_dict.get('下载流量')) if '下载流量' in info_dict else 0 self.ratio = float(info_dict.get('共享率')) if '共享率' in info_dict else 0 bonus_str = info_dict.get('星辰', '') self.bonus = float(re.search(r'累计([\d\.]+)', bonus_str).group(1)) if re.search(r'累计([\d\.]+)', bonus_str) else 0 @@ -71,7 +72,7 @@ class BitptSiteUserInfo(SiteParserBase): match = re.search(r'当前上传的种子\((\d+)个, 共([\d\.]+ [KMGT]B)\)', seeding_link) if match: self.seeding = int(match.group(1)) - self.seeding_size = StringUtils.num_filesize(match.group(2)) + self.seeding_size = size_tools.parse_size(match.group(2)) else: self.seeding = 0 self.seeding_size = 0 @@ -102,7 +103,7 @@ class BitptSiteUserInfo(SiteParserBase): size_text = size_a.text.strip() if size_a else size_td.text.strip() if size_text: page_seeding += 1 - page_seeding_size += StringUtils.num_filesize(size_text) + page_seeding_size += size_tools.parse_size(size_text) return page_seeding, page_seeding_size def _parse_message_unread_links(self, html_text: str, msg_links: list) -> Optional[str]: diff --git a/app/modules/indexer/parser/discuz.py b/app/modules/indexer/parser/discuz.py index 00c0ebcfb..6ad6ab7d1 100644 --- a/app/modules/indexer/parser/discuz.py +++ b/app/modules/indexer/parser/discuz.py @@ -5,7 +5,10 @@ from typing import Optional from lxml import etree from app.modules.indexer.parser import SiteParserBase, SiteSchema -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation import temporal as time_tools +from app.foundation import text as text_tools +from app.foundation.dom import DomUtils class DiscuzUserInfo(SiteParserBase): @@ -38,7 +41,7 @@ class DiscuzUserInfo(SiteParserBase): """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return None # 用户等级 @@ -49,29 +52,29 @@ class DiscuzUserInfo(SiteParserBase): # 加入日期 join_at_text = html.xpath('//li[em[text()="注册时间"]]/text()') if join_at_text: - self.join_at = StringUtils.unify_datetime_str(join_at_text[0].strip()) + self.join_at = time_tools.normalize_datetime(join_at_text[0].strip()) # 分享率 ratio_text = html.xpath('//li[contains(.//text(), "分享率")]//text()') if ratio_text: ratio_match = re.search(r"\(([\d,.]+)\)", ratio_text[0]) if ratio_match and ratio_match.group(1).strip(): - self.bonus = StringUtils.str_float(ratio_match.group(1)) + self.bonus = text_tools.parse_float(ratio_match.group(1)) # 积分 bouns_text = html.xpath('//li[em[text()="积分"]]/text()') if bouns_text: - self.bonus = StringUtils.str_float(bouns_text[0].strip()) + self.bonus = text_tools.parse_float(bouns_text[0].strip()) # 上传 upload_text = html.xpath('//li[em[contains(text(),"上传量")]]/text()') if upload_text: - self.upload = StringUtils.num_filesize(upload_text[0].strip().split('/')[-1]) + self.upload = size_tools.parse_size(upload_text[0].strip().split('/')[-1]) # 下载 download_text = html.xpath('//li[em[contains(text(),"下载量")]]/text()') if download_text: - self.download = StringUtils.num_filesize(download_text[0].strip().split('/')[-1]) + self.download = size_tools.parse_size(download_text[0].strip().split('/')[-1]) finally: if html is not None: del html @@ -85,7 +88,7 @@ class DiscuzUserInfo(SiteParserBase): """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return None size_col = 3 @@ -108,8 +111,8 @@ class DiscuzUserInfo(SiteParserBase): page_seeding = len(seeding_sizes) for i in range(0, len(seeding_sizes)): - size = StringUtils.num_filesize(seeding_sizes[i].xpath("string(.)").strip()) - seeders = StringUtils.str_int(seeding_seeders[i]) + size = size_tools.parse_size(seeding_sizes[i].xpath("string(.)").strip()) + seeders = text_tools.parse_int(seeding_seeders[i]) page_seeding_size += size page_seeding_info.append([seeders, size]) diff --git a/app/modules/indexer/parser/file_list.py b/app/modules/indexer/parser/file_list.py index efeb2ad2a..000bf807f 100644 --- a/app/modules/indexer/parser/file_list.py +++ b/app/modules/indexer/parser/file_list.py @@ -5,7 +5,10 @@ from typing import Optional from lxml import etree from app.modules.indexer.parser import SiteParserBase, SiteSchema -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation import temporal as time_tools +from app.foundation import text as text_tools +from app.foundation.dom import DomUtils class FileListSiteUserInfo(SiteParserBase): @@ -46,22 +49,22 @@ class FileListSiteUserInfo(SiteParserBase): try: upload_html = html.xpath('//table//tr/td[text()="Uploaded"]/following-sibling::td//text()') if upload_html: - self.upload = StringUtils.num_filesize(upload_html[0]) + self.upload = size_tools.parse_size(upload_html[0]) download_html = html.xpath('//table//tr/td[text()="Downloaded"]/following-sibling::td//text()') if download_html: - self.download = StringUtils.num_filesize(download_html[0]) + self.download = size_tools.parse_size(download_html[0]) ratio_html = html.xpath('//table//tr/td[text()="Share ratio"]/following-sibling::td//text()') if ratio_html: - share_ratio = StringUtils.str_float(ratio_html[0]) + share_ratio = text_tools.parse_float(ratio_html[0]) else: share_ratio = 0 self.ratio = 0 if self.download == 0 else share_ratio seed_html = html.xpath('//table//tr/td[text()="Seed bonus"]/following-sibling::td//text()') if seed_html: - self.seeding = StringUtils.str_int(seed_html[1]) - self.seeding_size = StringUtils.num_filesize(seed_html[3]) + self.seeding = text_tools.parse_int(seed_html[1]) + self.seeding_size = size_tools.parse_size(seed_html[3]) user_level_html = html.xpath('//table//tr/td[text()="Class"]/following-sibling::td//text()') if user_level_html: @@ -70,11 +73,11 @@ class FileListSiteUserInfo(SiteParserBase): join_at_html = html.xpath('//table//tr/td[contains(text(), "Join")]/following-sibling::td//text()') if join_at_html: join_at = (join_at_html[0].split("("))[0].strip() - self.join_at = StringUtils.unify_datetime_str(join_at) + self.join_at = time_tools.normalize_datetime(join_at) bonus_html = html.xpath('//a[contains(@href, "shop.php")]') if bonus_html: - self.bonus = StringUtils.str_float(bonus_html[0].xpath("string(.)").strip()) + self.bonus = text_tools.parse_float(bonus_html[0].xpath("string(.)").strip()) finally: if html is not None: del html @@ -88,7 +91,7 @@ class FileListSiteUserInfo(SiteParserBase): """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return None size_col = 6 @@ -100,8 +103,8 @@ class FileListSiteUserInfo(SiteParserBase): seeding_seeders = html.xpath(f'//table/tr[position()>1]/td[{seeders_col}]') if seeding_sizes and seeding_seeders: for i in range(0, len(seeding_sizes)): - size = StringUtils.num_filesize(seeding_sizes[i].xpath("string(.)").strip()) - seeders = StringUtils.str_int(seeding_seeders[i].xpath("string(.)").strip()) + size = size_tools.parse_size(seeding_sizes[i].xpath("string(.)").strip()) + seeders = text_tools.parse_int(seeding_seeders[i].xpath("string(.)").strip()) page_seeding_size += size page_seeding_info.append([seeders, size]) diff --git a/app/modules/indexer/parser/gazelle.py b/app/modules/indexer/parser/gazelle.py index bc285e43e..b8440002d 100644 --- a/app/modules/indexer/parser/gazelle.py +++ b/app/modules/indexer/parser/gazelle.py @@ -5,7 +5,10 @@ from typing import Optional from lxml import etree from app.modules.indexer.parser import SiteParserBase, SiteSchema -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation import temporal as time_tools +from app.foundation import text as text_tools +from app.foundation.dom import DomUtils class GazelleSiteUserInfo(SiteParserBase): @@ -26,19 +29,19 @@ class GazelleSiteUserInfo(SiteParserBase): tmps = html.xpath('//*[@id="header-uploaded-value"]/@data-value') if tmps: - self.upload = StringUtils.num_filesize(tmps[0]) + self.upload = size_tools.parse_size(tmps[0]) else: tmps = html.xpath('//li[@id="stats_seeding"]/span/text()') if tmps: - self.upload = StringUtils.num_filesize(tmps[0]) + self.upload = size_tools.parse_size(tmps[0]) tmps = html.xpath('//*[@id="header-downloaded-value"]/@data-value') if tmps: - self.download = StringUtils.num_filesize(tmps[0]) + self.download = size_tools.parse_size(tmps[0]) else: tmps = html.xpath('//li[@id="stats_leeching"]/span/text()') if tmps: - self.download = StringUtils.num_filesize(tmps[0]) + self.download = size_tools.parse_size(tmps[0]) self.ratio = 0.0 if self.download <= 0.0 else round(self.upload / self.download, 3) @@ -46,14 +49,14 @@ class GazelleSiteUserInfo(SiteParserBase): if tmps: bonus_match = re.search(r"([\d,.]+)", tmps[0]) if bonus_match and bonus_match.group(1).strip(): - self.bonus = StringUtils.str_float(bonus_match.group(1)) + self.bonus = text_tools.parse_float(bonus_match.group(1)) else: tmps = html.xpath('//a[contains(@href, "bonus")]') if tmps: bonus_text = tmps[0].xpath("string(.)") bonus_match = re.search(r"([\d,.]+)", bonus_text) if bonus_match and bonus_match.group(1).strip(): - self.bonus = StringUtils.str_float(bonus_match.group(1)) + self.bonus = text_tools.parse_float(bonus_match.group(1)) finally: if html is not None: del html @@ -69,7 +72,7 @@ class GazelleSiteUserInfo(SiteParserBase): """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return None # 用户等级 @@ -84,12 +87,12 @@ class GazelleSiteUserInfo(SiteParserBase): # 加入日期 join_at_text = html.xpath('//*[@id="join-date-value"]/@data-value') if join_at_text: - self.join_at = StringUtils.unify_datetime_str(join_at_text[0].strip()) + self.join_at = time_tools.normalize_datetime(join_at_text[0].strip()) else: join_at_text = html.xpath( '//div[contains(@class, "box_userinfo_stats")]//li[contains(text(), "加入时间")]/span/text()') if join_at_text: - self.join_at = StringUtils.unify_datetime_str(join_at_text[0].strip()) + self.join_at = time_tools.normalize_datetime(join_at_text[0].strip()) finally: if html is not None: del html @@ -103,7 +106,7 @@ class GazelleSiteUserInfo(SiteParserBase): """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return None size_col = 3 @@ -122,7 +125,7 @@ class GazelleSiteUserInfo(SiteParserBase): page_seeding = len(seeding_sizes) for i in range(0, len(seeding_sizes)): - size = StringUtils.num_filesize(seeding_sizes[i].xpath("string(.)").strip()) + size = size_tools.parse_size(seeding_sizes[i].xpath("string(.)").strip()) seeders = int(seeding_seeders[i]) page_seeding_size += size diff --git a/app/modules/indexer/parser/hddolby.py b/app/modules/indexer/parser/hddolby.py index 05fa3032f..3ac4f19a8 100644 --- a/app/modules/indexer/parser/hddolby.py +++ b/app/modules/indexer/parser/hddolby.py @@ -5,7 +5,7 @@ from typing import Optional, Tuple from app.runtime.log import logger from app.modules.indexer.parser import SiteParserBase, SiteSchema from app.modules.indexer.parser.nexus_php import NexusPhpSiteUserInfo -from app.domain.string import StringUtils +from app.domain import site as site_rules class HDDolbySiteUserInfo(SiteParserBase): @@ -44,7 +44,7 @@ class HDDolbySiteUserInfo(SiteParserBase): 获取站点页面地址 """ # 更换api地址 - self._base_url = f"https://api.{StringUtils.get_url_domain(self._base_url)}" + self._base_url = f"https://api.{site_rules.extract_domain(self._base_url)}" self._user_traffic_page = None self._user_detail_page = None self._user_basic_page = "api/v1/user/data" diff --git a/app/modules/indexer/parser/ipt_project.py b/app/modules/indexer/parser/ipt_project.py index 901594bb6..d8cd867f1 100644 --- a/app/modules/indexer/parser/ipt_project.py +++ b/app/modules/indexer/parser/ipt_project.py @@ -5,7 +5,10 @@ from typing import Optional from lxml import etree from app.modules.indexer.parser import SiteParserBase, SiteSchema -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation import temporal as time_tools +from app.foundation import text as text_tools +from app.foundation.dom import DomUtils class IptSiteUserInfo(SiteParserBase): @@ -28,12 +31,12 @@ class IptSiteUserInfo(SiteParserBase): tmps = html.xpath('//div[@class = "stats"]/div/div') if tmps: - self.upload = StringUtils.num_filesize(str(tmps[0].xpath('span/text()')[1]).strip()) - self.download = StringUtils.num_filesize(str(tmps[0].xpath('span/text()')[2]).strip()) - self.seeding = StringUtils.str_int(tmps[0].xpath('a')[2].xpath('text()')[0]) - self.leeching = StringUtils.str_int(tmps[0].xpath('a')[2].xpath('text()')[1]) - self.ratio = StringUtils.str_float(str(tmps[0].xpath('span/text()')[0]).strip().replace('-', '0')) - self.bonus = StringUtils.str_float(tmps[0].xpath('a')[3].xpath('text()')[0]) + self.upload = size_tools.parse_size(str(tmps[0].xpath('span/text()')[1]).strip()) + self.download = size_tools.parse_size(str(tmps[0].xpath('span/text()')[2]).strip()) + self.seeding = text_tools.parse_int(tmps[0].xpath('a')[2].xpath('text()')[0]) + self.leeching = text_tools.parse_int(tmps[0].xpath('a')[2].xpath('text()')[1]) + self.ratio = text_tools.parse_float(str(tmps[0].xpath('span/text()')[0]).strip().replace('-', '0')) + self.bonus = text_tools.parse_float(tmps[0].xpath('a')[3].xpath('text()')[0]) finally: if html is not None: del html @@ -44,7 +47,7 @@ class IptSiteUserInfo(SiteParserBase): def _parse_user_detail_info(self, html_text: str): html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return user_levels_text = html.xpath('//tr/th[text()="Class"]/following-sibling::td[1]/text()') @@ -54,7 +57,7 @@ class IptSiteUserInfo(SiteParserBase): # 加入日期 join_at_text = html.xpath('//tr/th[text()="Join date"]/following-sibling::td[1]/text()') if join_at_text: - self.join_at = StringUtils.unify_datetime_str(join_at_text[0].split(' (')[0]) + self.join_at = time_tools.normalize_datetime(join_at_text[0].split(' (')[0]) finally: if html is not None: del html @@ -62,7 +65,7 @@ class IptSiteUserInfo(SiteParserBase): def _parse_user_torrent_seeding_info(self, html_text: str, multi_page: bool = False) -> Optional[str]: html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return None # seeding start seeding_end_pos = 3 @@ -80,7 +83,7 @@ class IptSiteUserInfo(SiteParserBase): per_size = per_size.split('(')[-1] per_size = per_size.split(')')[0] - page_seeding_size += StringUtils.num_filesize(per_size) + page_seeding_size += size_tools.parse_size(per_size) self.seeding = page_seeding self.seeding_size = page_seeding_size diff --git a/app/modules/indexer/parser/mtorrent.py b/app/modules/indexer/parser/mtorrent.py index fbe67495d..0df4f83df 100644 --- a/app/modules/indexer/parser/mtorrent.py +++ b/app/modules/indexer/parser/mtorrent.py @@ -5,7 +5,7 @@ from urllib.parse import urljoin from app.runtime.log import logger from app.modules.indexer.parser import SiteParserBase, SiteSchema -from app.domain.string import StringUtils +from app.domain import site as site_rules class MTorrentSiteUserInfo(SiteParserBase): @@ -39,7 +39,7 @@ class MTorrentSiteUserInfo(SiteParserBase): 获取站点页面地址 """ # 更换api地址 - self._base_url = f"https://api.{StringUtils.get_url_domain(self._base_url)}" + self._base_url = f"https://api.{site_rules.extract_domain(self._base_url)}" self._user_traffic_page = None self._user_detail_page = None self._user_basic_page = "api/member/profile" diff --git a/app/modules/indexer/parser/nexus_audiences.py b/app/modules/indexer/parser/nexus_audiences.py index a010e83a9..e1289545a 100644 --- a/app/modules/indexer/parser/nexus_audiences.py +++ b/app/modules/indexer/parser/nexus_audiences.py @@ -8,7 +8,9 @@ from lxml import etree from app.runtime.log import logger from app.modules.indexer.parser import SiteSchema from app.modules.indexer.parser.nexus_php import NexusPhpSiteUserInfo -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation import text as text_tools +from app.foundation.dom import DomUtils class NexusAudiencesSiteUserInfo(NexusPhpSiteUserInfo): @@ -32,7 +34,7 @@ class NexusAudiencesSiteUserInfo(NexusPhpSiteUserInfo): """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): super()._parse_message_unread(html_text) return @@ -62,7 +64,7 @@ class NexusAudiencesSiteUserInfo(NexusPhpSiteUserInfo): """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return None message_links = self.__parse_table_unread_message_links(html) @@ -86,7 +88,7 @@ class NexusAudiencesSiteUserInfo(NexusPhpSiteUserInfo): """ html = etree.HTML(html_text) try: - if StringUtils.is_valid_html_element(html): + if DomUtils.has_child_elements(html): head = self.__extract_first_text( html, '//*[contains(concat(" ", normalize-space(@class), " "), " pm-hero__title ")]' @@ -350,7 +352,7 @@ class NexusAudiencesSiteUserInfo(NexusPhpSiteUserInfo): """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return for user_node in html.xpath('//*[@data-uploader-url or @data-uploader-stats]'): @@ -432,18 +434,18 @@ class NexusAudiencesSiteUserInfo(NexusPhpSiteUserInfo): metric_key = field or tone or label if metric_key in {"uploaded", "上传量", "upload"}: - self.upload = StringUtils.num_filesize(value) + self.upload = size_tools.parse_size(value) elif metric_key in {"downloaded", "下载量", "download"}: - self.download = StringUtils.num_filesize(value) + self.download = size_tools.parse_size(value) elif metric_key in {"bonus", "爆米花"}: - self.bonus = StringUtils.str_float(value) + self.bonus = text_tools.parse_float(value) elif metric_key == "ratio": - self.ratio = StringUtils.str_float(value) + self.ratio = text_tools.parse_float(value) elif metric_key in {"active", "活跃"}: active_match = re.search(r"↑\s*(\d+)\s*/\s*↓\s*(\d+)", value) if active_match: - self.seeding = StringUtils.str_int(active_match.group(1)) - self.leeching = StringUtils.str_int(active_match.group(2)) + self.seeding = text_tools.parse_int(active_match.group(1)) + self.leeching = text_tools.parse_int(active_match.group(2)) def __parse_inbox_unread(self, message_link): """ @@ -482,7 +484,7 @@ class NexusAudiencesSiteUserInfo(NexusPhpSiteUserInfo): inbox_count = re.search(r"(?:收件箱\s*)?(\d[\d,]*)\s*/\s*(\d[\d,]*)", text) if inbox_count: - return StringUtils.str_int(inbox_count.group(2)) + return text_tools.parse_int(inbox_count.group(2)) return None @@ -500,7 +502,7 @@ class NexusAudiencesSiteUserInfo(NexusPhpSiteUserInfo): text = re.sub(r"\s+", " ", text.replace("\xa0", " ")).strip() single_count = re.fullmatch(r"(\d[\d,]*)", text) if single_count: - return StringUtils.str_int(single_count.group(1)) + return text_tools.parse_int(single_count.group(1)) return None @staticmethod @@ -526,15 +528,15 @@ class NexusAudiencesSiteUserInfo(NexusPhpSiteUserInfo): return html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return total_row = html.xpath('//table[@class="table table-bordered"]//tr[td[1][normalize-space()="Total"]]') if not total_row: return seeding_count = total_row[0].xpath('./td[2]/text()') seeding_size = total_row[0].xpath('./td[3]/text()') - self.seeding = StringUtils.str_int(seeding_count[0]) if seeding_count else 0 - self.seeding_size = StringUtils.num_filesize(seeding_size[0].strip()) if seeding_size else 0 + self.seeding = text_tools.parse_int(seeding_count[0]) if seeding_count else 0 + self.seeding_size = size_tools.parse_size(seeding_size[0].strip()) if seeding_size else 0 finally: if html is not None: del html diff --git a/app/modules/indexer/parser/nexus_hhanclub.py b/app/modules/indexer/parser/nexus_hhanclub.py index 6ff4f1f25..4adec07a9 100644 --- a/app/modules/indexer/parser/nexus_hhanclub.py +++ b/app/modules/indexer/parser/nexus_hhanclub.py @@ -5,7 +5,10 @@ from lxml import etree from app.modules.indexer.parser import SiteSchema from app.modules.indexer.parser.nexus_php import NexusPhpSiteUserInfo -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation import temporal as time_tools +from app.foundation import text as text_tools +from app.foundation.dom import DomUtils class NexusHhanclubSiteUserInfo(NexusPhpSiteUserInfo): @@ -27,11 +30,11 @@ class NexusHhanclubSiteUserInfo(NexusPhpSiteUserInfo): html.xpath('//*[@id="user-info-panel"]/div[2]/div[1]/div[1]/div/text()')[0]) # 计算分享率 - self.upload = StringUtils.num_filesize(upload_match.group(1).strip()) if upload_match else 0 - self.download = StringUtils.num_filesize(download_match.group(1).strip()) if download_match else 0 + self.upload = size_tools.parse_size(upload_match.group(1).strip()) if upload_match else 0 + self.download = size_tools.parse_size(download_match.group(1).strip()) if download_match else 0 # 优先使用页面上的分享率 calc_ratio = 0.0 if self.download <= 0.0 else round(self.upload / self.download, 3) - self.ratio = StringUtils.str_float(ratio_match.group(1)) if ( + self.ratio = text_tools.parse_float(ratio_match.group(1)) if ( ratio_match and ratio_match.group(1).strip()) else calc_ratio finally: if html is not None: @@ -47,12 +50,12 @@ class NexusHhanclubSiteUserInfo(NexusPhpSiteUserInfo): html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return # 加入时间 join_at_text = html.xpath('//span[contains(text(), "加入日期")]/following-sibling::span/span/@title') if join_at_text: - self.join_at = StringUtils.unify_datetime_str(join_at_text[0].strip()) + self.join_at = time_tools.normalize_datetime(join_at_text[0].strip()) finally: if html is not None: del html diff --git a/app/modules/indexer/parser/nexus_php.py b/app/modules/indexer/parser/nexus_php.py index e062c8e37..97e262826 100644 --- a/app/modules/indexer/parser/nexus_php.py +++ b/app/modules/indexer/parser/nexus_php.py @@ -7,7 +7,9 @@ from lxml import etree from app.runtime.log import logger from app.modules.indexer.parser import SiteParserBase, SiteSchema -from app.domain.string import StringUtils +from app.foundation import temporal as time_tools +from app.foundation import text as text_tools +from app.foundation.dom import DomUtils class NexusPhpSiteUserInfo(SiteParserBase): @@ -40,7 +42,7 @@ class NexusPhpSiteUserInfo(SiteParserBase): """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return message_labels = html.xpath('//a[@href="messages.php"]/..') @@ -52,9 +54,9 @@ class NexusPhpSiteUserInfo(SiteParserBase): message_unread_match = re.findall(r"[^Date](信息箱\s*|\((?![^)]*:)|你有\xa0)(\d+)", message_text) if message_unread_match and len(message_unread_match[-1]) == 2: - self.message_unread = StringUtils.str_int(message_unread_match[-1][1]) + self.message_unread = text_tools.parse_int(message_unread_match[-1][1]) elif message_text.isdigit(): - self.message_unread = StringUtils.str_int(message_text) + self.message_unread = text_tools.parse_int(message_text) finally: if html is not None: del html @@ -71,7 +73,7 @@ class NexusPhpSiteUserInfo(SiteParserBase): html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return ret = html.xpath(f'//a[contains(@href, "userdetails") and contains(@href, "{self.userid}")]//b//text()') @@ -106,10 +108,10 @@ class NexusPhpSiteUserInfo(SiteParserBase): # 计算分享率 calc_ratio = 0.0 if self.download <= 0.0 else round(self.upload / self.download, 3) # 优先使用页面上的分享率 - self.ratio = StringUtils.str_float(ratio_match.group(1)) if ( + self.ratio = text_tools.parse_float(ratio_match.group(1)) if ( ratio_match and ratio_match.group(1).strip()) else calc_ratio leeching_match = re.search(r"(Torrents leeching|下载中)[\u4E00-\u9FA5\D\s]+(\d+)[\s\S]+<", html_text) - self.leeching = StringUtils.str_int(leeching_match.group(2)) if leeching_match and leeching_match.group( + self.leeching = text_tools.parse_int(leeching_match.group(2)) if leeching_match and leeching_match.group( 2).strip() else 0 html = etree.HTML(html_text) try: @@ -121,18 +123,18 @@ class NexusPhpSiteUserInfo(SiteParserBase): bonus_text = str(tmps[0]).strip() bonus_match = re.search(r"([\d,.]+)", bonus_text) if bonus_match and bonus_match.group(1).strip(): - self.bonus = StringUtils.str_float(bonus_match.group(1)) + self.bonus = text_tools.parse_float(bonus_match.group(1)) return bonus_match = re.search(r"mybonus.[\[\]::<>/a-zA-Z_\-=\"'\s#;.(使用魔力值豆]+\s*([\d,.]+)[<()&\s]", html_text) try: if bonus_match and bonus_match.group(1).strip(): - self.bonus = StringUtils.str_float(bonus_match.group(1)) + self.bonus = text_tools.parse_float(bonus_match.group(1)) return bonus_match = re.search(r"[魔力值|\]][\[\]::<>/a-zA-Z_\-=\"'\s#;]+\s*([\d,.]+|\"[\d,.]+\")[<>()&\s]", html_text, flags=re.S) if bonus_match and bonus_match.group(1).strip(): - self.bonus = StringUtils.str_float(bonus_match.group(1).strip('"')) + self.bonus = text_tools.parse_float(bonus_match.group(1).strip('"')) except Exception as err: logger.error(f"{self._site_name} 解析魔力值出错, 错误信息: {str(err)}") finally: @@ -146,18 +148,18 @@ class NexusPhpSiteUserInfo(SiteParserBase): :param html: :return: """ - if StringUtils.is_valid_html_element(html): + if DomUtils.has_child_elements(html): gold, silver, copper = None, None, None golds = html.xpath('//span[@class = "ucoin-symbol ucoin-gold"]//text()') if golds: - gold = StringUtils.str_float(str(golds[-1])) + gold = text_tools.parse_float(str(golds[-1])) silvers = html.xpath('//span[@class = "ucoin-symbol ucoin-silver"]//text()') if silvers: - silver = StringUtils.str_float(str(silvers[-1])) + silver = text_tools.parse_float(str(silvers[-1])) coppers = html.xpath('//span[@class = "ucoin-symbol ucoin-copper"]//text()') if coppers: - copper = StringUtils.str_float(str(coppers[-1])) + copper = text_tools.parse_float(str(coppers[-1])) if gold or silver or copper: gold = gold if gold else 0 silver = silver if silver else 0 @@ -174,7 +176,7 @@ class NexusPhpSiteUserInfo(SiteParserBase): """ html = etree.HTML(str(html_text).replace(r'\/', '/')) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return None # 首页存在扩展链接,使用扩展链接 @@ -215,7 +217,7 @@ class NexusPhpSiteUserInfo(SiteParserBase): for i in range(0, len(seeding_sizes)): size = self.num_filesize(seeding_sizes[i].xpath("string(.)").strip()) - seeders = StringUtils.str_int(seeding_seeders[i]) + seeders = text_tools.parse_int(seeding_seeders[i]) page_seeding_size += size page_seeding_info.append([seeders, size]) @@ -274,7 +276,7 @@ class NexusPhpSiteUserInfo(SiteParserBase): """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return self._get_user_level(html) @@ -287,7 +289,7 @@ class NexusPhpSiteUserInfo(SiteParserBase): '|//div/b[text()="加入日期"]/../text()' '|//*[@id="outer"]/table/tr/td/div/div[1]/div[2]/div[3]/span[1]/span/@title') if join_at_text: - self.join_at = StringUtils.unify_datetime_str(join_at_text[0].split(' (')[0].strip()) + self.join_at = time_tools.normalize_datetime(join_at_text[0].split(' (')[0].strip()) # 做种体积 & 做种数 # seeding 页面获取不到的话,此处再获取一次 @@ -300,7 +302,7 @@ class NexusPhpSiteUserInfo(SiteParserBase): tmp_seeding_info = [] for i in range(0, len(seeding_sizes)): size = self.num_filesize(seeding_sizes[i].xpath("string(.)").strip()) - seeders = StringUtils.str_int(seeding_seeders[i]) + seeders = text_tools.parse_int(seeding_seeders[i]) tmp_seeding_size += size tmp_seeding_info.append([seeders, size]) @@ -316,7 +318,7 @@ class NexusPhpSiteUserInfo(SiteParserBase): if seeding_sizes: seeding_match = re.search(r"总做种数:\s+(\d+)", seeding_sizes[0], re.IGNORECASE) seeding_size_match = re.search(r"总做种体积:\s+([\d,.\s]+[KMGTPI]*B)", seeding_sizes[0], re.IGNORECASE) - tmp_seeding = StringUtils.str_int(seeding_match.group(1)) if ( + tmp_seeding = text_tools.parse_int(seeding_match.group(1)) if ( seeding_match and seeding_match.group(1)) else 0 tmp_seeding_size = self.num_filesize( seeding_size_match.group(1).strip()) if seeding_size_match else 0 @@ -396,7 +398,7 @@ class NexusPhpSiteUserInfo(SiteParserBase): def _parse_message_unread_links(self, html_text: str, msg_links: list) -> Optional[str]: html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return None message_links = html.xpath('//tr[not(./td/img[@alt="Read"])]/td/a[contains(@href, "viewmessage")]/@href') @@ -415,7 +417,7 @@ class NexusPhpSiteUserInfo(SiteParserBase): def _parse_message_content(self, html_text): html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return None, None, None # 标题 message_head_text = None @@ -448,4 +450,4 @@ class NexusPhpSiteUserInfo(SiteParserBase): if not self.bonus: bonus_text = html.xpath('//tr/td[text()="魔力值" or text()="猫粮"]/following-sibling::td[1]/text()') if bonus_text: - self.bonus = StringUtils.str_float(bonus_text[0].strip()) + self.bonus = text_tools.parse_float(bonus_text[0].strip()) diff --git a/app/modules/indexer/parser/nexus_rabbit.py b/app/modules/indexer/parser/nexus_rabbit.py index 5e8b37120..40e05ed44 100644 --- a/app/modules/indexer/parser/nexus_rabbit.py +++ b/app/modules/indexer/parser/nexus_rabbit.py @@ -7,7 +7,10 @@ from urllib.parse import urljoin from app.runtime.log import logger from app.modules.indexer.parser import SiteSchema from app.modules.indexer.parser import SiteParserBase -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation import temporal as time_tools +from app.foundation import text as text_tools +from app.foundation.dom import DomUtils class NexusRabbitSiteUserInfo(SiteParserBase): @@ -73,7 +76,7 @@ class NexusRabbitSiteUserInfo(SiteParserBase): for torrent in torrents: seeders = int(torrent.get("seeders", 0)) - size = StringUtils.num_filesize(torrent.get("size")) + size = size_tools.parse_size(torrent.get("size")) seeding_size += size seeding_info.append([seeders, size]) @@ -115,13 +118,13 @@ class NexusRabbitSiteUserInfo(SiteParserBase): """只有奶糖余额才需要在 base 中获取,其它均可以在详情页拿到""" html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return bonus = html.xpath( '//div[contains(text(), "奶糖余额")]/following-sibling::div[1]/text()' ) if bonus: - self.bonus = StringUtils.str_float(bonus[0].strip()) + self.bonus = text_tools.parse_float(bonus[0].strip()) finally: if html is not None: del html @@ -129,7 +132,7 @@ class NexusRabbitSiteUserInfo(SiteParserBase): def _parse_user_detail_info(self, html_text: str): html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return # 缩小一下查找范围,所有的信息都在这个 div 里 user_info = html.xpath('//div[contains(@class, "layui-hares-user-info-right")]') @@ -147,20 +150,20 @@ class NexusRabbitSiteUserInfo(SiteParserBase): # 加入日期 if join_date := user_info.xpath('.//span[contains(text(), "注册日期")]/text()'): join_date = join_date[0].strip().split("\r")[0].removeprefix("注册日期:") - self.join_at = StringUtils.unify_datetime_str(join_date) + self.join_at = time_tools.normalize_datetime(join_date) # 上传量 if upload := user_info.xpath('.//span[contains(text(), "上传量")]/text()'): - self.upload = StringUtils.num_filesize( + self.upload = size_tools.parse_size( upload[0].strip().removeprefix("上传量:") ) # 下载量 if download := user_info.xpath('.//span[contains(text(), "下载量")]/text()'): - self.download = StringUtils.num_filesize( + self.download = size_tools.parse_size( download[0].strip().removeprefix("下载量:") ) # 分享率 if ratio := user_info.xpath('.//span[contains(text(), "分享率")]/em/text()'): - self.ratio = StringUtils.str_float(ratio[0].strip()) + self.ratio = text_tools.parse_float(ratio[0].strip()) finally: if html is not None: del html diff --git a/app/modules/indexer/parser/rousi.py b/app/modules/indexer/parser/rousi.py index 4ba6d72c9..26d079ae7 100644 --- a/app/modules/indexer/parser/rousi.py +++ b/app/modules/indexer/parser/rousi.py @@ -6,7 +6,8 @@ from typing import Optional, Tuple from app.runtime.log import logger from app.runtime.config import settings from app.adapters.network.http import RequestUtils -from app.domain.string import StringUtils +from app.domain import site as site_rules +from app.foundation import temporal as time_tools from app.modules.indexer.parser import SiteParserBase, SiteSchema @@ -23,7 +24,7 @@ class RousiSiteUserInfo(SiteParserBase): 配置 API 请求地址和请求头 使用 API v1 的 /profile 接口获取用户信息 """ - self._base_url = f"https://{StringUtils.get_url_domain(self._site_url)}" + self._base_url = f"https://{site_rules.extract_domain(self._site_url)}" self._user_basic_page = "api/v1/profile?include_fields[user]=seeding_leeching_data" self._user_basic_params = {} self._user_basic_headers = { @@ -97,7 +98,7 @@ class RousiSiteUserInfo(SiteParserBase): self.user_level = user_info.get("level_text") or user_info.get("role_text") # 注册时间:统一格式为 YYYY-MM-DD HH:MM:SS - join_at = StringUtils.unify_datetime_str(user_info.get("registered_at")) + join_at = time_tools.normalize_datetime(user_info.get("registered_at")) if join_at: # 确保格式为 YYYY-MM-DD HH:MM:SS (19位) if len(join_at) >= 19: @@ -219,7 +220,7 @@ class RousiSiteUserInfo(SiteParserBase): self.message_unread = len(messages) for messsage in messages: head = messsage.get("title") - date = StringUtils.unify_datetime_str(messsage.get("created_at")) + date = time_tools.normalize_datetime(messsage.get("created_at")) content = messsage.get("content") logger.debug(f"{self._site_name} 标题 {head} 时间 {date} 内容 {content}") self.message_unread_contents.append((head, date, content)) diff --git a/app/modules/indexer/parser/small_horse.py b/app/modules/indexer/parser/small_horse.py index 87a28f6c1..83b8203e0 100644 --- a/app/modules/indexer/parser/small_horse.py +++ b/app/modules/indexer/parser/small_horse.py @@ -5,7 +5,10 @@ from typing import Optional from lxml import etree from app.modules.indexer.parser import SiteParserBase, SiteSchema -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation import temporal as time_tools +from app.foundation import text as text_tools +from app.foundation.dom import DomUtils class SmallHorseSiteUserInfo(SiteParserBase): @@ -44,17 +47,17 @@ class SmallHorseSiteUserInfo(SiteParserBase): tmps = html.xpath('//ul[@class = "stats nobullet"]') if tmps: if tmps[1].xpath("li") and tmps[1].xpath("li")[0].xpath("span//text()"): - self.join_at = StringUtils.unify_datetime_str(tmps[1].xpath("li")[0].xpath("span//text()")[0]) - self.upload = StringUtils.num_filesize(str(tmps[1].xpath("li")[2].xpath("text()")[0]).split(":")[1].strip()) - self.download = StringUtils.num_filesize( + self.join_at = time_tools.normalize_datetime(tmps[1].xpath("li")[0].xpath("span//text()")[0]) + self.upload = size_tools.parse_size(str(tmps[1].xpath("li")[2].xpath("text()")[0]).split(":")[1].strip()) + self.download = size_tools.parse_size( str(tmps[1].xpath("li")[3].xpath("text()")[0]).split(":")[1].strip()) if tmps[1].xpath("li")[4].xpath("span//text()"): - self.ratio = StringUtils.str_float(str(tmps[1].xpath("li")[4].xpath("span//text()")[0]).replace('∞', '0')) + self.ratio = text_tools.parse_float(str(tmps[1].xpath("li")[4].xpath("span//text()")[0]).replace('∞', '0')) else: - self.ratio = StringUtils.str_float(str(tmps[1].xpath("li")[5].xpath("text()")[0]).split(":")[1]) - self.bonus = StringUtils.str_float(str(tmps[1].xpath("li")[5].xpath("text()")[0]).split(":")[1]) + self.ratio = text_tools.parse_float(str(tmps[1].xpath("li")[5].xpath("text()")[0]).split(":")[1]) + self.bonus = text_tools.parse_float(str(tmps[1].xpath("li")[5].xpath("text()")[0]).split(":")[1]) self.user_level = str(tmps[3].xpath("li")[0].xpath("text()")[0]).split(":")[1].strip() - self.leeching = StringUtils.str_int( + self.leeching = text_tools.parse_int( (tmps[4].xpath("li")[6].xpath("text()")[0]).split(":")[1].replace("[", "")) finally: if html is not None: @@ -72,7 +75,7 @@ class SmallHorseSiteUserInfo(SiteParserBase): """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return None size_col = 6 @@ -87,8 +90,8 @@ class SmallHorseSiteUserInfo(SiteParserBase): page_seeding = len(seeding_sizes) for i in range(0, len(seeding_sizes)): - size = StringUtils.num_filesize(seeding_sizes[i].xpath("string(.)").strip()) - seeders = StringUtils.str_int(seeding_seeders[i].xpath("string(.)").strip()) + size = size_tools.parse_size(seeding_sizes[i].xpath("string(.)").strip()) + seeders = text_tools.parse_int(seeding_seeders[i].xpath("string(.)").strip()) page_seeding_size += size page_seeding_info.append([seeders, size]) diff --git a/app/modules/indexer/parser/sunnypt.py b/app/modules/indexer/parser/sunnypt.py index cc16f33fa..60c2f5380 100644 --- a/app/modules/indexer/parser/sunnypt.py +++ b/app/modules/indexer/parser/sunnypt.py @@ -5,7 +5,7 @@ from urllib.parse import urlencode, urljoin from app.runtime.log import logger from app.modules.indexer.parser import SiteParserBase, SiteSchema -from app.domain.string import StringUtils +from app.foundation import temporal as time_tools class SunnyPTSiteUserInfo(SiteParserBase): @@ -75,7 +75,7 @@ class SunnyPTSiteUserInfo(SiteParserBase): self.userid = user_info.get("id") self.username = user_info.get("username") self.user_level = user_info.get("level") or str(user_info.get("class") or "") - self.join_at = StringUtils.unify_datetime_str(user_info.get("registered_at")) + self.join_at = time_tools.normalize_datetime(user_info.get("registered_at")) self.upload = int(user_info.get("uploaded") or 0) self.download = int(user_info.get("downloaded") or 0) self.ratio = float(user_info.get("ratio") or 0) @@ -122,7 +122,7 @@ class SunnyPTSiteUserInfo(SiteParserBase): continue title = message.get("title") content = message.get("content") - created_at = StringUtils.unify_datetime_str(message.get("created_at")) + created_at = time_tools.normalize_datetime(message.get("created_at")) message_id = message.get("id") if title and content and created_at: message_source = f"sunnypt-message:{message_id}" if message_id is not None else None diff --git a/app/modules/indexer/parser/tnode.py b/app/modules/indexer/parser/tnode.py index 63c596878..8adb33d5d 100644 --- a/app/modules/indexer/parser/tnode.py +++ b/app/modules/indexer/parser/tnode.py @@ -5,7 +5,7 @@ from typing import Optional from app.runtime.log import logger from app.modules.indexer.parser import SiteParserBase, SiteSchema -from app.domain.string import StringUtils +from app.foundation import temporal as time_tools class TNodeSiteUserInfo(SiteParserBase): @@ -49,7 +49,7 @@ class TNodeSiteUserInfo(SiteParserBase): self.username = user_info.get("username") self.user_level = user_info.get("class", {}).get("name") self.join_at = user_info.get("regTime", 0) - self.join_at = StringUtils.unify_datetime_str(str(self.join_at)) + self.join_at = time_tools.normalize_datetime(str(self.join_at)) self.upload = user_info.get("upload") self.download = user_info.get("download") diff --git a/app/modules/indexer/parser/torrent_leech.py b/app/modules/indexer/parser/torrent_leech.py index 06cc65fc4..57b5334af 100644 --- a/app/modules/indexer/parser/torrent_leech.py +++ b/app/modules/indexer/parser/torrent_leech.py @@ -5,7 +5,10 @@ from typing import Optional from lxml import etree from app.modules.indexer.parser import SiteParserBase, SiteSchema -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation import temporal as time_tools +from app.foundation import text as text_tools +from app.foundation.dom import DomUtils class TorrentLeechSiteUserInfo(SiteParserBase): @@ -26,7 +29,7 @@ class TorrentLeechSiteUserInfo(SiteParserBase): html = etree.HTML(html_text) current_userid = None try: - if StringUtils.is_valid_html_element(html): + if DomUtils.has_child_elements(html): profile_routes = html.xpath( '//span[contains(concat(" ", normalize-space(@class), " "), " centerTopBar ")]' '//*[@onclick]/@onclick' @@ -71,7 +74,7 @@ class TorrentLeechSiteUserInfo(SiteParserBase): html_text = self._prepare_html_text(html_text) html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return username_html = html.xpath('//div[contains(concat(" ", normalize-space(@class), " "), ' @@ -85,13 +88,13 @@ class TorrentLeechSiteUserInfo(SiteParserBase): upload_html = html.xpath('//div[contains(@class,"profile-uploaded")]//span/text()') if upload_html: - self.upload = StringUtils.num_filesize(upload_html[0]) + self.upload = size_tools.parse_size(upload_html[0]) download_html = html.xpath('//div[contains(@class,"profile-downloaded")]//span/text()') if download_html: - self.download = StringUtils.num_filesize(download_html[0]) + self.download = size_tools.parse_size(download_html[0]) ratio_html = html.xpath('//div[contains(@class,"profile-ratio")]//span/text()') if ratio_html: - self.ratio = StringUtils.str_float(ratio_html[0].replace('∞', '0')) + self.ratio = text_tools.parse_float(ratio_html[0].replace('∞', '0')) user_level_html = html.xpath('//table[contains(@class, "profileViewTable")]' '//tr/td[normalize-space()="Class"]/' @@ -103,11 +106,11 @@ class TorrentLeechSiteUserInfo(SiteParserBase): '//tr/td[normalize-space()="Registration date"]/' 'following-sibling::td[1]/text()') if join_at_html: - self.join_at = StringUtils.unify_datetime_str(join_at_html[0].strip()) + self.join_at = time_tools.normalize_datetime(join_at_html[0].strip()) bonus_html = html.xpath('//span[contains(@class, "total-TL-points")]/text()') if bonus_html: - self.bonus = StringUtils.str_float(bonus_html[0].strip()) + self.bonus = text_tools.parse_float(bonus_html[0].strip()) finally: if html is not None: del html @@ -129,7 +132,7 @@ class TorrentLeechSiteUserInfo(SiteParserBase): """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return None size_col = 2 @@ -144,8 +147,8 @@ class TorrentLeechSiteUserInfo(SiteParserBase): page_seeding = len(seeding_sizes) for i in range(0, len(seeding_sizes)): - size = StringUtils.num_filesize(seeding_sizes[i].xpath("string(.)").strip()) - seeders = StringUtils.str_int(seeding_seeders[i]) + size = size_tools.parse_size(seeding_sizes[i].xpath("string(.)").strip()) + seeders = text_tools.parse_int(seeding_seeders[i]) page_seeding_size += size page_seeding_info.append([seeders, size]) diff --git a/app/modules/indexer/parser/unit3d.py b/app/modules/indexer/parser/unit3d.py index 533446f9f..9104f5847 100644 --- a/app/modules/indexer/parser/unit3d.py +++ b/app/modules/indexer/parser/unit3d.py @@ -5,7 +5,10 @@ from typing import Optional from lxml import etree from app.modules.indexer.parser import SiteParserBase, SiteSchema -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation import temporal as time_tools +from app.foundation import text as text_tools +from app.foundation.dom import DomUtils class Unit3dSiteUserInfo(SiteParserBase): @@ -28,7 +31,7 @@ class Unit3dSiteUserInfo(SiteParserBase): bonus_text = tmps[0].xpath("string(.)") bonus_match = re.search(r"([\d,.]+)", bonus_text) if bonus_match and bonus_match.group(1).strip(): - self.bonus = StringUtils.str_float(bonus_match.group(1)) + self.bonus = text_tools.parse_float(bonus_match.group(1)) finally: if html is not None: del html @@ -44,7 +47,7 @@ class Unit3dSiteUserInfo(SiteParserBase): """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return None # 用户等级 @@ -57,7 +60,7 @@ class Unit3dSiteUserInfo(SiteParserBase): 'or contains(text(), "註冊日期") ' 'or contains(text(), "Registration date")]/text()') if join_at_text: - self.join_at = StringUtils.unify_datetime_str( + self.join_at = time_tools.normalize_datetime( join_at_text[0].replace('注册日期', '').replace('註冊日期', '').replace('Registration date', '')) finally: if html is not None: @@ -72,7 +75,7 @@ class Unit3dSiteUserInfo(SiteParserBase): """ html = etree.HTML(html_text) try: - if not StringUtils.is_valid_html_element(html): + if not DomUtils.has_child_elements(html): return None size_col = 9 @@ -93,8 +96,8 @@ class Unit3dSiteUserInfo(SiteParserBase): page_seeding = len(seeding_sizes) for i in range(0, len(seeding_sizes)): - size = StringUtils.num_filesize(seeding_sizes[i].xpath("string(.)").strip()) - seeders = StringUtils.str_int(seeding_seeders[i].xpath("string(.)").strip()) + size = size_tools.parse_size(seeding_sizes[i].xpath("string(.)").strip()) + seeders = text_tools.parse_int(seeding_seeders[i].xpath("string(.)").strip()) page_seeding_size += size page_seeding_info.append([seeders, size]) @@ -120,12 +123,12 @@ class Unit3dSiteUserInfo(SiteParserBase): html_text = self._prepare_html_text(html_text) upload_match = re.search(r"[^总]上[传傳]量?[::_<>/a-zA-Z-=\"'\s#;]+([\d,.\s]+[KMGTPI]*B)", html_text, re.IGNORECASE) - self.upload = StringUtils.num_filesize(upload_match.group(1).strip()) if upload_match else 0 + self.upload = size_tools.parse_size(upload_match.group(1).strip()) if upload_match else 0 download_match = re.search(r"[^总子影力]下[载載]量?[::_<>/a-zA-Z-=\"'\s#;]+([\d,.\s]+[KMGTPI]*B)", html_text, re.IGNORECASE) - self.download = StringUtils.num_filesize(download_match.group(1).strip()) if download_match else 0 + self.download = size_tools.parse_size(download_match.group(1).strip()) if download_match else 0 ratio_match = re.search(r"分享率[::_<>/a-zA-Z-=\"'\s#;]+([\d,.\s]+)", html_text) - self.ratio = StringUtils.str_float(ratio_match.group(1)) if ( + self.ratio = text_tools.parse_float(ratio_match.group(1)) if ( ratio_match and ratio_match.group(1).strip()) else 0.0 def _parse_message_unread_links(self, html_text: str, msg_links: list) -> Optional[str]: diff --git a/app/modules/indexer/parser/yema.py b/app/modules/indexer/parser/yema.py index 99d8c65d1..2e7121399 100644 --- a/app/modules/indexer/parser/yema.py +++ b/app/modules/indexer/parser/yema.py @@ -4,7 +4,7 @@ from typing import Optional, Tuple from app.runtime.log import logger from app.modules.indexer.parser import SiteParserBase, SiteSchema -from app.domain.string import StringUtils +from app.foundation import temporal as time_tools class YemaSiteUserInfo(SiteParserBase): @@ -65,7 +65,7 @@ class YemaSiteUserInfo(SiteParserBase): self.username = user_info.get("name") self.user_level = str(user_info.get("level")) \ if user_info.get("level") is not None else None - self.join_at = StringUtils.unify_datetime_str(user_info.get("registerTime")) + self.join_at = time_tools.normalize_datetime(user_info.get("registerTime")) self.upload = int(user_info.get("promotionUploadSize") or 0) self.download = int(user_info.get("promotionDownloadSize") or 0) self.ratio = round(self.upload / (self.download or 1), 2) diff --git a/app/modules/indexer/parser/zhixing.py b/app/modules/indexer/parser/zhixing.py index dd8b9eff4..f6179897b 100644 --- a/app/modules/indexer/parser/zhixing.py +++ b/app/modules/indexer/parser/zhixing.py @@ -8,7 +8,7 @@ import re from typing import Optional, Tuple from app.modules.indexer.parser import SiteParserBase, SiteSchema -from app.domain.string import StringUtils +from app.foundation import temporal as time_tools from bs4 import BeautifulSoup from urllib.parse import urljoin @@ -68,7 +68,7 @@ class ZhixingSiteUserInfo(SiteParserBase): self.userid = info_dict.get('UID') self.username = info_dict.get('用户名') self.user_level = info_dict.get('用户组') - self.join_at = StringUtils.unify_datetime_str(info_dict.get('注册时间')) if '注册时间' in info_dict else None + self.join_at = time_tools.normalize_datetime(info_dict.get('注册时间')) if '注册时间' in info_dict else None def num_filesize_safe(s: str): if s: diff --git a/app/modules/indexer/spider/__init__.py b/app/modules/indexer/spider/__init__.py index 3932075c1..24f768812 100644 --- a/app/modules/indexer/spider/__init__.py +++ b/app/modules/indexer/spider/__init__.py @@ -14,7 +14,9 @@ from app.runtime.log import logger from app.schemas.types import MediaType from app.adapters.system import rust as rust_accel from app.adapters.network.http import RequestUtils, AsyncRequestUtils -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation import temporal as time_tools +from app.foundation import url as url_tools from app.foundation.url import UrlUtils @@ -479,7 +481,7 @@ class SiteSpider: if download_link: if not download_link.startswith("http") \ and not download_link.startswith("magnet"): - _scheme, _domain = StringUtils.get_url_netloc(self.domain) + _scheme, _domain = url_tools.split_netloc(self.domain) if _domain in download_link: if download_link.startswith("/"): self.torrents_info['enclosure'] = f"{_scheme}:{download_link}" @@ -535,7 +537,7 @@ class SiteSpider: size_val = item.replace("\n", "").strip() size_val = self.__filter_text(size_val, selector.get('filters')) - self.torrents_info['size'] = StringUtils.num_filesize(size_val) + self.torrents_info['size'] = size_tools.parse_size(size_val) else: self.torrents_info['size'] = 0 @@ -600,7 +602,7 @@ class SiteSpider: else: datetime.datetime.strptime(str(self.torrents_info['pubdate']), '%Y-%m-%d %H:%M:%S') except (ValueError, TypeError): - self.torrents_info['pubdate'] = StringUtils.unify_datetime_str(str(self.torrents_info['pubdate'])) + self.torrents_info['pubdate'] = time_tools.normalize_datetime(str(self.torrents_info['pubdate'])) if self.__is_invalid_pubdate_text(self.torrents_info.get('pubdate')): self.torrents_info.pop('pubdate', None) diff --git a/app/modules/indexer/spider/haidan.py b/app/modules/indexer/spider/haidan.py index 13b37db51..f7ff1cd96 100644 --- a/app/modules/indexer/spider/haidan.py +++ b/app/modules/indexer/spider/haidan.py @@ -6,7 +6,8 @@ from app.db.oper.systemconfig import SystemConfigOper from app.runtime.log import logger from app.schemas import MediaType from app.adapters.network.http import RequestUtils, AsyncRequestUtils -from app.domain.string import StringUtils +from app.domain import site as site_rules +from app.foundation import temporal as time_tools class HaiDanSpider: @@ -63,7 +64,7 @@ class HaiDanSpider: if indexer: self._indexerid = indexer.get('id') self._url = indexer.get('domain') - self._domain = StringUtils.get_url_domain(self._url) + self._domain = site_rules.extract_domain(self._url) self._searchurl = self._searchurl % self._url self._name = indexer.get('name') if indexer.get('proxy'): @@ -132,7 +133,7 @@ class HaiDanSpider: 'title': item.get('name'), 'description': item.get('small_descr'), 'enclosure': item.get('url'), - 'pubdate': StringUtils.format_timestamp(item.get('added')), + 'pubdate': time_tools.format_timestamp(item.get('added')), 'size': int(item.get('size') or '0'), 'seeders': int(item.get('seeders') or '0'), 'peers': int(item.get("leechers") or '0'), diff --git a/app/modules/indexer/spider/hddolby.py b/app/modules/indexer/spider/hddolby.py index 1031a8dfe..5a9dfd2d7 100644 --- a/app/modules/indexer/spider/hddolby.py +++ b/app/modules/indexer/spider/hddolby.py @@ -5,7 +5,7 @@ from app.db.oper.systemconfig import SystemConfigOper from app.runtime.log import logger from app.schemas import MediaType from app.adapters.network.http import RequestUtils, AsyncRequestUtils -from app.domain.string import StringUtils +from app.domain import site as site_rules class HddolbySpider: @@ -71,7 +71,7 @@ class HddolbySpider: if indexer: self._indexerid = indexer.get('id') self._domain = indexer.get('domain') - self._domain_host = StringUtils.get_url_domain(self._domain) + self._domain_host = site_rules.extract_domain(self._domain) self._name = indexer.get('name') if indexer.get('proxy'): self._proxy = settings.PROXY diff --git a/app/modules/indexer/spider/mtorrent.py b/app/modules/indexer/spider/mtorrent.py index 50b0976b8..4447eaa8b 100644 --- a/app/modules/indexer/spider/mtorrent.py +++ b/app/modules/indexer/spider/mtorrent.py @@ -9,7 +9,8 @@ from app.db.oper.systemconfig import SystemConfigOper from app.runtime.log import logger from app.schemas import MediaType from app.adapters.network.http import RequestUtils, AsyncRequestUtils -from app.domain.string import StringUtils +from app.domain import site as site_rules +from app.foundation import temporal as time_tools class MTorrentSpider: @@ -68,7 +69,7 @@ class MTorrentSpider: if indexer: self._indexerid = indexer.get('id') self._url = indexer.get('domain') - self._domain = StringUtils.get_url_domain(self._url) + self._domain = site_rules.extract_domain(self._url) self._searchurl = self._searchurl % self._domain self._name = indexer.get('name') if indexer.get('proxy'): @@ -137,7 +138,7 @@ class MTorrentSpider: 'title': result.get('name'), 'description': result.get('smallDescr'), 'enclosure': self.__get_download_url(result.get('id')), - 'pubdate': StringUtils.format_timestamp(result.get('createdDate')), + 'pubdate': time_tools.format_timestamp(result.get('createdDate')), 'size': int(result.get('size') or '0'), 'seeders': int(status.get("seeders") or '0'), 'peers': int(status.get("leechers") or '0'), @@ -150,18 +151,18 @@ class MTorrentSpider: 'category': category } if discount_end_time := status.get('discountEndTime'): - torrent['freedate'] = StringUtils.format_timestamp(discount_end_time) + torrent['freedate'] = time_tools.format_timestamp(discount_end_time) # 解析全站促销时的规则(当前馒头只有下载促销) if promotion_rule := status.get("promotionRule"): discount = promotion_rule.get("discount", "NORMAL") torrent["downloadvolumefactor"] = self.__get_downloadvolumefactor(discount) if end_time := promotion_rule.get("endTime"): - torrent["freedate"] = StringUtils.format_timestamp(end_time) + torrent["freedate"] = time_tools.format_timestamp(end_time) if mall_single_free := status.get("mallSingleFree"): if mall_single_free.get("status") == "ONGOING": torrent["downloadvolumefactor"] = self.__get_downloadvolumefactor("FREE") if end_date := mall_single_free.get("endDate"): - torrent["freedate"] = StringUtils.format_timestamp(end_date) + torrent["freedate"] = time_tools.format_timestamp(end_date) torrents.append(torrent) return torrents diff --git a/app/modules/indexer/spider/rousi.py b/app/modules/indexer/spider/rousi.py index cdbdfab87..7204c2636 100644 --- a/app/modules/indexer/spider/rousi.py +++ b/app/modules/indexer/spider/rousi.py @@ -7,7 +7,8 @@ from app.db.oper.systemconfig import SystemConfigOper from app.runtime.log import logger from app.schemas import MediaType from app.adapters.network.http import RequestUtils, AsyncRequestUtils -from app.domain.string import StringUtils +from app.domain import site as site_rules +from app.foundation import temporal as time_tools class RousiSpider: @@ -52,7 +53,7 @@ class RousiSpider: if indexer: self._indexerid = indexer.get('id') self._url = indexer.get('domain') - self._domain = StringUtils.get_url_domain(self._url) + self._domain = site_rules.extract_domain(self._url) self._searchurl = self._searchurl % self._domain self._downloadurl = self._downloadurl % (self._domain, "%s") self._name = indexer.get('name') @@ -199,13 +200,13 @@ class RousiSpider: uploadvolumefactor = float(promotion.get('up_multiplier', 1.0)) # 促销到期时间,格式化为 YYYY-MM-DD HH:MM:SS if promotion.get('until'): - freedate = StringUtils.unify_datetime_str(promotion.get('until')) + freedate = time_tools.normalize_datetime(promotion.get('until')) torrent = { 'title': result.get('title'), 'description': result.get('subtitle'), 'enclosure': self.__get_download_url(result.get('id')), - 'pubdate': StringUtils.unify_datetime_str(result.get('created_at')), + 'pubdate': time_tools.normalize_datetime(result.get('created_at')), 'size': int(result.get('size') or 0), 'seeders': int(result.get('seeders') or 0), 'peers': int(result.get('leechers') or 0), diff --git a/app/modules/indexer/spider/sunnypt.py b/app/modules/indexer/spider/sunnypt.py index e033a5e69..3c1d750d7 100644 --- a/app/modules/indexer/spider/sunnypt.py +++ b/app/modules/indexer/spider/sunnypt.py @@ -7,7 +7,7 @@ from app.runtime.config import settings from app.runtime.log import logger from app.schemas import MediaType from app.adapters.network.http import AsyncRequestUtils, RequestUtils -from app.domain.string import StringUtils +from app.foundation import temporal as time_tools class SunnyPTSpider: @@ -272,14 +272,14 @@ class SunnyPTSpider: if promotion_active else 1.0 upload_factor = float(promotion.get("up_multiplier", 1.0)) \ if promotion_active else 1.0 - freedate = StringUtils.unify_datetime_str(promotion.get("until")) \ + freedate = time_tools.normalize_datetime(promotion.get("until")) \ if promotion_active and promotion.get("until") else None torrent_id = result.get("id") torrents.append({ "title": result.get("title"), "description": result.get("subtitle"), "enclosure": self._build_download_url(torrent_id), - "pubdate": StringUtils.unify_datetime_str(result.get("created_at")), + "pubdate": time_tools.normalize_datetime(result.get("created_at")), "size": int(result.get("size") or 0), "seeders": int(result.get("seeders") or 0), "peers": int(result.get("leechers") or 0), diff --git a/app/modules/indexer/spider/tnode.py b/app/modules/indexer/spider/tnode.py index fb47451ea..19bbcd6c0 100644 --- a/app/modules/indexer/spider/tnode.py +++ b/app/modules/indexer/spider/tnode.py @@ -6,7 +6,7 @@ from app.runtime.config import settings from app.runtime.log import logger from app.adapters.network.http import RequestUtils, AsyncRequestUtils from app.foundation.singleton import SingletonClass -from app.domain.string import StringUtils +from app.foundation import temporal as time_tools class TNodeSpider(metaclass=SingletonClass): @@ -98,7 +98,7 @@ class TNodeSpider(metaclass=SingletonClass): 'title': result.get('title'), 'description': result.get('subtitle'), 'enclosure': self._downloadurl % (self._domain, result.get('id')), - 'pubdate': StringUtils.format_timestamp(result.get('upload_time')), + 'pubdate': time_tools.format_timestamp(result.get('upload_time')), 'size': result.get('size'), 'seeders': result.get('seeding'), 'peers': result.get('leeching'), diff --git a/app/modules/indexer/spider/torrentleech.py b/app/modules/indexer/spider/torrentleech.py index 2e0b96bfe..142daf80b 100644 --- a/app/modules/indexer/spider/torrentleech.py +++ b/app/modules/indexer/spider/torrentleech.py @@ -5,7 +5,8 @@ from app.runtime.config import settings from app.runtime.log import logger from app.schemas import MediaType from app.adapters.network.http import RequestUtils, AsyncRequestUtils -from app.domain.string import StringUtils +from app.foundation import temporal as time_tools +from app.foundation import text as text_tools class TorrentLeech: @@ -76,7 +77,7 @@ class TorrentLeech: 'enclosure': self._downloadurl % (self._indexer.get('domain'), result.get('fid'), result.get('filename')), - 'pubdate': StringUtils.format_timestamp(result.get('addedTimestamp')), + 'pubdate': time_tools.format_timestamp(result.get('addedTimestamp')), 'size': result.get('size'), 'seeders': result.get('seeders'), 'peers': result.get('leechers'), @@ -104,7 +105,7 @@ class TorrentLeech: """ 搜索种子 """ - if StringUtils.is_chinese(keyword): + if text_tools.contains_chinese(keyword): # 不支持中文 return True, [] @@ -141,7 +142,7 @@ class TorrentLeech: """ 异步搜索种子 """ - if StringUtils.is_chinese(keyword): + if text_tools.contains_chinese(keyword): # 不支持中文 return True, [] diff --git a/app/modules/indexer/spider/yema.py b/app/modules/indexer/spider/yema.py index 7df11ab10..2cc360668 100644 --- a/app/modules/indexer/spider/yema.py +++ b/app/modules/indexer/spider/yema.py @@ -6,7 +6,7 @@ from app.runtime.config import settings from app.runtime.log import logger from app.schemas import MediaType from app.adapters.network.http import AsyncRequestUtils, RequestUtils -from app.domain.string import StringUtils +from app.foundation import temporal as time_tools class YemaSpider: @@ -162,14 +162,14 @@ class YemaSpider: "title": result.get("showName"), "description": result.get("shortDesc"), "enclosure": self._build_download_url(torrent_id), - "pubdate": StringUtils.unify_datetime_str(result.get("listingTime")), + "pubdate": time_tools.normalize_datetime(result.get("listingTime")), "size": result.get("fileSize"), "seeders": result.get("seedNum"), "peers": result.get("leechNum"), "grabs": result.get("completedNum"), "downloadvolumefactor": self._download_factor(result.get("downloadPromotion")), "uploadvolumefactor": self._upload_factor(result.get("uploadPromotion")), - "freedate": StringUtils.unify_datetime_str(result.get("downloadPromotionEndTime")), + "freedate": time_tools.normalize_datetime(result.get("downloadPromotionEndTime")), "page_url": f"{self._site_url}/#/torrent/detail/{torrent_id}/", "labels": labels, "hit_and_run": bool(result.get("hrPunishEnable")), diff --git a/app/modules/qbittorrent/__init__.py b/app/modules/qbittorrent/__init__.py index e1298ee6d..ee5425df4 100644 --- a/app/modules/qbittorrent/__init__.py +++ b/app/modules/qbittorrent/__init__.py @@ -19,7 +19,10 @@ from app.schemas.types import ( TorrentQueryStatus, TorrentStatus, ) -from app.domain.string import StringUtils +from app.domain import torrent as torrent_rules +from app.foundation import size as size_tools +from app.foundation import temporal as time_tools +from app.foundation import text as text_tools _QBITTORRENT_DOWNLOADING_STATES = { "allocating", @@ -147,7 +150,7 @@ class QbittorrentModule(_ModuleBase, _DownloaderBase[Qbittorrent]): if torrent_content: # 检查是否为磁力链接 - if StringUtils.is_magnet_link(torrent_content): + if torrent_rules.is_magnet_link(torrent_content): return None, torrent_content else: torrent_info = Torrent.from_string(torrent_content) @@ -175,7 +178,7 @@ class QbittorrentModule(_ModuleBase, _DownloaderBase[Qbittorrent]): return None # 生成随机Tag - tag = StringUtils.generate_random_str(10) + tag = text_tools.random_string(10) if label: tags = label.split(',') + [tag] elif settings.TORRENT_TAG: @@ -341,9 +344,9 @@ class QbittorrentModule(_ModuleBase, _DownloaderBase[Qbittorrent]): seeding_time_limit=torrent_data.get('seeding_time_limit'), progress=(torrent_data.get('progress') or 0) * 100, state=self.__normalize_torrent_state(torrent_data.get('state')), - dlspeed=StringUtils.str_filesize(dlspeed), - upspeed=StringUtils.str_filesize(torrent_data.get('upspeed')), - left_time=StringUtils.str_secends( + dlspeed=size_tools.format_compact_size(dlspeed), + upspeed=size_tools.format_compact_size(torrent_data.get('upspeed')), + left_time=time_tools.format_duration( (total_size - completed_size) / dlspeed ) if dlspeed > 0 else '', ) diff --git a/app/modules/qbittorrent/qbittorrent.py b/app/modules/qbittorrent/qbittorrent.py index db745d28a..466187ed4 100644 --- a/app/modules/qbittorrent/qbittorrent.py +++ b/app/modules/qbittorrent/qbittorrent.py @@ -11,7 +11,8 @@ from qbittorrentapi.client import Client from qbittorrentapi.transfer import TransferInfoDictionary from app.runtime.log import logger -from app.domain.string import StringUtils +from app.domain import torrent as torrent_rules +from app.foundation import url as url_tools class Qbittorrent: @@ -32,7 +33,7 @@ class Qbittorrent: if host and port: self._host, self._port = host, port elif host: - self._host, self._port = StringUtils.get_domain_address(address=host, prefix=True) + self._host, self._port = url_tools.parse_address(address=host, include_scheme=True) else: logger.error("Qbittorrent配置不完整!") return @@ -453,7 +454,7 @@ class Qbittorrent: category = None try: cookie_to_use = cookie - if urls and cookie and not StringUtils.is_magnet_link(urls): + if urls and cookie and not torrent_rules.is_magnet_link(urls): if self.__sync_download_cookies(url=urls, cookie_header=cookie): cookie_to_use = None diff --git a/app/modules/qqbot/qqbot.py b/app/modules/qqbot/qqbot.py index 0e38a67f9..63b3c4727 100644 --- a/app/modules/qqbot/qqbot.py +++ b/app/modules/qqbot/qqbot.py @@ -24,7 +24,7 @@ from app.modules.qqbot.api import ( ) from app.modules.qqbot.gateway import run_gateway from app.adapters.network.http import RequestUtils -from app.domain.string import StringUtils +from app.foundation import size as size_tools # QQ Markdown 图片展示尺寸限制,避免竖版海报被客户端拉伸变形 _DEFAULT_IMAGE_SIZE: Tuple[int, int] = (208, 320) @@ -449,7 +449,7 @@ class QQBot: meta = MetaInfo(t.title, t.description) name = f"{meta.season_episode} {meta.resource_term} {meta.video_term}" name = " ".join(name.split()) - lines.append(f"{i + 1}.【{t.site_name}】{name} {StringUtils.str_filesize(t.size)} {t.seeders}↑") + lines.append(f"{i + 1}.【{t.site_name}】{name} {size_tools.format_compact_size(t.size)} {t.seeders}↑") text = "\n".join(lines) return self.send_msg( title=title or "种子列表", diff --git a/app/modules/rtorrent/__init__.py b/app/modules/rtorrent/__init__.py index d142bf612..8dc71388e 100644 --- a/app/modules/rtorrent/__init__.py +++ b/app/modules/rtorrent/__init__.py @@ -18,7 +18,10 @@ from app.schemas.types import ( TorrentQueryStatus, TorrentStatus, ) -from app.domain.string import StringUtils +from app.domain import torrent as torrent_rules +from app.foundation import size as size_tools +from app.foundation import temporal as time_tools +from app.foundation import text as text_tools class RtorrentModule(_ModuleBase, _DownloaderBase[Rtorrent]): @@ -122,7 +125,7 @@ class RtorrentModule(_ModuleBase, _DownloaderBase[Rtorrent]): torrent_content = content if torrent_content: - if StringUtils.is_magnet_link(torrent_content): + if torrent_rules.is_magnet_link(torrent_content): return None, torrent_content else: torrent_info = Torrent.from_string(torrent_content) @@ -153,7 +156,7 @@ class RtorrentModule(_ModuleBase, _DownloaderBase[Rtorrent]): return None # 生成随机Tag - tag = StringUtils.generate_random_str(10) + tag = text_tools.random_string(10) if label: tags = label.split(",") + [tag] elif settings.TORRENT_TAG: @@ -347,10 +350,10 @@ class RtorrentModule(_ModuleBase, _DownloaderBase[Rtorrent]): state=self.__normalize_torrent_state( torrent_data.get("state"), torrent_data.get("complete") ), - dlspeed=StringUtils.str_filesize(dlspeed), - upspeed=StringUtils.str_filesize(upspeed), + dlspeed=size_tools.format_compact_size(dlspeed), + upspeed=size_tools.format_compact_size(upspeed), tags=torrent_data.get("tags"), - left_time=StringUtils.str_secends((total_size - completed_size) / dlspeed) + left_time=time_tools.format_duration((total_size - completed_size) / dlspeed) if dlspeed > 0 else "", ) diff --git a/app/modules/slack/slack.py b/app/modules/slack/slack.py index cba8a30aa..e537a1bb4 100644 --- a/app/modules/slack/slack.py +++ b/app/modules/slack/slack.py @@ -15,7 +15,7 @@ from app.domain.context import MediaInfo, Context from app.domain.metainfo import MetaInfo from app.runtime.log import logger from app.adapters.network.http import RequestUtils -from app.domain.string import StringUtils +from app.foundation import size as size_tools lock = Lock() @@ -692,7 +692,7 @@ class Slack: seeder = f"{torrent.seeders}↑" description = torrent.description text = f"{index}. 【{site_name}】<{link}|{title_text}> " \ - f"{StringUtils.str_filesize(torrent.size)} {free} {seeder}\n" \ + f"{size_tools.format_compact_size(torrent.size)} {free} {seeder}\n" \ f"{description}" blocks.append( { @@ -752,7 +752,7 @@ class Slack: seeder = f"{torrent.seeders}↑" description = torrent.description text = f"{index}. 【{site_name}】<{link}|{title_text}> " \ - f"{StringUtils.str_filesize(torrent.size)} {free} {seeder}\n" \ + f"{size_tools.format_compact_size(torrent.size)} {free} {seeder}\n" \ f"{description}" blocks.append( { diff --git a/app/modules/synologychat/synologychat.py b/app/modules/synologychat/synologychat.py index 299ecdd5d..4c9eca947 100644 --- a/app/modules/synologychat/synologychat.py +++ b/app/modules/synologychat/synologychat.py @@ -8,7 +8,8 @@ from app.domain.context import MediaInfo, Context from app.domain.metainfo import MetaInfo from app.runtime.log import logger from app.adapters.network.http import RequestUtils -from app.domain.string import StringUtils +from app.foundation import size as size_tools +from app.foundation import url as url_tools lock = Lock() @@ -22,7 +23,7 @@ class SynologyChat: self._webhook_url = SYNOLOGYCHAT_WEBHOOK self._token = SYNOLOGYCHAT_TOKEN if self._webhook_url: - self._domain = StringUtils.get_base_url(self._webhook_url) + self._domain = url_tools.base_url(self._webhook_url) def check_token(self, token: str) -> bool: return True if token == self._token else False @@ -161,7 +162,7 @@ class SynologyChat: seeder = f"{torrent.seeders}↑" description = torrent.description caption = f"{caption}\n{index}.【{site_name}】<{link}|{title}> " \ - f"{StringUtils.str_filesize(torrent.size)} {free} {seeder}\n" \ + f"{size_tools.format_compact_size(torrent.size)} {free} {seeder}\n" \ f"_{description}_" index += 1 diff --git a/app/modules/telegram/telegram.py b/app/modules/telegram/telegram.py index 1f45f5ae0..a1ab9a5ed 100644 --- a/app/modules/telegram/telegram.py +++ b/app/modules/telegram/telegram.py @@ -42,7 +42,7 @@ from app.runtime.thread import ThreadHelper # noqa: E402 from app.runtime.log import logger # noqa: E402 from app.runtime.execution import retry # noqa: E402 from app.adapters.network.http import RequestUtils # noqa: E402 -from app.domain.string import StringUtils # noqa: E402 +from app.foundation import size as size_tools # noqa: E402 TELEGRAM_PARSE_MODE_MARKDOWN = "MarkdownV2" @@ -1036,7 +1036,7 @@ class Telegram: title_link = self._format_link(title, link, parse_mode) caption = ( f"{caption}\n{index}.【{site_name}】{title_link} " - f"{StringUtils.str_filesize(torrent.size)} {free} {seeder}" + f"{size_tools.format_compact_size(torrent.size)} {free} {seeder}" ) index += 1 diff --git a/app/modules/themoviedb/tmdbapi.py b/app/modules/themoviedb/tmdbapi.py index 1d6cae598..859fda704 100644 --- a/app/modules/themoviedb/tmdbapi.py +++ b/app/modules/themoviedb/tmdbapi.py @@ -5,7 +5,7 @@ from typing import Optional, List from app.runtime.config import settings from app.runtime.log import logger from app.schemas.types import MediaType -from app.domain.string import StringUtils +from app.foundation import text as text_tools from app.foundation.text import convert as zhconv_convert from .tmdbv3api import TMDb, Search, Movie, TV, Season, Episode, Discover, Trending, Person, Collection from .tmdbv3api.exceptions import TMDbException, TMDbConnectionError @@ -124,9 +124,9 @@ class TmdbApi: return False if not isinstance(tmdb_names, list): tmdb_names = [tmdb_names] - file_name = StringUtils.clear(file_name).upper() + file_name = text_tools.remove_punctuation(file_name).upper() for tmdb_name in tmdb_names: - tmdb_name = StringUtils.clear(tmdb_name).strip().upper() + tmdb_name = text_tools.remove_punctuation(tmdb_name).strip().upper() if file_name == tmdb_name: return True return False @@ -727,7 +727,7 @@ class TmdbApi: iso_3166_1 = alternative_title.get("iso_3166_1") if iso_3166_1 == "CN": title = alternative_title.get("title") - if title and StringUtils.is_chinese(title) \ + if title and text_tools.contains_chinese(title) \ and zhconv_convert(title, "zh-hans") == title: return title return tmdbinfo.get("title") if tmdbinfo.get("media_type") == MediaType.MOVIE else tmdbinfo.get("name") @@ -737,7 +737,7 @@ class TmdbApi: if tmdb_info.get("media_type") == MediaType.MOVIE \ else tmdb_info.get("name") # 查找中文名 - if not StringUtils.is_chinese(org_title): + if not text_tools.contains_chinese(org_title): cn_title = __get_tmdb_chinese_title(tmdb_info) if cn_title and cn_title != org_title: # 使用中文别名 @@ -748,7 +748,7 @@ class TmdbApi: else: # 使用新加坡名 sg_title = tmdb_info.get("sg_title") - if sg_title and sg_title != org_title and StringUtils.is_chinese(sg_title): + if sg_title and sg_title != org_title and text_tools.contains_chinese(sg_title): if tmdb_info.get("media_type") == MediaType.MOVIE: tmdb_info['title'] = sg_title else: diff --git a/app/modules/transmission/__init__.py b/app/modules/transmission/__init__.py index 504bbbbe0..0b2a72b85 100644 --- a/app/modules/transmission/__init__.py +++ b/app/modules/transmission/__init__.py @@ -19,7 +19,9 @@ from app.schemas.types import ( TorrentQueryStatus, TorrentStatus, ) -from app.domain.string import StringUtils +from app.domain import torrent as torrent_rules +from app.foundation import size as size_tools +from app.foundation import temporal as time_tools _TRANSMISSION_DOWNLOADING_STATES = { "download_pending", @@ -125,7 +127,7 @@ class TransmissionModule(_ModuleBase, _DownloaderBase[Transmission]): if torrent_content: # 检查是否为磁力链接 - if StringUtils.is_magnet_link(torrent_content): + if torrent_rules.is_magnet_link(torrent_content): return None, torrent_content else: torrent_info = Torrent.from_string(torrent_content) @@ -329,14 +331,14 @@ class TransmissionModule(_ModuleBase, _DownloaderBase[Transmission]): progress=__get_torrent_progress(torrent_data), size=__get_torrent_size(torrent_data), state=self.__normalize_torrent_state(torrent_data.status), - dlspeed=StringUtils.str_filesize(dlspeed), - upspeed=StringUtils.str_filesize(upspeed), + dlspeed=size_tools.format_compact_size(dlspeed), + upspeed=size_tools.format_compact_size(upspeed), tags=__get_torrent_labels(torrent_data), download_limit=__get_torrent_attr(torrent_data, "download_limit", "downloadLimit"), upload_limit=__get_torrent_attr(torrent_data, "upload_limit", "uploadLimit"), ratio_limit=ratio_limit, seeding_time_limit=seeding_time_limit, - left_time=StringUtils.str_secends( + left_time=time_tools.format_duration( left_until_done / dlspeed ) if dlspeed > 0 else '' ) diff --git a/app/modules/vocechat/vocechat.py b/app/modules/vocechat/vocechat.py index 905817f8d..3c9a2f30e 100644 --- a/app/modules/vocechat/vocechat.py +++ b/app/modules/vocechat/vocechat.py @@ -9,7 +9,7 @@ from app.domain.metainfo import MetaInfo from app.runtime.log import logger from app.runtime.execution import retry from app.adapters.network.http import RequestUtils -from app.domain.string import StringUtils +from app.foundation import size as size_tools lock = threading.Lock() @@ -215,7 +215,7 @@ class VoceChat: free = torrent.volume_factor seeder = f"{torrent.seeders}↑" caption = f"{caption}\n{index}.【{site_name}】[{title}]({link}) " \ - f"{StringUtils.str_filesize(torrent.size)} {free} {seeder}" + f"{size_tools.format_compact_size(torrent.size)} {free} {seeder}" index += 1 if link: diff --git a/app/modules/wechat/wechat.py b/app/modules/wechat/wechat.py index 12cb281a2..44146960c 100644 --- a/app/modules/wechat/wechat.py +++ b/app/modules/wechat/wechat.py @@ -12,7 +12,7 @@ from app.domain.metainfo import MetaInfo from app.runtime.log import logger from app.runtime.execution import retry from app.adapters.network.http import RequestUtils -from app.domain.string import StringUtils +from app.foundation import size as size_tools from app.foundation.url import UrlUtils lock = threading.Lock() @@ -563,7 +563,7 @@ class WeChat: f"{meta.resource_term} " \ f"{meta.video_term} " \ f"{meta.release_group} " \ - f"{StringUtils.str_filesize(torrent.size)} " \ + f"{size_tools.format_compact_size(torrent.size)} " \ f"{torrent.volume_factor} " \ f"{torrent.seeders}↑" torrent_title = re.sub(r"\s+", " ", torrent_title).strip() diff --git a/app/modules/wechat/wechatbot.py b/app/modules/wechat/wechatbot.py index 5f0b87c4e..bef6b144c 100644 --- a/app/modules/wechat/wechatbot.py +++ b/app/modules/wechat/wechatbot.py @@ -20,7 +20,7 @@ from app.runtime.log import logger from app.schemas import CommingMessage from app.schemas.types import MessageChannel from app.adapters.network.http import RequestUtils -from app.domain.string import StringUtils +from app.foundation import size as size_tools class WeChatBot: @@ -629,7 +629,7 @@ class WeChatBot: f"{meta.resource_term} " f"{meta.video_term} " f"{meta.release_group} " - f"{StringUtils.str_filesize(torrent.size)} " + f"{size_tools.format_compact_size(torrent.size)} " f"{torrent.volume_factor} " f"{torrent.seeders}↑" ) diff --git a/app/modules/wechatclawbot/wechatclawbot.py b/app/modules/wechatclawbot/wechatclawbot.py index f41b55d37..efafe34a3 100644 --- a/app/modules/wechatclawbot/wechatclawbot.py +++ b/app/modules/wechatclawbot/wechatclawbot.py @@ -22,7 +22,7 @@ from app.domain.context import Context, MediaInfo from app.domain.metainfo import MetaInfo from app.runtime.log import logger from app.adapters.network.http import RequestUtils -from app.domain.string import StringUtils +from app.foundation import size as size_tools @dataclass @@ -2204,7 +2204,7 @@ class WechatClawBot: meta = MetaInfo(title=torrent.title, subtitle=torrent.description) text = ( f"{index}.【{torrent.site_name}】{meta.season_episode} {meta.resource_term} " - f"{meta.video_term} {meta.release_group} {StringUtils.str_filesize(torrent.size)} " + f"{meta.video_term} {meta.release_group} {size_tools.format_compact_size(torrent.size)} " f"{torrent.volume_factor} {torrent.seeders}↑" ) text = re.sub(r"\s+", " ", text).strip() diff --git a/app/runtime/compat/manifest.py b/app/runtime/compat/manifest.py index b79fe2804..758c0bd58 100644 --- a/app/runtime/compat/manifest.py +++ b/app/runtime/compat/manifest.py @@ -30,6 +30,12 @@ MODULE_ALIASES: Dict[str, ModuleAlias] = { introduced="v3.0.0", owner="sdk", ), + "app.domain.string": ModuleAlias( + target="app.sdk.string", + replacement="app.sdk.utilities", + introduced="v3.0.0", + owner="sdk", + ), "app.db.agentchat_oper": ModuleAlias( target="app.db.oper.agentchat", replacement="app.db.oper.agentchat", @@ -283,10 +289,10 @@ MODULE_ALIASES: Dict[str, ModuleAlias] = { owner="domain", ), "app.utils.string": ModuleAlias( - target="app.domain.string", + target="app.sdk.string", replacement="app.sdk.utilities", introduced="v3.0.0", - owner="domain", + owner="sdk", ), "app.utils.url": ModuleAlias( target="app.foundation.url", diff --git a/app/sdk/string.py b/app/sdk/string.py new file mode 100644 index 000000000..cc8c1f623 --- /dev/null +++ b/app/sdk/string.py @@ -0,0 +1,165 @@ +"""为插件保留历史 StringUtils 类的轻量兼容门面。""" + +from functools import wraps +from inspect import signature +from typing import Callable + +from app.domain.episode import compact_numbers, format_ranges +from app.domain.site import extract_domain, urls_match +from app.domain.title import is_media_title_like, parse_search_keyword +from app.domain.torrent import is_magnet_link +from app.foundation.crypto import HashUtils +from app.foundation.dom import DomUtils +from app.foundation.size import format_compact_size, format_size, parse_size +from app.foundation.temporal import ( + format_approx_duration, + format_duration, + format_minutes, + format_remaining, + format_timestamp, + normalize_datetime, + parse_datetime, + parse_timestamp, +) +from app.foundation.text import ( + common_prefix, + contains_chinese, + contains_japanese, + contains_korean, + cookiejar_to_string, + count_words, + escape_markdown, + extract_named_ids, + format_amount, + is_all_chinese, + is_english_word, + is_number, + natural_sort_key, + normalize_upper, + parse_bool, + parse_float, + parse_int, + random_string, + remove_punctuation, + sanitize_filename, + split_by_bytes, + strip_optional, + title_case, +) +from app.foundation.url import ( + base_url, + host_label, + is_link, + parse_address, + second_level_label, + split_netloc, +) +from app.foundation.version import compare_version + + +def _legacy_alias(function: Callable, **keyword_aliases: str) -> Callable: + """创建支持旧关键字名称的静态方法转发器。""" + + @wraps(function) + def call(*args, **kwargs): + """把旧关键字转换为 canonical 参数后调用真实实现。""" + for legacy_name, canonical_name in keyword_aliases.items(): + if legacy_name in kwargs: + kwargs[canonical_name] = kwargs.pop(legacy_name) + return function(*args, **kwargs) + + canonical_to_legacy = { + canonical_name: legacy_name + for legacy_name, canonical_name in keyword_aliases.items() + } + call.__signature__ = signature(function).replace( + parameters=[ + parameter.replace( + name=canonical_to_legacy.get(parameter.name, parameter.name) + ) + for parameter in signature(function).parameters.values() + ] + ) + return call + + +def _legacy_md5_hash(data) -> str: + """保持 StringUtils.md5_hash 对空值和对象文本化的历史语义。""" + if not data: + return "" + return HashUtils.md5(str(data)) + + +class StringUtils: + """组合已拆分实现,保持插件使用的历史静态方法接口。""" + + num_filesize = staticmethod(parse_size) + str_timelong = staticmethod(_legacy_alias(format_approx_duration, time_sec="seconds")) + str_secends = staticmethod(_legacy_alias(format_duration, time_sec="seconds")) + is_chinese = staticmethod(_legacy_alias(contains_chinese, word="value")) + is_japanese = staticmethod(_legacy_alias(contains_japanese, word="value")) + is_korean = staticmethod(_legacy_alias(contains_korean, word="value")) + is_all_chinese = staticmethod(_legacy_alias(is_all_chinese, word="value")) + is_english_word = staticmethod(_legacy_alias(is_english_word, word="value")) + str_int = staticmethod(_legacy_alias(parse_int, text="value")) + str_float = staticmethod(_legacy_alias(parse_float, text="value")) + clear = staticmethod( + _legacy_alias( + remove_punctuation, + text="value", + replace_word="replacement", + ) + ) + clear_upper = staticmethod(_legacy_alias(normalize_upper, text="value")) + str_filesize = staticmethod(_legacy_alias(format_compact_size, pre="precision")) + format_size = staticmethod(format_size) + url_equal = staticmethod(_legacy_alias(urls_match, url1="first", url2="second")) + get_url_netloc = staticmethod(split_netloc) + get_url_domain = staticmethod(extract_domain) + get_url_sld = staticmethod(second_level_label) + get_url_host = staticmethod(host_label) + get_base_url = staticmethod(base_url) + clear_file_name = staticmethod(_legacy_alias(sanitize_filename, name="value")) + generate_random_str = staticmethod(_legacy_alias(random_string, randomlength="length")) + get_time = staticmethod(_legacy_alias(parse_datetime, date="value")) + unify_datetime_str = staticmethod(_legacy_alias(normalize_datetime, datetime_str="value")) + format_timestamp = staticmethod(format_timestamp) + str_to_timestamp = staticmethod(_legacy_alias(parse_timestamp, date_str="value")) + to_bool = staticmethod(_legacy_alias(parse_bool, text="value", default_val="default")) + str_from_cookiejar = staticmethod(_legacy_alias(cookiejar_to_string, cj="cookiejar")) + get_idlist = staticmethod(_legacy_alias(extract_named_ids, dicts="entries")) + md5_hash = staticmethod(_legacy_md5_hash) + str_timehours = staticmethod(format_minutes) + str_amount = staticmethod(_legacy_alias(format_amount, curr="currency")) + count_words = staticmethod(_legacy_alias(count_words, text="value")) + is_media_title_like = staticmethod(_legacy_alias(is_media_title_like, text="value")) + split_text = staticmethod(_legacy_alias(split_by_bytes, text="value")) + get_keyword = staticmethod(parse_search_keyword) + str_title = staticmethod(_legacy_alias(title_case, s="value")) + escape_markdown = staticmethod(_legacy_alias(escape_markdown, content="value")) + get_domain_address = staticmethod( + _legacy_alias(parse_address, prefix="include_scheme") + ) + str_series = staticmethod(_legacy_alias(compact_numbers, array="numbers")) + format_ep = staticmethod(_legacy_alias(format_ranges, nums="numbers")) + is_number = staticmethod(_legacy_alias(is_number, text="value")) + find_common_prefix = staticmethod( + _legacy_alias(common_prefix, str1="first", str2="second") + ) + compare_version = staticmethod( + _legacy_alias( + compare_version, + v1="source", + compare_type="comparison", + v2="target", + ) + ) + diff_time_str = staticmethod(_legacy_alias(format_remaining, time_str="value")) + safe_strip = staticmethod(strip_optional) + is_valid_html_element = staticmethod(_legacy_alias(DomUtils.has_child_elements, elem="element")) + is_link = staticmethod(_legacy_alias(is_link, text="value")) + is_magnet_link = staticmethod(is_magnet_link) + natural_sort_key = staticmethod(_legacy_alias(natural_sort_key, text="value")) + + +__all__ = ["StringUtils"] diff --git a/app/sdk/utilities.py b/app/sdk/utilities.py index 6980dac18..81d9eb385 100644 --- a/app/sdk/utilities.py +++ b/app/sdk/utilities.py @@ -1,6 +1,5 @@ """插件常用的无状态通用工具。""" -from app.domain.string import StringUtils from app.foundation.crypto import CryptoJsUtils from app.foundation.dom import DomUtils from app.foundation.text import cut @@ -11,6 +10,7 @@ from app.runtime.execution import log_execution_time, retry from app.runtime.localization import LocaleHelper from app.runtime.scheduling import TimerUtils from app.application.security.otp import OtpUtils +from app.sdk.string import StringUtils decrypt = CryptoJsUtils.decrypt diff --git a/docs/backend-module-refactor-compatibility.md b/docs/backend-module-refactor-compatibility.md index 273b2bbf1..47846f00e 100644 --- a/docs/backend-module-refactor-compatibility.md +++ b/docs/backend-module-refactor-compatibility.md @@ -114,7 +114,8 @@ Entrypoints / Plugins --> Application / Chain --> Domain + Ports --> Foundation | `core.event` 中的 EventManager | `runtime.events` | 移除按类名猜路径及直接实例化 PluginManager/ModuleManager/MessageHelper | | `core.module`、`core.plugin` | `runtime.extensions` | 安装、发现、生命周期和业务上报通过接口/装配连接 | | `core.cache` | `app.runtime.cache` + `app.adapters.cache.backends` | runtime 保留契约、内存策略和装饰器;cache adapters 实现 Redis/文件 I/O;SDK 维持旧完整符号集 | -| `utils.string/url/identity/coalesce/structures` 等纯函数 | `foundation` 对应领域文件 | 确认不读取全局配置、不执行 I/O、不导入高层模块 | +| `utils.string` 聚合类 | `foundation.text/size/temporal/url/dom/crypto/version` + `domain.title/episode/site/torrent` | 宿主按真实职责直接调用;完整 `StringUtils` 静态方法面只在 `app.sdk.string` 组合,旧 `app.utils.string` 和 `app.domain.string` 精确映射到该 SDK 模块 | +| `utils.url/identity/coalesce/structures` 等纯函数 | `foundation` 对应能力文件 | 确认不读取全局配置、不执行 I/O、不导入高层模块 | | `utils.http` | `app.adapters.network.http` | 去除对 `settings` 的反向读取,由启动层注入宿主 User-Agent | | `utils.web` | `app.adapters.external.location` | 外部 IP 归属服务是具体生态集成,不是通用网络基础设施 | | `utils.gc` | `app.runtime.gc` | 进程内存观测和回收是运行平台策略,不是外部适配器 | @@ -512,7 +513,7 @@ SYMBOL_ALIASES = { ## 14. 实施结果 -1. `app/core`、`app/helper`、`app/utils` 已无物理 Python 源码,宿主全部使用 canonical 路径。 +1. `app/core`、`app/helper`、`app/utils` 物理目录均已删除,宿主全部使用 canonical 路径,插件旧导入只由虚拟兼容包解析。 2. `app.runtime.compat` 在 `app` 包初始化时安装精确白名单 Finder,旧叶子模块与 canonical 模块保持同一身份。 3. DEBUG 诊断通过运行时命中和插件 AST 扫描互补发现旧引用,生产模式静默。 4. Event、模块、插件和安全边界改为由 startup composition root 注入 resolver、回调和错误处理器,迁移模块不再处于强连通分量。 diff --git a/docs/rules/05-architecture.md b/docs/rules/05-architecture.md index 3b5dfbf31..0034b3451 100644 --- a/docs/rules/05-architecture.md +++ b/docs/rules/05-architecture.md @@ -8,13 +8,10 @@ MoviePilot keeps the established product packages such as `app/chain`, `app/helper` and `app/utils` roots are virtual compatibility packages only; physical Python sources must not be recreated there. -The sole filesystem exception is the non-Python `app/helper/.resource-compat` -marker retained in source archives for old Docker images whose updater still -writes compiled site resources to `/app/app/helper`. When the canonical site -extension is absent, `app/application/site/__init__.py` may add that directory -as a package search fallback. Current images and update flows must still write -only to `app/application/site/`; no Python implementation may return to the -legacy root. +The legacy roots have no physical directories in the source tree. Current +images and update flows write site resources only to `app/application/site/`; +plugin imports under `app.helper.*` are resolved exclusively by the exact +runtime compatibility manifest. Capabilities migrated out of those legacy roots are organized by technical responsibility: @@ -125,16 +122,23 @@ mentions media, site or torrent: | Subdomain | Modules and ownership | |---|---| -| Media | `context.py` owns `Context`, `MediaInfo` and `TorrentInfo`; `media.py` owns source/ID normalization; `scraper.py` owns Kodi-style NFO reading and metadata document generation | +| Media | `context.py` owns `Context`, `MediaInfo` and `TorrentInfo`; `media.py` owns source/ID normalization; `title.py` owns title-candidate and search-keyword rules; `episode.py` owns episode-range display; `scraper.py` owns Kodi-style NFO reading and metadata document generation | | Recognition | `metainfo.py`, `meta/` and `tokens.py` parse names, paths, release groups, streaming platforms, anime, video and music metadata | -| Site | `site.py` interprets HTML into business states such as logged-in and checked-in; configured catalog/auth/index resources stay in `app/application/site/`, DOM parsing stays in foundation and network access stays in adapters | -| Torrent | Identity/title semantics live in the domain model; configured download/cache/file behavior stays in `app/application/torrent.py` | -| Shared business text | `string.py` contains MoviePilot-specific media/site/torrent normalization; generic text primitives stay in `app/foundation/text.py` | +| Site | `site.py` owns site-domain exceptions and interprets HTML into business states such as logged-in and checked-in; configured catalog/auth/index resources stay in `app/application/site/`, generic URL/DOM parsing stays in foundation and network access stays in adapters | +| Torrent | `torrent.py` owns magnet-link semantics; configured download/cache/file behavior stays in `app/application/torrent.py` | `app/domain` may depend only on schemas and foundation. It must not read global settings, access DB/network/filesystem adapters, import Rust, discover services or initialize process runtime state. +`StringUtils` is not a canonical implementation type. Generic text, capacity, +time, URL, DOM, hash and version functions live under `app.foundation`; media +title, episode, site and torrent rules live in their owning domain modules. Host +code must import those implementations directly. `app.sdk.string.StringUtils` +only composes the complete historical static-method surface for plugins, and +both `app.utils.string` and the retired `app.domain.string` resolve to that same +SDK module through the compatibility manifest. + ## Established Packages That Stay in Place The following roots predate this migration and must not be moved or renamed as diff --git a/tests/test_architecture_dependencies.py b/tests/test_architecture_dependencies.py index bdb9c662c..957ed1ab3 100644 --- a/tests/test_architecture_dependencies.py +++ b/tests/test_architecture_dependencies.py @@ -46,6 +46,7 @@ RETIRED_CANONICAL_FILES = ( "app/security/url_safety.py", "app/domain/mediaserver.py", "app/domain/nfo.py", + "app/domain/string.py", "app/log.py", "app/foundation/diagnostics.py", "app/infrastructure/log.py", @@ -226,6 +227,16 @@ def test_legacy_roots_contain_no_python_sources(): assert leftovers == [] +def test_legacy_source_directories_do_not_exist(): + """core/helper/utils 物理目录应完全退役,旧导入只由虚拟兼容包解析。""" + leftovers = [ + root_name + for root_name in ("core", "helper", "utils") + if (APP_ROOT / root_name).exists() + ] + assert leftovers == [] + + def test_retired_canonical_filenames_do_not_return(): """能力包应使用包内语境明确的短文件名,避免再次出现冗余角色后缀。""" leftovers = [ @@ -326,6 +337,31 @@ def test_capability_packages_do_not_import_forbidden_upper_layers(): assert violations == {} +def test_site_domain_uses_foundation_dom_boundary(): + """站点领域规则应依赖 DOM 原语,不得重新耦合聚合字符串工具。""" + modules = _discover_modules() + dependencies = _resolve_imports( + "app.domain.site", + modules["app.domain.site"], + set(modules), + ) + assert "app.foundation.dom" in dependencies + assert "app.domain.string" not in dependencies + + +def test_host_code_does_not_use_string_utils_facade(): + """聚合 StringUtils 只服务插件兼容,宿主实现必须使用拆分后的能力。""" + violations: list[str] = [] + for path in APP_ROOT.rglob("*.py"): + relative = path.relative_to(APP_ROOT) + if relative.parts[0] in {"plugins", "sdk"}: + continue + tree = ast.parse(path.read_text(encoding="utf-8-sig"), filename=str(path)) + if any(isinstance(node, ast.Name) and node.id == "StringUtils" for node in ast.walk(tree)): + violations.append(str(relative)) + assert violations == [] + + def test_runtime_log_is_a_dependency_leaf(): """底层可引用运行时日志,但日志模块本身不得反向导入应用模块。""" modules = _discover_modules() diff --git a/tests/test_downloader_path_mapping.py b/tests/test_downloader_path_mapping.py index 420c0fff0..e275f5333 100644 --- a/tests/test_downloader_path_mapping.py +++ b/tests/test_downloader_path_mapping.py @@ -101,6 +101,13 @@ def _load_transmission_module(): app_module.__path__ = [] core_module = types.ModuleType("app.core") core_module.__path__ = [] + domain_module = types.ModuleType("app.domain") + domain_module.__path__ = [] + foundation_module = types.ModuleType("app.foundation") + foundation_module.__path__ = [] + torrent_rules_module = types.ModuleType("app.domain.torrent") + size_tools_module = types.ModuleType("app.foundation.size") + temporal_tools_module = types.ModuleType("app.foundation.temporal") cache_module = types.ModuleType("app.runtime.cache") modules_module = types.ModuleType("app.modules") modules_module.__path__ = [] @@ -112,9 +119,6 @@ def _load_transmission_module(): config_module = types.ModuleType("app.runtime.config") metainfo_module = types.ModuleType("app.domain.metainfo") log_module = types.ModuleType("app.runtime.log") - utils_module = types.ModuleType("app.utils") - utils_module.__path__ = [] - string_module = types.ModuleType("app.domain.string") transmission_rpc_module = types.ModuleType("transmission_rpc") torrentool_module = types.ModuleType("torrentool") torrentool_module.__path__ = [] @@ -172,22 +176,17 @@ def _load_transmission_module(): self.season_episode = "" self.episode_list = [] - class _StringUtils: - @staticmethod - def is_magnet_link(value): - return isinstance(value, str) and value.startswith("magnet:") + def _is_magnet_link(value): + """按生产领域规则识别测试磁力链接。""" + return isinstance(value, str) and value.startswith("magnet:") - @staticmethod - def generate_random_str(_length): - return "tmp-tag-01" + def _format_size(value): + """返回隔离测试需要的简化容量文本。""" + return str(value) - @staticmethod - def str_filesize(value): - return str(value) - - @staticmethod - def str_secends(value): - return str(value) + def _format_duration(value): + """返回隔离测试需要的简化时长文本。""" + return str(value) class _FileCache: def get(self, *_args, **_kwargs): @@ -211,28 +210,38 @@ def _load_transmission_module(): log_module.logger = _Logger() modules_module._ModuleBase = _ModuleBase modules_module._DownloaderBase = _DownloaderBase - string_module.StringUtils = _StringUtils + torrent_rules_module.is_magnet_link = _is_magnet_link + size_tools_module.format_compact_size = _format_size + temporal_tools_module.format_duration = _format_duration transmission_rpc_module.File = object torrentool_torrent_module.Torrent = SimpleNamespace( from_string=lambda _content: SimpleNamespace(name="test", total_size=1) ) app_module.core = core_module + app_module.domain = domain_module + app_module.foundation = foundation_module app_module.modules = modules_module app_module.schemas = schemas_module - app_module.utils = utils_module + domain_module.torrent = torrent_rules_module + foundation_module.size = size_tools_module + foundation_module.temporal = temporal_tools_module core_module.cache = cache_module core_module.config = config_module core_module.metainfo = metainfo_module modules_module.transmission = transmission_package_module transmission_package_module.transmission = transmission_client_module schemas_module.types = schema_types_module - utils_module.string = string_module torrentool_module.torrent = torrentool_torrent_module stub_modules = { "app": app_module, "app.core": core_module, + "app.domain": domain_module, + "app.domain.torrent": torrent_rules_module, + "app.foundation": foundation_module, + "app.foundation.size": size_tools_module, + "app.foundation.temporal": temporal_tools_module, "app.runtime.cache": cache_module, "app.runtime.config": config_module, "app.domain.metainfo": metainfo_module, @@ -242,8 +251,6 @@ def _load_transmission_module(): "app.modules.transmission.transmission": transmission_client_module, "app.schemas": schemas_module, "app.schemas.types": schema_types_module, - "app.utils": utils_module, - "app.domain.string": string_module, "transmission_rpc": transmission_rpc_module, "torrentool": torrentool_module, "torrentool.torrent": torrentool_torrent_module, diff --git a/tests/test_nexus_audiences_parser.py b/tests/test_nexus_audiences_parser.py index 5a0a8f538..be2df4728 100644 --- a/tests/test_nexus_audiences_parser.py +++ b/tests/test_nexus_audiences_parser.py @@ -1,6 +1,6 @@ # -*- coding: utf-8 -*- from app.modules.indexer.parser.nexus_audiences import NexusAudiencesSiteUserInfo -from app.domain.string import StringUtils +from app.foundation.size import parse_size def test_audiences_userbar_metrics_override_generic_nexus_regex(): @@ -38,8 +38,8 @@ def test_audiences_userbar_metrics_override_generic_nexus_regex(): assert parser.userid == "18978" assert parser.username == "jxxghp" assert parser.user_level == "(江湖儿女)Elite User" - assert parser.upload == StringUtils.num_filesize("10.150 TB") - assert parser.download == StringUtils.num_filesize("3.624 TB") + assert parser.upload == parse_size("10.150 TB") + assert parser.download == parse_size("3.624 TB") assert parser.ratio == 2.801 assert parser.bonus == 1973896.2 assert parser.seeding == 355 diff --git a/tests/test_plugin_sdk.py b/tests/test_plugin_sdk.py index 64a9ffd52..b20437765 100644 --- a/tests/test_plugin_sdk.py +++ b/tests/test_plugin_sdk.py @@ -19,13 +19,15 @@ def test_sdk_exports_canonical_plugin_interfaces(): from app.domain.meta.metamusic import MetaMusic as CanonicalMetaMusic from app.domain.metainfo import MetaInfo as CanonicalMetaInfo from app.domain.scraper import NfoReader as CanonicalNfoReader - from app.domain.string import StringUtils as CanonicalStringUtils + LegacyDomainStringUtils = importlib.import_module( + "app.domain.string" + ).StringUtils from app.foundation.crypto import CryptoJsUtils from app.runtime.extensions.module_manager import ModuleManager as CanonicalModuleManager from app.runtime.extensions.plugin_manager import PluginManager as CanonicalPluginManager from app.adapters.network.http import RequestUtils as CanonicalRequestUtils from app.application.rss import RssHelper as CanonicalRssHelper - from app.application.site.sites import SitesHelper as CanonicalSitesHelper + from app.application.site.sites import SitesHelper as CanonicalSitesHelper # pylint: disable=no-name-in-module from app.runtime.cache import Cache as CanonicalCache from app.runtime.cache import cached as canonical_cached from app.runtime.config import settings as canonical_settings @@ -49,7 +51,7 @@ def test_sdk_exports_canonical_plugin_interfaces(): assert RssHelper is CanonicalRssHelper assert SitesHelper is CanonicalSitesHelper assert NotificationHelper is CanonicalNotificationHelper - assert UtilityStringUtils is CanonicalStringUtils + assert UtilityStringUtils is LegacyDomainStringUtils assert decrypt is CryptoJsUtils.decrypt assert encrypt is CryptoJsUtils.encrypt assert ModuleManager is CanonicalModuleManager diff --git a/tests/test_qbittorrent_compat.py b/tests/test_qbittorrent_compat.py index 510ca6d52..ebbedcc2b 100644 --- a/tests/test_qbittorrent_compat.py +++ b/tests/test_qbittorrent_compat.py @@ -13,8 +13,15 @@ def _load_qbittorrent_modules(): app_module.__path__ = [] core_module = types.ModuleType("app.core") core_module.__path__ = [] - utils_module = types.ModuleType("app.utils") - utils_module.__path__ = [] + domain_module = types.ModuleType("app.domain") + domain_module.__path__ = [] + foundation_module = types.ModuleType("app.foundation") + foundation_module.__path__ = [] + torrent_rules_module = types.ModuleType("app.domain.torrent") + size_tools_module = types.ModuleType("app.foundation.size") + temporal_tools_module = types.ModuleType("app.foundation.temporal") + text_tools_module = types.ModuleType("app.foundation.text") + url_tools_module = types.ModuleType("app.foundation.url") modules_module = types.ModuleType("app.modules") modules_module.__path__ = [] qbittorrent_package_module = types.ModuleType("app.modules.qbittorrent") @@ -25,7 +32,6 @@ def _load_qbittorrent_modules(): metainfo_module = types.ModuleType("app.domain.metainfo") schemas_module = types.ModuleType("app.schemas") schema_types_module = types.ModuleType("app.schemas.types") - string_module = types.ModuleType("app.domain.string") torrentool_module = types.ModuleType("torrentool") torrentool_module.__path__ = [] torrentool_torrent_module = types.ModuleType("torrentool.torrent") @@ -46,28 +52,27 @@ def _load_qbittorrent_modules(): def error(self, *_args, **_kwargs): pass - class _StringUtils: - @staticmethod - def get_domain_address(address, prefix=False): - return address, 8080 + def _is_magnet_link(value): + """按生产领域规则识别测试磁力链接。""" + if isinstance(value, bytes): + return value.startswith(b"magnet:") + return isinstance(value, str) and value.startswith("magnet:") - @staticmethod - def is_magnet_link(value): - if isinstance(value, bytes): - return value.startswith(b"magnet:") - return isinstance(value, str) and value.startswith("magnet:") + def _parse_address(address, include_scheme=False): + """返回隔离测试使用的主机和固定端口。""" + return address, 8080 - @staticmethod - def generate_random_str(_length): - return "tmp-tag-01" + def _random_string(_length): + """生成可断言的固定临时标签。""" + return "tmp-tag-01" - @staticmethod - def str_filesize(value): - return str(value) + def _format_size(value): + """返回隔离测试需要的简化容量文本。""" + return str(value) - @staticmethod - def str_secends(value): - return str(value) + def _format_duration(value): + """返回隔离测试需要的简化时长文本。""" + return str(value) class _FileCache: def get(self, *_args, **_kwargs): @@ -131,7 +136,11 @@ def _load_qbittorrent_modules(): schema_types_module.DownloadTaskState = DownloadTaskState schema_types_module.ModuleType = ModuleType schema_types_module.DownloaderType = DownloaderType - string_module.StringUtils = _StringUtils + torrent_rules_module.is_magnet_link = _is_magnet_link + url_tools_module.parse_address = _parse_address + text_tools_module.random_string = _random_string + size_tools_module.format_compact_size = _format_size + temporal_tools_module.format_duration = _format_duration modules_module._ModuleBase = _ModuleBase modules_module._DownloaderBase = _DownloaderBase torrentool_torrent_module.Torrent = _Torrent @@ -145,14 +154,19 @@ def _load_qbittorrent_modules(): qbittorrentapi_transfer_module.TransferInfoDictionary = dict app_module.core = core_module + app_module.domain = domain_module + app_module.foundation = foundation_module app_module.log = log_module app_module.modules = modules_module app_module.schemas = schemas_module - app_module.utils = utils_module + domain_module.torrent = torrent_rules_module + foundation_module.size = size_tools_module + foundation_module.temporal = temporal_tools_module + foundation_module.text = text_tools_module + foundation_module.url = url_tools_module core_module.cache = cache_module core_module.config = config_module core_module.metainfo = metainfo_module - utils_module.string = string_module schemas_module.types = schema_types_module modules_module.qbittorrent = qbittorrent_package_module torrentool_module.torrent = torrentool_torrent_module @@ -160,6 +174,13 @@ def _load_qbittorrent_modules(): stub_modules = { "app": app_module, "app.core": core_module, + "app.domain": domain_module, + "app.domain.torrent": torrent_rules_module, + "app.foundation": foundation_module, + "app.foundation.size": size_tools_module, + "app.foundation.temporal": temporal_tools_module, + "app.foundation.text": text_tools_module, + "app.foundation.url": url_tools_module, "app.runtime.cache": cache_module, "app.runtime.config": config_module, "app.domain.metainfo": metainfo_module, @@ -168,8 +189,6 @@ def _load_qbittorrent_modules(): "app.modules.qbittorrent": qbittorrent_package_module, "app.schemas": schemas_module, "app.schemas.types": schema_types_module, - "app.utils": utils_module, - "app.domain.string": string_module, "qbittorrentapi": qbittorrentapi_module, "qbittorrentapi.client": qbittorrentapi_client_module, "qbittorrentapi.transfer": qbittorrentapi_transfer_module, diff --git a/tests/test_resource_v3.py b/tests/test_resource_v3.py index 109ae3ea7..37fb42971 100644 --- a/tests/test_resource_v3.py +++ b/tests/test_resource_v3.py @@ -1,7 +1,5 @@ -from importlib.machinery import EXTENSION_SUFFIXES, PathFinder from pathlib import Path -from app.application.site import _include_legacy_resource_directory from app.runtime.config import settings from app.adapters.system.resource import ( ResourceHelper, @@ -13,41 +11,6 @@ from app.startup import modules_initializer ROOT_DIR = Path(__file__).resolve().parents[1] -def test_legacy_docker_updater_resource_directory_remains_importable(tmp_path): - """旧镜像把资源写入 helper 时,canonical 站点包仍应找到对应扩展。""" - package_dir = tmp_path / "app" / "application" / "site" - legacy_dir = tmp_path / "app" / "helper" - package_dir.mkdir(parents=True) - legacy_dir.mkdir(parents=True) - extension_path = legacy_dir / f"sites{EXTENSION_SUFFIXES[0]}" - extension_path.touch() - package_paths = [str(package_dir)] - - _include_legacy_resource_directory(package_paths, package_dir) - - assert (ROOT_DIR / "app" / "helper" / ".resource-compat").is_file() - assert package_paths == [str(package_dir), str(legacy_dir)] - spec = PathFinder.find_spec("app.application.site.sites", package_paths) - assert spec is not None - assert spec.origin == str(extension_path) - - -def test_canonical_site_extension_takes_priority_over_legacy_directory(tmp_path): - """canonical 扩展存在时不得把旧资源目录加入站点包搜索路径。""" - package_dir = tmp_path / "app" / "application" / "site" - legacy_dir = tmp_path / "app" / "helper" - package_dir.mkdir(parents=True) - legacy_dir.mkdir(parents=True) - extension_name = f"sites{EXTENSION_SUFFIXES[0]}" - (package_dir / extension_name).touch() - (legacy_dir / extension_name).touch() - package_paths = [str(package_dir)] - - _include_legacy_resource_directory(package_paths, package_dir) - - assert package_paths == [str(package_dir)] - - def test_resource_helper_uses_v3_only(): """在线资源更新器必须只请求 V3 清单、目录和站点索引文件。""" assert settings.VERSION_FLAG == "v3" diff --git a/tests/test_site_domain.py b/tests/test_site_domain.py new file mode 100644 index 000000000..4478c8a35 --- /dev/null +++ b/tests/test_site_domain.py @@ -0,0 +1,42 @@ +from lxml import etree + +from app.domain.site import SiteUtils +from app.foundation.dom import DomUtils +from app.sdk.string import StringUtils + + +def test_dom_child_element_check_preserves_existing_semantics(): + """DOM 基础判断应区分空树和至少包含一个子元素的树。""" + empty_tree = etree.HTML("") + populated_tree = etree.HTML("") + + assert DomUtils.has_child_elements(None) is False + assert DomUtils.has_child_elements(empty_tree) is False + assert DomUtils.has_child_elements(populated_tree) is True + + +def test_plugin_string_facade_delegates_to_dom_primitive(): + """存量 StringUtils 调用应继续获得与 DOM 原语一致的结果。""" + html = etree.HTML("") + + assert StringUtils.is_valid_html_element(html) is DomUtils.has_child_elements(html) + + +def test_site_login_state_is_derived_from_html_markers(): + """站点登录规则应识别退出入口并拒绝密码登录页。""" + logged_in_html = '退出' + login_form_html = '' + + assert SiteUtils.is_logged_in(logged_in_html) is True + assert SiteUtils.is_logged_in(login_form_html) is False + assert SiteUtils.is_logged_in("") is False + + +def test_site_checkin_state_is_derived_from_html_markers(): + """站点签到规则应把仍存在签到入口的页面识别为未签到。""" + pending_html = '签到' + completed_html = '退出' + + assert SiteUtils.is_checkin(pending_html) is False + assert SiteUtils.is_checkin(completed_html) is True + assert SiteUtils.is_checkin("") is False diff --git a/tests/test_string.py b/tests/test_string.py index e520cf23d..bf2215747 100644 --- a/tests/test_string.py +++ b/tests/test_string.py @@ -1,26 +1,27 @@ from unittest import TestCase -from app.domain.string import StringUtils +from app.domain.title import is_media_title_like -class StringUtilsTest(TestCase): +class MediaTitleTest(TestCase): + """验证媒体标题候选规则。""" def test_is_media_title_like_true(self): - self.assertTrue(StringUtils.is_media_title_like("盗梦空间")) - self.assertTrue(StringUtils.is_media_title_like("The Lord of the Rings")) - self.assertTrue(StringUtils.is_media_title_like("庆余年 第2季")) - self.assertTrue(StringUtils.is_media_title_like("The Office S01E01")) - self.assertTrue(StringUtils.is_media_title_like("权力的游戏 Game of Thrones")) - self.assertTrue(StringUtils.is_media_title_like("Spider-Man: No Way Home 2021")) + self.assertTrue(is_media_title_like("盗梦空间")) + self.assertTrue(is_media_title_like("The Lord of the Rings")) + self.assertTrue(is_media_title_like("庆余年 第2季")) + self.assertTrue(is_media_title_like("The Office S01E01")) + self.assertTrue(is_media_title_like("权力的游戏 Game of Thrones")) + self.assertTrue(is_media_title_like("Spider-Man: No Way Home 2021")) def test_is_media_title_like_false(self): - self.assertFalse(StringUtils.is_media_title_like("")) - self.assertFalse(StringUtils.is_media_title_like(" ")) - self.assertFalse(StringUtils.is_media_title_like("a")) - self.assertFalse(StringUtils.is_media_title_like("第2季")) - self.assertFalse(StringUtils.is_media_title_like("S01E01")) - self.assertFalse(StringUtils.is_media_title_like("#推荐电影")) - self.assertFalse(StringUtils.is_media_title_like("请帮我推荐一部电影")) - self.assertFalse(StringUtils.is_media_title_like("盗梦空间怎么样?")) - self.assertFalse(StringUtils.is_media_title_like("我想看盗梦空间")) - self.assertFalse(StringUtils.is_media_title_like("继续")) + self.assertFalse(is_media_title_like("")) + self.assertFalse(is_media_title_like(" ")) + self.assertFalse(is_media_title_like("a")) + self.assertFalse(is_media_title_like("第2季")) + self.assertFalse(is_media_title_like("S01E01")) + self.assertFalse(is_media_title_like("#推荐电影")) + self.assertFalse(is_media_title_like("请帮我推荐一部电影")) + self.assertFalse(is_media_title_like("盗梦空间怎么样?")) + self.assertFalse(is_media_title_like("我想看盗梦空间")) + self.assertFalse(is_media_title_like("继续")) diff --git a/tests/test_string_compat.py b/tests/test_string_compat.py new file mode 100644 index 000000000..94cb605f8 --- /dev/null +++ b/tests/test_string_compat.py @@ -0,0 +1,120 @@ +import importlib +import inspect + +from lxml import etree + +from app.sdk.string import StringUtils + + +EXPECTED_METHODS = { + "clear", + "clear_file_name", + "clear_upper", + "compare_version", + "count_words", + "diff_time_str", + "escape_markdown", + "find_common_prefix", + "format_ep", + "format_size", + "format_timestamp", + "generate_random_str", + "get_base_url", + "get_domain_address", + "get_idlist", + "get_keyword", + "get_time", + "get_url_domain", + "get_url_host", + "get_url_netloc", + "get_url_sld", + "is_all_chinese", + "is_chinese", + "is_english_word", + "is_japanese", + "is_korean", + "is_link", + "is_magnet_link", + "is_media_title_like", + "is_number", + "is_valid_html_element", + "md5_hash", + "natural_sort_key", + "num_filesize", + "safe_strip", + "split_text", + "str_amount", + "str_filesize", + "str_float", + "str_from_cookiejar", + "str_int", + "str_secends", + "str_series", + "str_timehours", + "str_timelong", + "str_title", + "str_to_timestamp", + "to_bool", + "unify_datetime_str", + "url_equal", +} + + +def test_string_utils_keeps_complete_plugin_method_surface(): + """SDK 门面必须保留拆分前全部静态方法名称。""" + assert EXPECTED_METHODS <= set(dir(StringUtils)) + + +def test_legacy_string_modules_share_sdk_facade_identity(): + """旧 utils/domain 路径与 SDK 应解析到同一个轻量兼容模块。""" + sdk_module = importlib.import_module("app.sdk.string") + legacy_utils = importlib.import_module("app.utils.string") + legacy_domain = importlib.import_module("app.domain.string") + + assert legacy_utils is sdk_module + assert legacy_domain is sdk_module + assert legacy_utils.StringUtils is StringUtils + assert legacy_domain.StringUtils is StringUtils + + +def test_string_utils_preserves_legacy_keyword_arguments(): + """插件按旧参数名调用时应正确转交到拆分后的实现。""" + assert StringUtils.clear(text="A.B C", replace_word="-", allow_space=True) == "A-B C" + assert StringUtils.str_filesize(size=1024 ** 3, pre=2) == "1.0G" + assert StringUtils.url_equal(url1="https://www.example.com", url2="example.com") is True + assert StringUtils.generate_random_str(randomlength=8) + assert StringUtils.to_bool(text="", default_val=True) is True + assert StringUtils.str_amount(amount=1234, curr="¥") == "¥1,234" + assert StringUtils.get_domain_address( + address="example.com:8080", prefix=True + ) == ("http://example.com", 8080) + assert StringUtils.compare_version( + v1="1.2.0", compare_type="<", v2="1.3.0" + ) is True + + +def test_string_utils_preserves_legacy_method_signatures(): + """反射静态方法签名时也应继续看到插件熟悉的旧参数名。""" + assert list(inspect.signature(StringUtils.clear).parameters) == [ + "text", + "replace_word", + "allow_space", + ] + assert list(inspect.signature(StringUtils.compare_version).parameters) == [ + "v1", + "compare_type", + "v2", + "verbose", + ] + + +def test_string_utils_routes_representative_capabilities(): + """容量、站点、媒体、剧集、种子和 DOM 能力应保持历史结果。""" + assert StringUtils.num_filesize("10.150 TB") == 11160043021926 + assert StringUtils.get_url_domain("https://u2.dmhy.org/torrents.php") == "u2.dmhy.org" + assert StringUtils.is_media_title_like("The Office S01E01") is True + assert StringUtils.format_ep([1, 2, 3, 5]) == "E01-E03、E05" + assert StringUtils.is_magnet_link("magnet:?xt=urn:btih:abc") is True + assert StringUtils.is_valid_html_element( + etree.HTML("") + ) is True diff --git a/tests/test_torrent_leech_parser.py b/tests/test_torrent_leech_parser.py index 1be1ef3e7..462b5ba67 100644 --- a/tests/test_torrent_leech_parser.py +++ b/tests/test_torrent_leech_parser.py @@ -1,7 +1,7 @@ # -*- coding: utf-8 -*- from app.modules.indexer import parser as parser_module from app.modules.indexer.parser.torrent_leech import TorrentLeechSiteUserInfo -from app.domain.string import StringUtils +from app.foundation.size import parse_size PROFILE_VIEW_HTML = """ @@ -93,8 +93,8 @@ def test_torrent_leech_refresh_prefers_topbar_user_and_parses_profile_once(monke assert parser.userid == "example_user" assert parser.username == "example_user" - assert parser.upload == StringUtils.num_filesize("41.54 GB") - assert parser.download == StringUtils.num_filesize("10.16 GB") + assert parser.upload == parse_size("41.54 GB") + assert parser.download == parse_size("10.16 GB") assert parser.ratio == 4.089 assert parser.user_level == "Registered" assert parser.join_at == "2022-09-04 00:00:00"