feat: 用户数据解析器自动探测 + Gazelle/NexusProject 变种兼容 (#6403)

This commit is contained in:
SayItDitto
2026-08-23 09:13:55 +08:00
committed by GitHub
parent 0c05b260ea
commit b1a55b76e5
3 changed files with 87 additions and 3 deletions
+21
View File
@@ -93,6 +93,27 @@ class GazelleSiteUserInfo(SiteParserBase):
'//div[contains(@class, "box_userinfo_stats")]//li[contains(text(), "加入时间")]/span/text()')
if join_at_text:
self.join_at = time_tools.normalize_datetime(join_at_text[0].strip())
# 兼容部分 Gazelle 站点(如 JPopsuki)以文本形式展示上传/下载:
# <li>Uploaded: 77.44 GB</li> / <li>Downloaded: 8.51 GB</li>
if not self.upload:
upload_text = html.xpath(
'//li[starts-with(normalize-space(text()), "Uploaded:")]')
if upload_text:
size_match = re.search(
r"([\d.,]+\s*[GMKT]?i?B)", upload_text[0].xpath("string(.)"), re.I)
if size_match:
self.upload = size_tools.parse_size(size_match.group(1))
if not self.download:
download_text = html.xpath(
'//li[starts-with(normalize-space(text()), "Downloaded:")]')
if download_text:
size_match = re.search(
r"([\d.,]+\s*[GMKT]?i?B)", download_text[0].xpath("string(.)"), re.I)
if size_match:
self.download = size_tools.parse_size(size_match.group(1))
if not self.ratio and self.upload and self.download:
self.ratio = round(self.upload / self.download, 3)
finally:
if html is not None:
del html
+39 -1
View File
@@ -1,6 +1,8 @@
# -*- coding: utf-8 -*-
import re
from lxml import etree
from app.modules.indexer.parser import SiteSchema
from app.modules.indexer.parser.nexus_php import NexusPhpSiteUserInfo
@@ -11,9 +13,45 @@ class NexusProjectSiteUserInfo(NexusPhpSiteUserInfo):
def _parse_site_page(self, html_text: str):
html_text = self._prepare_html_text(html_text)
user_detail = re.search(r"userdetails.php\?id=(\d+)", html_text)
user_detail = re.search(r"userdetails\.php\?id=(\d+)", html_text)
if user_detail and user_detail.group().strip():
self._user_detail_page = user_detail.group().strip().lstrip('/')
self.userid = user_detail.group(1)
else:
# 兼容部分 NexusProject 变种站点(如 star-space.net)的
# p_user/user_detail.php?uid= 用户定位格式
user_detail = re.search(r"user_detail\.php\?uid=(\d+)", html_text)
if user_detail and user_detail.group().strip():
self._user_detail_page = user_detail.group().strip().lstrip('/')
self.userid = user_detail.group(1)
uname = re.search(
r"user_detail\.php\?uid=\d+[^>]*?>\s*<[^>]*>([^<]+)<", html_text)
if uname:
self.username = uname.group(1).strip()
self._torrent_seeding_page = f"viewusertorrents.php?id={self.userid}&show=seeding"
def _parse_user_traffic_info(self, html_text):
# 兼容部分 NexusProject 变种站点(如 star-space.net)以
# span#user_info / span#user_info_no_hover 文本展示流量:
# "上传:445.11 G" / "下载:29.61 G"(单位无 B 后缀)
try:
html = etree.HTML(html_text)
if html is not None:
body_text = " ".join(html.xpath(
'//span[@id="user_info"]//text() | //span[@id="user_info_no_hover"]//text()'))
size_match = re.search(r"上传[:]\s*([\d.,]+)\s*([GMKT]?i?B?)", body_text, re.I)
if size_match:
self.upload = self.num_filesize(
f"{size_match.group(1).replace(',', '')} {size_match.group(2).upper()}B")
size_match = re.search(r"下载[:]\s*([\d.,]+)\s*([GMKT]?i?B?)", body_text, re.I)
if size_match:
self.download = self.num_filesize(
f"{size_match.group(1).replace(',', '')} {size_match.group(2).upper()}B")
if self.upload and self.download:
self.ratio = round(self.upload / self.download, 3)
if self.upload or self.download:
return
except Exception:
pass
super()._parse_user_traffic_info(html_text)