mirror of
https://github.com/krau/SaveAny-Bot.git
synced 2026-09-04 23:26:39 +08:00
* fix: prevent queue deadlock after cancelling tasks Get no longer recurses while holding the mutex. Cancelled queued tasks leave the map, making their IDs reusable. Closed empty queues return ErrQueueClosed. * fix: init task queue lazily and close on shutdown AddTask is safe before Run by initializing the queue once. Close unblocks workers waiting in Get. * fix: make resource fingerprints deterministic Sort map keys before hashing so Resource.ID is stable. * fix: harden per-item processing tracking in tasks Use the resource fingerprint as the dedup key on insert and delete. Check and set processing entries atomically instead of TOCTOU. Count only successful downloads. * fix: honor IgnoreErrors in batch tasks Element failures no longer cancel sibling elements or stop later groups. * fix: report streamed upload bytes in batch tasks Stream downloads report their byte count as the upload total. * fix: validate i18n key parity across locales geni18n fails when a language file misses any key. Align the syncpeers completion key between en and zh-Hans. Translate three untranslated parse keys in zh-Hans. * refactor: share progress throttling helpers Move size-tiered and count-based throttling into progressutil. Localize hardcoded Chinese progress strings. Drop five duplicated implementations and dead local copies. * refactor: share unique filename logic Storage backends use fsutil.UniquePath instead of local loops. * fix: sanitize local storage paths Reject absolute paths and dot-dot escapes in Save. Check close errors and wrap creation failures. * fix: preserve webdav error causes Wrap mkdir and write failures with %w and drop dead error values. * fix: kill rclone subprocess on reader close Prevent cat processes from hanging after the pipe is closed. * fix: fail alist init instead of exiting Replace log.Fatalf with wrapped errors so a bad alist cannot kill the bot. Cancel token refresh with the init context and re-login on 401. * fix: enforce telegram album limits and track saved paths Use tglimit.MaxAlbumItems for album batching and video splitting. Exists now reports previously saved paths instead of always false. * fix: guard storage registry maps Protect Storages and UserStorages with mutexes and expose read accessors. * fix: harden JS parser plugin runtime Validate semver instead of panicking and require canHandle. Recover plugin worker panics and time out CanHandle calls. Return a copy from the registry and sanitize install filenames. * fix: guard kemono parser against nil fields Skip sparse preview and attachment entries instead of panicking. * refactor: remove commented-out dead code Drop unused kemono legacy types and commented response structs. * test: cover invalid plugin version rejection * fix: show all queued tasks in /task list Render up to ten tasks and append the truncation note once. * fix: guard callback data parsing Reject malformed callback payloads before indexing split parts. * fix: isolate media groups per user Key pending groups by chat, user and group id. * fix: require permission for callback handlers * fix: initialize userbot context once Replace the racy lazy init with sync.OnceValue. * fix: avoid leaking raw errors in /dir reply * fix: notify users on invalid update version * fix: fail fast when API listen fails Bind synchronously and surface errors instead of logging them. * fix: add timeouts and backoff to webhook delivery * refactor: remove dead code from api and bot Drop the empty ProgressTracker shim, unused token context key and a redundant SetBotCommands call. * fix: load remote config without local lookup Skip the local file search after reading a config URL and add a timeout. * refactor: drop unused hook config * docs: document parser plugin config * ci: fix BuildTime formatting and align checkout Actions format does not format dates; pass the raw timestamp. * chore: ignore cache directory * fix: make cache init idempotent * fix: upload only downloaded batch elements Failed elements keep partial cache files and never reach the backend. Successfully downloaded siblings still upload when one element fails. * fix: record telegram saved paths only after upload Skip-large returns a sentinel so skipped files are not marked as saved. * fix: deduplicate alist token refreshes Guard token access with a mutex and merge concurrent logins. Reuse a recent refresh to avoid login storms. * test: cover concurrent alist 401 retry Ten parallel uploads share a single re-login under -race. * fix: count parsed resources in progress text * fix: localize storage lookup errors in /dir Use the shared i18n key and escape the dynamic error. * fix: deduplicate concurrent storage initialization singleflight merges first-time inits so side effects are not duplicated. * fix: guard nil progress trackers in api-created tasks Batch, telegraph and transfer tasks run without a Telegram tracker when created through the API; their callbacks must not panic. * fix: keep album order when filtering failed batch items Download results are stored by original index so surviving elements keep their source order. * test: cover nil-tracker task execution * fix: report partial failure in batch done message IgnoreErrors runs with failed elements show success and failed counts instead of claiming every file completed. * fix: skip login refresh for token-only alist storage 401 responses surface the auth error instead of sending a credential-less login; the refresh window never exceeds TokenExp. * test: cover alist refresh semantics Concurrent refresh uses username/password; token-only storage never attempts a login on 401. * fix: call tracker from notifyProgress instead of recursing The helper called itself, overflowing the stack on any batch task with a progress tracker. * fix: propagate cancellation past IgnoreErrors Cancelled tasks must not be reported as successful: only ordinary element failures are ignored. * test: cover notifyProgress tracker call * test: cover cancellation with IgnoreErrors * fix: refresh alist token after startup 401s The init login no longer satisfies the dedup window, so an early 401 triggers a real refresh; later 401s reuse it. Clarify that streaming uploads cannot replay their body. * test: exercise the alist 401 refresh path Concurrent uploads now reject the init token, share one refresh and retry with the new token; token-only stays inert. * style: trim verbose comments Drop process-style explanations; keep one-line behavior notes. * style: gofmt test files * fix: drop unbounded saved-path cache in telegram storage Telegram cannot reliably query remote file existence, so the cache answered a question it could not answer and grew without bound. Exists returns false again, as before. * style: format codes
183 lines
5.0 KiB
Go
183 lines
5.0 KiB
Go
package kemono
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"errors"
|
|
"fmt"
|
|
"net/http"
|
|
"net/url"
|
|
"path"
|
|
"strings"
|
|
|
|
"github.com/charmbracelet/log"
|
|
"github.com/duke-git/lancet/v2/strutil"
|
|
"github.com/krau/SaveAny-Bot/common/utils/netutil"
|
|
"github.com/krau/SaveAny-Bot/pkg/parser"
|
|
)
|
|
|
|
type KemonoParser struct{}
|
|
|
|
var (
|
|
kemonoDomains = []string{
|
|
"kemono.su",
|
|
"kemono.cr",
|
|
}
|
|
ErrFailedToExtractInfo = errors.New("failed to extract download info from URL")
|
|
)
|
|
|
|
const (
|
|
kemonoApiBase = "https://kemono.cr/api/v1"
|
|
)
|
|
|
|
func (k *KemonoParser) CanHandle(text string) bool {
|
|
text = strings.TrimPrefix(text, "https://")
|
|
text = strings.TrimPrefix(text, "http://")
|
|
|
|
var matchesDomain bool
|
|
for _, domain := range kemonoDomains {
|
|
if strings.Contains(text, domain) {
|
|
matchesDomain = true
|
|
break
|
|
}
|
|
}
|
|
if !matchesDomain {
|
|
return false
|
|
}
|
|
|
|
var path string
|
|
for _, domain := range kemonoDomains {
|
|
if _, after, ok := strings.Cut(text, domain); ok {
|
|
remaining := after
|
|
if len(remaining) > 0 && remaining[0] == '/' {
|
|
path = remaining[1:]
|
|
}
|
|
break
|
|
}
|
|
}
|
|
|
|
if path == "" {
|
|
return false
|
|
}
|
|
|
|
parts := strings.Split(path, "/")
|
|
// servicename/user/id (user profile page)
|
|
// servicename/user/id/post/id (post page)
|
|
return len(parts) == 3 || (len(parts) == 5 && parts[3] == "post")
|
|
}
|
|
|
|
func (k *KemonoParser) Parse(ctx context.Context, u string) (*parser.Item, error) {
|
|
info := extractDownloadInfoFromURL(u)
|
|
if info == nil {
|
|
return nil, ErrFailedToExtractInfo
|
|
}
|
|
if info.PostID != "" {
|
|
return k.parseOne(ctx, info)
|
|
}
|
|
return k.parseUserPage(ctx, info)
|
|
}
|
|
|
|
func (k *KemonoParser) parseOne(ctx context.Context, info *DownloadInfo) (*parser.Item, error) {
|
|
client := netutil.DefaultParserHTTPClient()
|
|
endpoint := fmt.Sprintf("%s/%s/user/%s/post/%s", kemonoApiBase, info.ServiceName, info.UserID, info.PostID)
|
|
req, err := http.NewRequestWithContext(ctx, http.MethodGet, endpoint, nil)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("failed to create request to Kemono API: %w", err)
|
|
}
|
|
req.Header.Set("Accept", "text/css")
|
|
resp, err := client.Do(req)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("failed to fetch Kemono API: %w", err)
|
|
}
|
|
defer resp.Body.Close()
|
|
if resp.StatusCode != http.StatusOK {
|
|
return nil, fmt.Errorf("failed to fetch Kemono API, status code: %d", resp.StatusCode)
|
|
}
|
|
var postInfo PostInfo
|
|
if err := json.NewDecoder(resp.Body).Decode(&postInfo); err != nil {
|
|
return nil, fmt.Errorf("failed to decode Kemono API response: %w", err)
|
|
}
|
|
item := &parser.Item{
|
|
Site: "kemono",
|
|
Title: postInfo.Post.Title,
|
|
URL: fmt.Sprintf("https://kemono.cr/%s/user/%s/post/%s", info.ServiceName, info.UserID, info.PostID),
|
|
Author: postInfo.Post.User, // [TODO] request user profile
|
|
Description: postInfo.Post.Content,
|
|
Tags: func() []string {
|
|
if postInfo.Post.Tags != nil {
|
|
return *postInfo.Post.Tags
|
|
}
|
|
return nil
|
|
}(),
|
|
}
|
|
resources := make([]parser.Resource, 0)
|
|
for _, attachment := range postInfo.Attachments {
|
|
if attachment.Server == nil || attachment.Path == nil || attachment.Name == nil {
|
|
continue
|
|
}
|
|
var size int64
|
|
fileUrl := fmt.Sprintf("%s/data%s", *attachment.Server, *attachment.Path)
|
|
headReq, err := http.NewRequestWithContext(ctx, http.MethodHead, fileUrl, nil)
|
|
if err == nil {
|
|
resp, err := client.Do(headReq)
|
|
if err == nil {
|
|
size = resp.ContentLength
|
|
resp.Body.Close()
|
|
}
|
|
}
|
|
resources = append(resources, parser.Resource{
|
|
URL: fmt.Sprintf("%s/data%s", *attachment.Server, *attachment.Path),
|
|
Filename: *attachment.Name,
|
|
Size: size,
|
|
})
|
|
}
|
|
picCdnMap := make(map[string]string)
|
|
for _, preview := range postInfo.Previews {
|
|
if preview.Type == nil || *preview.Type != "thumbnail" {
|
|
continue
|
|
}
|
|
if preview.Path == nil || preview.Server == nil {
|
|
log.FromContext(ctx).Warnf("Skipping kemono preview with missing path or server: post %s", info.PostID)
|
|
continue
|
|
}
|
|
picCdnMap[*preview.Path] = *preview.Server
|
|
}
|
|
for _, attachment := range postInfo.Post.Attachments {
|
|
if attachment.Path == nil || attachment.Name == nil {
|
|
log.FromContext(ctx).Warnf("Skipping kemono post attachment with missing path or name: post %s", info.PostID)
|
|
continue
|
|
}
|
|
if !isImageExt(*attachment.Path) {
|
|
continue
|
|
}
|
|
picUrl, err := url.JoinPath(picCdnMap[*attachment.Path], "data", *attachment.Path)
|
|
if err != nil {
|
|
continue
|
|
}
|
|
var size int64
|
|
headReq, err := http.NewRequestWithContext(ctx, http.MethodHead, picUrl, nil)
|
|
if err == nil {
|
|
resp, err := client.Do(headReq)
|
|
if err == nil {
|
|
size = resp.ContentLength
|
|
resp.Body.Close()
|
|
}
|
|
}
|
|
resources = append(resources, parser.Resource{
|
|
URL: picUrl,
|
|
Filename: *attachment.Name,
|
|
Size: size,
|
|
})
|
|
}
|
|
item.Resources = resources
|
|
return item, nil
|
|
}
|
|
|
|
func (k *KemonoParser) parseUserPage(_ context.Context, _ *DownloadInfo) (*parser.Item, error) {
|
|
return nil, errors.New("kemono user page not implemented")
|
|
}
|
|
|
|
func isImageExt(attachmentPath string) bool {
|
|
return strutil.HasSuffixAny(path.Ext(strings.Split(attachmentPath, "?")[0]), []string{".jpg", ".jpeg", ".png", ".webp"})
|
|
}
|