mirror of
https://github.com/krau/SaveAny-Bot.git
synced 2026-08-30 20:56:42 +08:00
* fix: prevent queue deadlock after cancelling tasks Get no longer recurses while holding the mutex. Cancelled queued tasks leave the map, making their IDs reusable. Closed empty queues return ErrQueueClosed. * fix: init task queue lazily and close on shutdown AddTask is safe before Run by initializing the queue once. Close unblocks workers waiting in Get. * fix: make resource fingerprints deterministic Sort map keys before hashing so Resource.ID is stable. * fix: harden per-item processing tracking in tasks Use the resource fingerprint as the dedup key on insert and delete. Check and set processing entries atomically instead of TOCTOU. Count only successful downloads. * fix: honor IgnoreErrors in batch tasks Element failures no longer cancel sibling elements or stop later groups. * fix: report streamed upload bytes in batch tasks Stream downloads report their byte count as the upload total. * fix: validate i18n key parity across locales geni18n fails when a language file misses any key. Align the syncpeers completion key between en and zh-Hans. Translate three untranslated parse keys in zh-Hans. * refactor: share progress throttling helpers Move size-tiered and count-based throttling into progressutil. Localize hardcoded Chinese progress strings. Drop five duplicated implementations and dead local copies. * refactor: share unique filename logic Storage backends use fsutil.UniquePath instead of local loops. * fix: sanitize local storage paths Reject absolute paths and dot-dot escapes in Save. Check close errors and wrap creation failures. * fix: preserve webdav error causes Wrap mkdir and write failures with %w and drop dead error values. * fix: kill rclone subprocess on reader close Prevent cat processes from hanging after the pipe is closed. * fix: fail alist init instead of exiting Replace log.Fatalf with wrapped errors so a bad alist cannot kill the bot. Cancel token refresh with the init context and re-login on 401. * fix: enforce telegram album limits and track saved paths Use tglimit.MaxAlbumItems for album batching and video splitting. Exists now reports previously saved paths instead of always false. * fix: guard storage registry maps Protect Storages and UserStorages with mutexes and expose read accessors. * fix: harden JS parser plugin runtime Validate semver instead of panicking and require canHandle. Recover plugin worker panics and time out CanHandle calls. Return a copy from the registry and sanitize install filenames. * fix: guard kemono parser against nil fields Skip sparse preview and attachment entries instead of panicking. * refactor: remove commented-out dead code Drop unused kemono legacy types and commented response structs. * test: cover invalid plugin version rejection * fix: show all queued tasks in /task list Render up to ten tasks and append the truncation note once. * fix: guard callback data parsing Reject malformed callback payloads before indexing split parts. * fix: isolate media groups per user Key pending groups by chat, user and group id. * fix: require permission for callback handlers * fix: initialize userbot context once Replace the racy lazy init with sync.OnceValue. * fix: avoid leaking raw errors in /dir reply * fix: notify users on invalid update version * fix: fail fast when API listen fails Bind synchronously and surface errors instead of logging them. * fix: add timeouts and backoff to webhook delivery * refactor: remove dead code from api and bot Drop the empty ProgressTracker shim, unused token context key and a redundant SetBotCommands call. * fix: load remote config without local lookup Skip the local file search after reading a config URL and add a timeout. * refactor: drop unused hook config * docs: document parser plugin config * ci: fix BuildTime formatting and align checkout Actions format does not format dates; pass the raw timestamp. * chore: ignore cache directory * fix: make cache init idempotent * fix: upload only downloaded batch elements Failed elements keep partial cache files and never reach the backend. Successfully downloaded siblings still upload when one element fails. * fix: record telegram saved paths only after upload Skip-large returns a sentinel so skipped files are not marked as saved. * fix: deduplicate alist token refreshes Guard token access with a mutex and merge concurrent logins. Reuse a recent refresh to avoid login storms. * test: cover concurrent alist 401 retry Ten parallel uploads share a single re-login under -race. * fix: count parsed resources in progress text * fix: localize storage lookup errors in /dir Use the shared i18n key and escape the dynamic error. * fix: deduplicate concurrent storage initialization singleflight merges first-time inits so side effects are not duplicated. * fix: guard nil progress trackers in api-created tasks Batch, telegraph and transfer tasks run without a Telegram tracker when created through the API; their callbacks must not panic. * fix: keep album order when filtering failed batch items Download results are stored by original index so surviving elements keep their source order. * test: cover nil-tracker task execution * fix: report partial failure in batch done message IgnoreErrors runs with failed elements show success and failed counts instead of claiming every file completed. * fix: skip login refresh for token-only alist storage 401 responses surface the auth error instead of sending a credential-less login; the refresh window never exceeds TokenExp. * test: cover alist refresh semantics Concurrent refresh uses username/password; token-only storage never attempts a login on 401. * fix: call tracker from notifyProgress instead of recursing The helper called itself, overflowing the stack on any batch task with a progress tracker. * fix: propagate cancellation past IgnoreErrors Cancelled tasks must not be reported as successful: only ordinary element failures are ignored. * test: cover notifyProgress tracker call * test: cover cancellation with IgnoreErrors * fix: refresh alist token after startup 401s The init login no longer satisfies the dedup window, so an early 401 triggers a real refresh; later 401s reuse it. Clarify that streaming uploads cannot replay their body. * test: exercise the alist 401 refresh path Concurrent uploads now reject the init token, share one refresh and retry with the new token; token-only stays inert. * style: trim verbose comments Drop process-style explanations; keep one-line behavior notes. * style: gofmt test files * fix: drop unbounded saved-path cache in telegram storage Telegram cannot reliably query remote file existence, so the cache answered a question it could not answer and grew without bound. Exists returns false again, as before. * style: format codes
210 lines
5.9 KiB
Go
210 lines
5.9 KiB
Go
package directlinks
|
|
|
|
import (
|
|
"mime"
|
|
"net/url"
|
|
"strings"
|
|
"unicode/utf8"
|
|
|
|
"golang.org/x/text/encoding/simplifiedchinese"
|
|
)
|
|
|
|
// parseFilename extracts filename from Content-Disposition header
|
|
// It handles multiple encoding scenarios:
|
|
// 1. RFC 5987/RFC 2231 format: filename*=UTF-8”%E6%B5%8B%E8%AF%95.zip (preferred, checked first)
|
|
// 2. MIME encoded-word: filename="=?UTF-8?B?5rWL6K+VLnppcA==?="
|
|
// 3. URL-encoded: filename="%E6%B5%8B%E8%AF%95.zip"
|
|
// 4. Plain ASCII filename
|
|
//
|
|
// The key fix is checking filename*= first before mime.ParseMediaType, because
|
|
// some servers send Content-Disposition headers with invalid characters that cause
|
|
// mime.ParseMediaType to fail, but the filename*= parameter is still valid.
|
|
func parseFilename(contentDisposition string) string {
|
|
// First, try to find filename*= (RFC 5987 format, most reliable for non-ASCII)
|
|
if filename := parseFilenameExtended(contentDisposition); filename != "" {
|
|
return filename
|
|
}
|
|
|
|
// Try standard MIME parsing for regular filename= parameter
|
|
_, params, err := mime.ParseMediaType(contentDisposition)
|
|
if err == nil {
|
|
if filename := params["filename"]; filename != "" {
|
|
return decodeFilenameParam(filename)
|
|
}
|
|
}
|
|
|
|
// Fallback: manual parsing if mime.ParseMediaType fails
|
|
return parseFilenameFallback(contentDisposition)
|
|
}
|
|
|
|
// parseFilenameExtended parses RFC 5987/RFC 2231 extended parameter format
|
|
// Format: filename*=charset'language'value (e.g., UTF-8”%E6%B5%8B%E8%AF%95.zip)
|
|
func parseFilenameExtended(cd string) string {
|
|
// Look for filename*= (case-insensitive)
|
|
lower := strings.ToLower(cd)
|
|
idx := strings.Index(lower, "filename*=")
|
|
if idx == -1 {
|
|
return ""
|
|
}
|
|
|
|
// Extract the value after filename*=
|
|
value := cd[idx+len("filename*="):]
|
|
|
|
// Find the end of the value (next ; or end of string)
|
|
if endIdx := strings.Index(value, ";"); endIdx != -1 {
|
|
value = value[:endIdx]
|
|
}
|
|
value = strings.TrimSpace(value)
|
|
|
|
// Parse charset'language'encoded-value format
|
|
// Common format: UTF-8''%E6%B5%8B%E8%AF%95.zip
|
|
parts := strings.SplitN(value, "''", 2)
|
|
if len(parts) == 2 {
|
|
// parts[0] is charset (e.g., "UTF-8")
|
|
// parts[1] is percent-encoded value
|
|
decoded, err := url.QueryUnescape(parts[1])
|
|
if err == nil {
|
|
return decoded
|
|
}
|
|
}
|
|
|
|
// Try with single quote delimiter as well (some servers use this)
|
|
parts = strings.SplitN(value, "'", 3)
|
|
if len(parts) >= 3 {
|
|
decoded, err := url.QueryUnescape(parts[2])
|
|
if err == nil {
|
|
return decoded
|
|
}
|
|
}
|
|
|
|
return ""
|
|
}
|
|
|
|
// TryUrlQueryUnescape tries to unescape a URL-encoded string.
|
|
//
|
|
// If unescaping fails, it returns the original string.
|
|
func tryUrlQueryUnescape(s string) string {
|
|
if decoded, err := url.QueryUnescape(s); err == nil {
|
|
return decoded
|
|
}
|
|
return s
|
|
}
|
|
|
|
// decodeFilenameParam decodes a filename parameter value
|
|
// Handles MIME encoded-word, URL encoding, and GBK encoding fallback
|
|
func decodeFilenameParam(filename string) string {
|
|
// Check if the filename is MIME encoded-word (e.g., =?UTF-8?B?...?=)
|
|
if strings.HasPrefix(filename, "=?") {
|
|
decoder := new(mime.WordDecoder)
|
|
// Some servers use "UTF8" instead of "UTF-8", create a normalized copy
|
|
normalizedFilename := strings.Replace(filename, "UTF8", "UTF-8", 1)
|
|
if decoded, err := decoder.Decode(normalizedFilename); err == nil {
|
|
return decoded
|
|
}
|
|
}
|
|
|
|
// Try URL decoding
|
|
decoded := tryUrlQueryUnescape(filename)
|
|
|
|
// Check if the result is valid UTF-8. If not, try GBK decoding.
|
|
// This handles the case where Chinese Windows servers send GBK-encoded filenames
|
|
// which appear as garbled characters (e.g., "下载地址.zip" -> "���ص�ַ.zip")
|
|
if !utf8.ValidString(decoded) {
|
|
if gbkDecoded := tryDecodeGBK(decoded); gbkDecoded != "" {
|
|
return gbkDecoded
|
|
}
|
|
}
|
|
|
|
return decoded
|
|
}
|
|
|
|
// gbkDecoder is a reusable GBK decoder for better performance
|
|
var gbkDecoder = simplifiedchinese.GBK.NewDecoder()
|
|
|
|
// tryDecodeGBK attempts to decode a string as GBK/GB2312/GB18030 encoding
|
|
// Returns empty string if decoding fails or result is not valid UTF-8
|
|
func tryDecodeGBK(s string) string {
|
|
// GBK uses 1-2 bytes per character. Single-byte chars are 0x00-0x7F (ASCII compatible).
|
|
// Double-byte chars have first byte 0x81-0xFE and second byte 0x40-0xFE.
|
|
// Skip if string is empty or all ASCII (valid UTF-8)
|
|
if len(s) == 0 {
|
|
return ""
|
|
}
|
|
|
|
// Create a fresh decoder since the transform state may be corrupted
|
|
decoder := gbkDecoder
|
|
decoded, err := decoder.Bytes([]byte(s))
|
|
if err != nil {
|
|
return ""
|
|
}
|
|
result := string(decoded)
|
|
if utf8.ValidString(result) {
|
|
return result
|
|
}
|
|
return ""
|
|
}
|
|
|
|
// parseFilenameFromURL extracts filename from URL path
|
|
// This is used as a fallback when Content-Disposition is not available
|
|
func parseFilenameFromURL(rawURL string) string {
|
|
parsed, err := url.Parse(rawURL)
|
|
if err != nil {
|
|
return ""
|
|
}
|
|
|
|
// Get the path part and extract the last segment
|
|
path := parsed.Path
|
|
if path == "" {
|
|
return ""
|
|
}
|
|
|
|
// URL decode the path first
|
|
decodedPath, err := url.PathUnescape(path)
|
|
if err != nil {
|
|
decodedPath = path
|
|
}
|
|
|
|
// Get the last segment of the path
|
|
lastSlash := strings.LastIndex(decodedPath, "/")
|
|
if lastSlash == -1 {
|
|
return decodedPath
|
|
}
|
|
filename := decodedPath[lastSlash+1:]
|
|
|
|
// Remove query string if somehow still present
|
|
if idx := strings.Index(filename, "?"); idx != -1 {
|
|
filename = filename[:idx]
|
|
}
|
|
|
|
return filename
|
|
}
|
|
|
|
// parseFilenameFallback manually parses filename= when mime.ParseMediaType fails
|
|
func parseFilenameFallback(cd string) string {
|
|
// Look for filename= (case-insensitive)
|
|
lower := strings.ToLower(cd)
|
|
idx := strings.Index(lower, "filename=")
|
|
if idx == -1 {
|
|
return ""
|
|
}
|
|
|
|
// Skip "filename=" prefix
|
|
value := cd[idx+len("filename="):]
|
|
|
|
// Find the end of the value
|
|
if endIdx := strings.Index(value, ";"); endIdx != -1 {
|
|
value = value[:endIdx]
|
|
}
|
|
value = strings.TrimSpace(value)
|
|
|
|
// Remove quotes if present
|
|
if len(value) >= 2 {
|
|
if (value[0] == '"' && value[len(value)-1] == '"') ||
|
|
(value[0] == '\'' && value[len(value)-1] == '\'') {
|
|
value = value[1 : len(value)-1]
|
|
}
|
|
}
|
|
|
|
return decodeFilenameParam(value)
|
|
}
|