diff --git a/internal/handlers/core/full_text.go b/internal/handlers/core/full_text.go index 17e5efb8..ad7463a3 100644 --- a/internal/handlers/core/full_text.go +++ b/internal/handlers/core/full_text.go @@ -4,7 +4,6 @@ import ( "bytes" "context" "fmt" - "html" "io" "net/http" "net/url" @@ -121,9 +120,7 @@ func (h *Handler) FetchFullArticleContentContext(ctx context.Context, articleURL } extracted, extractErr := readability.FromReader(strings.NewReader(page), base) var output bytes.Buffer - leadImageURL := "" if extractErr == nil { - leadImageURL = extracted.ImageURL() extractErr = extracted.RenderHTML(&output) } content := output.String() @@ -131,12 +128,6 @@ func (h *Handler) FetchFullArticleContentContext(ctx context.Context, articleURL // Explicit semantic article containers are a useful fallback for short pages. content, _ = doc.Find("article,main,[role=main]").First().Html() } - if !strings.Contains(strings.ToLower(content), "

` + content - } - } content = textutil.PrepareArticleContent(content, base.String()) if strings.TrimSpace(content) == "" { return "", fmt.Errorf("no readable article content") diff --git a/internal/handlers/core/full_text_test.go b/internal/handlers/core/full_text_test.go index b57be4f3..61284af1 100644 --- a/internal/handlers/core/full_text_test.go +++ b/internal/handlers/core/full_text_test.go @@ -85,36 +85,6 @@ func TestFullTextReadabilityAndInvalidResponses(t *testing.T) { } } -func TestFullTextReadabilityRestoresLeadImageWithoutDuplicates(t *testing.T) { - h := fullTextHandler(t) - server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - w.Header().Set("Content-Type", "text/html; charset=utf-8") - body := strings.Repeat("A long article sentence, with useful detail and punctuation. ", 30) - if r.URL.Path == "/existing" { - fmt.Fprintf(w, `

Article

%s

`, body) - return - } - fmt.Fprintf(w, `

Article

%s

`, body) - })) - defer server.Close() - - content, err := h.FetchFullArticleContentContext(context.Background(), server.URL+"/lead", nil) - if err != nil { - t.Fatal(err) - } - if !strings.Contains(content, `src="`+server.URL+`/lead.jpg"`) || !strings.Contains(content, `referrerpolicy="no-referrer"`) { - t.Fatalf("lead image was not restored safely: %s", content) - } - - content, err = h.FetchFullArticleContentContext(context.Background(), server.URL+"/existing", nil) - if err != nil { - t.Fatal(err) - } - if !strings.Contains(content, server.URL+"/body.jpg") || strings.Contains(content, server.URL+"/og.jpg") { - t.Fatalf("existing article image was duplicated or replaced: %s", content) - } -} - func TestFullTextNoSelectorMatchIsAnError(t *testing.T) { h := fullTextHandler(t) id, err := h.DB.AddFeed(&models.Feed{Title: "f", URL: "https://example.org"}) diff --git a/internal/handlers/media/media_handlers_test.go b/internal/handlers/media/media_handlers_test.go index 67231721..afbeb028 100644 --- a/internal/handlers/media/media_handlers_test.go +++ b/internal/handlers/media/media_handlers_test.go @@ -1,11 +1,8 @@ package media import ( - "encoding/base64" "net/http" "net/http/httptest" - "regexp" - "strings" "testing" "MrRSS/internal/database" @@ -180,55 +177,6 @@ func TestProxyImagesInHTML_RelativeURLs(t *testing.T) { } } -func TestRewriteHTMLContent_ResponsiveImageCandidates(t *testing.T) { - baseURL := "https://example.com/news/article" - htmlContent := ` - - -` - - result := string(rewriteHTMLContent([]byte(htmlContent), baseURL)) - for _, descriptor := range []string{" 1x", " 2x", " 320w", " 1280w"} { - if !strings.Contains(result, descriptor) { - t.Errorf("missing srcset descriptor %q in %s", descriptor, result) - } - } - if strings.Contains(result, `srcSet="/_next/image`) || strings.Contains(result, `data-srcset="images/`) { - t.Fatalf("responsive image candidates were not proxied: %s", result) - } - - encodedURLs := regexp.MustCompile(`url_b64=([A-Za-z0-9+/=]+)`).FindAllStringSubmatch(result, -1) - var decodedURLs []string - for _, match := range encodedURLs { - decoded, err := base64.StdEncoding.DecodeString(match[1]) - if err != nil { - t.Fatalf("decode proxied URL: %v", err) - } - decodedURLs = append(decodedURLs, string(decoded)) - } - joinedURLs := strings.Join(decodedURLs, "\n") - for _, expected := range []string{ - "https://example.com/_next/image?url=%2Fhero.jpg&w=1280&q=75", - "https://cdn.example.com/hero.jpg", - "https://example.com/news/images/small.jpg", - "https://example.com/news/images/large.jpg", - } { - if !strings.Contains(joinedURLs, expected) { - t.Errorf("missing decoded candidate %q in %s", expected, joinedURLs) - } - } - if strings.Contains(joinedURLs, "&") { - t.Fatalf("HTML entities leaked into proxied URLs: %s", joinedURLs) - } -} - -func TestRewriteSrcsetAttribute_SkipsNonHTTPAndProxiedCandidates(t *testing.T) { - content := `` - if got := rewriteSrcsetAttribute(content, "img", "srcset", "https://example.com/article"); got != content { - t.Fatalf("special srcset candidates changed:\nwant: %s\n got: %s", content, got) - } -} - func contains(s, substr string) bool { return len(s) >= len(substr) && (s == substr || len(substr) == 0 || (len(s) > 0 && findInString(s, substr))) diff --git a/internal/handlers/media/media_proxy.go b/internal/handlers/media/media_proxy.go index b001e666..ec5a6c3f 100644 --- a/internal/handlers/media/media_proxy.go +++ b/internal/handlers/media/media_proxy.go @@ -823,8 +823,6 @@ func rewriteHTMLContent(bodyBytes []byte, baseURL string) []byte { // Then rewrite img src attributes (now including the converted lazy images) content = rewriteAttribute(content, "img", "src", baseURL) - content = rewriteSrcsetAttribute(content, "img", "srcset", baseURL) - content = rewriteSrcsetAttribute(content, "img", "data-srcset", baseURL) // Rewrite iframe src attributes content = rewriteAttribute(content, "iframe", "src", baseURL) @@ -838,8 +836,6 @@ func rewriteHTMLContent(bodyBytes []byte, baseURL string) []byte { // Rewrite source src attributes (for video/audio) content = rewriteAttribute(content, "source", "src", baseURL) - content = rewriteSrcsetAttribute(content, "source", "srcset", baseURL) - content = rewriteSrcsetAttribute(content, "source", "data-srcset", baseURL) // Rewrite track src attributes content = rewriteAttribute(content, "track", "src", baseURL) @@ -1059,101 +1055,89 @@ func parseHTMLAttributes(tag string) []htmlAttribute { return attrs } -// rewriteAttribute rewrites a specific URL attribute in HTML tags. +// rewriteAttribute rewrites a specific attribute in HTML tags func rewriteAttribute(content, tag, attr, baseURL string) string { - return rewriteAttributeValue(content, tag, attr, func(value string) (string, bool) { - return proxyWebpageResourceURL(value, baseURL) - }) -} + // Match all tags first + tagRe := regexp.MustCompile(fmt.Sprintf(`<%s[^>]*>`, tag)) -// rewriteSrcsetAttribute proxies every candidate URL while preserving its -// density or width descriptor (for example, 2x or 640w). -func rewriteSrcsetAttribute(content, tag, attr, baseURL string) string { - return rewriteAttributeValue(content, tag, attr, func(value string) (string, bool) { - var result strings.Builder - changed := false - for position := 0; position < len(value); { - prefixStart := position - for position < len(value) && (isHTMLSpace(value[position]) || value[position] == ',') { - position++ - } - result.WriteString(value[prefixStart:position]) - if position >= len(value) { - break - } + matchCount := 0 + rewriteCount := 0 - urlStart := position - isDataURL := strings.HasPrefix(strings.ToLower(value[position:]), "data:") - for position < len(value) && !isHTMLSpace(value[position]) && (isDataURL || value[position] != ',') { - position++ - } - candidate := value[urlStart:position] - if proxied, ok := proxyWebpageResourceURL(candidate, baseURL); ok { - result.WriteString(proxied) - changed = true + result := tagRe.ReplaceAllStringFunc(content, func(match string) string { + matchCount++ + // Try to find the attribute with double quotes + doubleQuoteRe := regexp.MustCompile(fmt.Sprintf(`\s%s\s*=\s*"([^"]*)"`, attr)) + doubleQuoteMatch := doubleQuoteRe.FindStringSubmatch(match) + + var urlValue, quote string + + if len(doubleQuoteMatch) >= 2 { + // Found double-quoted attribute + urlValue = doubleQuoteMatch[1] + quote = `"` + } else { + // Try single quotes + singleQuoteRe := regexp.MustCompile(fmt.Sprintf(`\s%s\s*=\s*'([^']*)'`, attr)) + singleQuoteMatch := singleQuoteRe.FindStringSubmatch(match) + if len(singleQuoteMatch) >= 2 { + urlValue = singleQuoteMatch[1] + quote = `'` } else { - result.WriteString(candidate) + // Try unquoted + unquotedRe := regexp.MustCompile(fmt.Sprintf(`\s%s\s*=\s*([^\s>]+)`, attr)) + unquotedMatch := unquotedRe.FindStringSubmatch(match) + if len(unquotedMatch) >= 2 { + urlValue = unquotedMatch[1] + quote = "" + } else { + // Attribute not found + return match + } } + } - descriptorStart := position - for position < len(value) && value[position] != ',' { - position++ - } - result.WriteString(value[descriptorStart:position]) + // Skip data: URLs, blob: URLs, and already proxied URLs + if strings.HasPrefix(urlValue, "data:") || + strings.HasPrefix(urlValue, "blob:") || + strings.HasPrefix(urlValue, "/api/") || + strings.HasPrefix(urlValue, "#") { + return match } - return result.String(), changed - }) -} -func rewriteAttributeValue(content, tag, attr string, rewrite func(string) (string, bool)) string { - tagRe := regexp.MustCompile(`(?i)<` + regexp.QuoteMeta(tag) + `\b[^>]*>`) - patterns := []*regexp.Regexp{ - regexp.MustCompile(`(?i)\s+` + regexp.QuoteMeta(attr) + `\s*=\s*"([^"]*)"`), - regexp.MustCompile(`(?i)\s+` + regexp.QuoteMeta(attr) + `\s*=\s*'([^']*)'`), - regexp.MustCompile(`(?i)\s+` + regexp.QuoteMeta(attr) + `\s*=\s*([^\s>]+)`), - } + rewriteCount++ + // if tag == "script" || tag == "link" { + // log.Printf("[%s Rewrite] Rewriting %s %d: %s", strings.ToUpper(tag), attr, rewriteCount, urlValue) + // } - return tagRe.ReplaceAllStringFunc(content, func(match string) string { - for index, pattern := range patterns { - location := pattern.FindStringSubmatchIndex(match) - if len(location) < 4 { - continue - } - valueStart, valueEnd := location[2], location[3] - rewritten, changed := rewrite(match[valueStart:valueEnd]) - if !changed { - return match - } - if index == len(patterns)-1 { - rewritten = `"` + rewritten + `"` - } - return match[:valueStart] + rewritten + match[valueEnd:] + // Resolve relative URLs + resolvedURL := resolveURL(urlValue, baseURL) + + // Create proxied URL with base64 encoding + proxiedURL := fmt.Sprintf("/api/webpage/resource?url_b64=%s&referer_b64=%s", + base64.StdEncoding.EncodeToString([]byte(resolvedURL)), + base64.StdEncoding.EncodeToString([]byte(baseURL))) + + // Replace the URL in the match + // Use regex to replace attribute value more reliably + if quote != "" { + // Quoted value - replace using regex for more flexibility + attrPattern := regexp.MustCompile(`(` + attr + `)\s*=\s*` + regexp.QuoteMeta(quote) + regexp.QuoteMeta(urlValue) + regexp.QuoteMeta(quote)) + replacement := fmt.Sprintf(`%s=%s%s%s`, attr, quote, proxiedURL, quote) + return attrPattern.ReplaceAllString(match, replacement) + } else { + // Unquoted value - match until whitespace or > character + // We need to capture the delimiter (space or >) to preserve it + attrPattern := regexp.MustCompile(`(` + attr + `)\s*=\s*` + regexp.QuoteMeta(urlValue) + `([\s>])`) + replacement := fmt.Sprintf(`%s="%s"$2`, attr, proxiedURL) + return attrPattern.ReplaceAllString(match, replacement) } - return match }) -} - -func proxyWebpageResourceURL(value, baseURL string) (string, bool) { - value = strings.TrimSpace(html.UnescapeString(value)) - lowerValue := strings.ToLower(value) - if value == "" || strings.HasPrefix(lowerValue, "data:") || - strings.HasPrefix(lowerValue, "blob:") || strings.HasPrefix(value, "#") || - strings.HasPrefix(value, "/api/") || strings.Contains(value, "/api/webpage/resource?") { - return value, false - } - resolvedURL := resolveURL(value, baseURL) - parsedURL, err := url.Parse(resolvedURL) - if err != nil || (parsedURL.Scheme != "http" && parsedURL.Scheme != "https") { - return value, false - } - return fmt.Sprintf("/api/webpage/resource?url_b64=%s&referer_b64=%s", - base64.StdEncoding.EncodeToString([]byte(resolvedURL)), - base64.StdEncoding.EncodeToString([]byte(baseURL))), true -} + // if matchCount > 0 && (tag == "script" || tag == "link") { + // log.Printf("[%s Rewrite] Found %d %s tags, rewrote %d %s attributes", strings.ToUpper(tag), matchCount, tag, rewriteCount, attr) + // } -func isHTMLSpace(value byte) bool { - return value == ' ' || value == '\t' || value == '\n' || value == '\r' || value == '\f' + return result } // rewriteLinkHref rewrites href attributes in link tags