From 545752ad6547511578677f134eb9e69fdc37c6dc Mon Sep 17 00:00:00 2001
From: marcomarcogd <35049765+marcomarcogd@users.noreply.github.com>
Date: Tue, 29 Sep 2026 14:15:24 +0800
Subject: [PATCH] =?UTF-8?q?revert=EF=BC=9A=E5=B0=86=20#1231=20=E5=9B=BE?=
=?UTF-8?q?=E7=89=87=E4=BF=AE=E5=A4=8D=E8=BF=81=E8=87=B3=E5=8F=91=E5=B8=83?=
=?UTF-8?q?=E5=88=86=E6=94=AF?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
---
internal/handlers/core/full_text.go | 9 -
internal/handlers/core/full_text_test.go | 30 ----
.../handlers/media/media_handlers_test.go | 52 ------
internal/handlers/media/media_proxy.go | 156 ++++++++----------
4 files changed, 70 insertions(+), 177 deletions(-)
diff --git a/internal/handlers/core/full_text.go b/internal/handlers/core/full_text.go
index 17e5efb88..ad7463a33 100644
--- a/internal/handlers/core/full_text.go
+++ b/internal/handlers/core/full_text.go
@@ -4,7 +4,6 @@ import (
"bytes"
"context"
"fmt"
- "html"
"io"
"net/http"
"net/url"
@@ -121,9 +120,7 @@ func (h *Handler) FetchFullArticleContentContext(ctx context.Context, articleURL
}
extracted, extractErr := readability.FromReader(strings.NewReader(page), base)
var output bytes.Buffer
- leadImageURL := ""
if extractErr == nil {
- leadImageURL = extracted.ImageURL()
extractErr = extracted.RenderHTML(&output)
}
content := output.String()
@@ -131,12 +128,6 @@ func (h *Handler) FetchFullArticleContentContext(ctx context.Context, articleURL
// Explicit semantic article containers are a useful fallback for short pages.
content, _ = doc.Find("article,main,[role=main]").First().Html()
}
- if !strings.Contains(strings.ToLower(content), "![]()
) + `)
` + content
- }
- }
content = textutil.PrepareArticleContent(content, base.String())
if strings.TrimSpace(content) == "" {
return "", fmt.Errorf("no readable article content")
diff --git a/internal/handlers/core/full_text_test.go b/internal/handlers/core/full_text_test.go
index b57be4f3b..61284af1c 100644
--- a/internal/handlers/core/full_text_test.go
+++ b/internal/handlers/core/full_text_test.go
@@ -85,36 +85,6 @@ func TestFullTextReadabilityAndInvalidResponses(t *testing.T) {
}
}
-func TestFullTextReadabilityRestoresLeadImageWithoutDuplicates(t *testing.T) {
- h := fullTextHandler(t)
- server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
- w.Header().Set("Content-Type", "text/html; charset=utf-8")
- body := strings.Repeat("A long article sentence, with useful detail and punctuation. ", 30)
- if r.URL.Path == "/existing" {
- fmt.Fprintf(w, `Article
%s
`, body)
- return
- }
- fmt.Fprintf(w, `Article
%s
`, body)
- }))
- defer server.Close()
-
- content, err := h.FetchFullArticleContentContext(context.Background(), server.URL+"/lead", nil)
- if err != nil {
- t.Fatal(err)
- }
- if !strings.Contains(content, `src="`+server.URL+`/lead.jpg"`) || !strings.Contains(content, `referrerpolicy="no-referrer"`) {
- t.Fatalf("lead image was not restored safely: %s", content)
- }
-
- content, err = h.FetchFullArticleContentContext(context.Background(), server.URL+"/existing", nil)
- if err != nil {
- t.Fatal(err)
- }
- if !strings.Contains(content, server.URL+"/body.jpg") || strings.Contains(content, server.URL+"/og.jpg") {
- t.Fatalf("existing article image was duplicated or replaced: %s", content)
- }
-}
-
func TestFullTextNoSelectorMatchIsAnError(t *testing.T) {
h := fullTextHandler(t)
id, err := h.DB.AddFeed(&models.Feed{Title: "f", URL: "https://example.org"})
diff --git a/internal/handlers/media/media_handlers_test.go b/internal/handlers/media/media_handlers_test.go
index 672317217..afbeb0280 100644
--- a/internal/handlers/media/media_handlers_test.go
+++ b/internal/handlers/media/media_handlers_test.go
@@ -1,11 +1,8 @@
package media
import (
- "encoding/base64"
"net/http"
"net/http/httptest"
- "regexp"
- "strings"
"testing"
"MrRSS/internal/database"
@@ -180,55 +177,6 @@ func TestProxyImagesInHTML_RelativeURLs(t *testing.T) {
}
}
-func TestRewriteHTMLContent_ResponsiveImageCandidates(t *testing.T) {
- baseURL := "https://example.com/news/article"
- htmlContent := `
-
-
-`
-
- result := string(rewriteHTMLContent([]byte(htmlContent), baseURL))
- for _, descriptor := range []string{" 1x", " 2x", " 320w", " 1280w"} {
- if !strings.Contains(result, descriptor) {
- t.Errorf("missing srcset descriptor %q in %s", descriptor, result)
- }
- }
- if strings.Contains(result, `srcSet="/_next/image`) || strings.Contains(result, `data-srcset="images/`) {
- t.Fatalf("responsive image candidates were not proxied: %s", result)
- }
-
- encodedURLs := regexp.MustCompile(`url_b64=([A-Za-z0-9+/=]+)`).FindAllStringSubmatch(result, -1)
- var decodedURLs []string
- for _, match := range encodedURLs {
- decoded, err := base64.StdEncoding.DecodeString(match[1])
- if err != nil {
- t.Fatalf("decode proxied URL: %v", err)
- }
- decodedURLs = append(decodedURLs, string(decoded))
- }
- joinedURLs := strings.Join(decodedURLs, "\n")
- for _, expected := range []string{
- "https://example.com/_next/image?url=%2Fhero.jpg&w=1280&q=75",
- "https://cdn.example.com/hero.jpg",
- "https://example.com/news/images/small.jpg",
- "https://example.com/news/images/large.jpg",
- } {
- if !strings.Contains(joinedURLs, expected) {
- t.Errorf("missing decoded candidate %q in %s", expected, joinedURLs)
- }
- }
- if strings.Contains(joinedURLs, "&") {
- t.Fatalf("HTML entities leaked into proxied URLs: %s", joinedURLs)
- }
-}
-
-func TestRewriteSrcsetAttribute_SkipsNonHTTPAndProxiedCandidates(t *testing.T) {
- content := `
`
- if got := rewriteSrcsetAttribute(content, "img", "srcset", "https://example.com/article"); got != content {
- t.Fatalf("special srcset candidates changed:\nwant: %s\n got: %s", content, got)
- }
-}
-
func contains(s, substr string) bool {
return len(s) >= len(substr) && (s == substr || len(substr) == 0 ||
(len(s) > 0 && findInString(s, substr)))
diff --git a/internal/handlers/media/media_proxy.go b/internal/handlers/media/media_proxy.go
index b001e666d..ec5a6c3f2 100644
--- a/internal/handlers/media/media_proxy.go
+++ b/internal/handlers/media/media_proxy.go
@@ -823,8 +823,6 @@ func rewriteHTMLContent(bodyBytes []byte, baseURL string) []byte {
// Then rewrite img src attributes (now including the converted lazy images)
content = rewriteAttribute(content, "img", "src", baseURL)
- content = rewriteSrcsetAttribute(content, "img", "srcset", baseURL)
- content = rewriteSrcsetAttribute(content, "img", "data-srcset", baseURL)
// Rewrite iframe src attributes
content = rewriteAttribute(content, "iframe", "src", baseURL)
@@ -838,8 +836,6 @@ func rewriteHTMLContent(bodyBytes []byte, baseURL string) []byte {
// Rewrite source src attributes (for video/audio)
content = rewriteAttribute(content, "source", "src", baseURL)
- content = rewriteSrcsetAttribute(content, "source", "srcset", baseURL)
- content = rewriteSrcsetAttribute(content, "source", "data-srcset", baseURL)
// Rewrite track src attributes
content = rewriteAttribute(content, "track", "src", baseURL)
@@ -1059,101 +1055,89 @@ func parseHTMLAttributes(tag string) []htmlAttribute {
return attrs
}
-// rewriteAttribute rewrites a specific URL attribute in HTML tags.
+// rewriteAttribute rewrites a specific attribute in HTML tags
func rewriteAttribute(content, tag, attr, baseURL string) string {
- return rewriteAttributeValue(content, tag, attr, func(value string) (string, bool) {
- return proxyWebpageResourceURL(value, baseURL)
- })
-}
+ // Match all tags first
+ tagRe := regexp.MustCompile(fmt.Sprintf(`<%s[^>]*>`, tag))
-// rewriteSrcsetAttribute proxies every candidate URL while preserving its
-// density or width descriptor (for example, 2x or 640w).
-func rewriteSrcsetAttribute(content, tag, attr, baseURL string) string {
- return rewriteAttributeValue(content, tag, attr, func(value string) (string, bool) {
- var result strings.Builder
- changed := false
- for position := 0; position < len(value); {
- prefixStart := position
- for position < len(value) && (isHTMLSpace(value[position]) || value[position] == ',') {
- position++
- }
- result.WriteString(value[prefixStart:position])
- if position >= len(value) {
- break
- }
+ matchCount := 0
+ rewriteCount := 0
- urlStart := position
- isDataURL := strings.HasPrefix(strings.ToLower(value[position:]), "data:")
- for position < len(value) && !isHTMLSpace(value[position]) && (isDataURL || value[position] != ',') {
- position++
- }
- candidate := value[urlStart:position]
- if proxied, ok := proxyWebpageResourceURL(candidate, baseURL); ok {
- result.WriteString(proxied)
- changed = true
+ result := tagRe.ReplaceAllStringFunc(content, func(match string) string {
+ matchCount++
+ // Try to find the attribute with double quotes
+ doubleQuoteRe := regexp.MustCompile(fmt.Sprintf(`\s%s\s*=\s*"([^"]*)"`, attr))
+ doubleQuoteMatch := doubleQuoteRe.FindStringSubmatch(match)
+
+ var urlValue, quote string
+
+ if len(doubleQuoteMatch) >= 2 {
+ // Found double-quoted attribute
+ urlValue = doubleQuoteMatch[1]
+ quote = `"`
+ } else {
+ // Try single quotes
+ singleQuoteRe := regexp.MustCompile(fmt.Sprintf(`\s%s\s*=\s*'([^']*)'`, attr))
+ singleQuoteMatch := singleQuoteRe.FindStringSubmatch(match)
+ if len(singleQuoteMatch) >= 2 {
+ urlValue = singleQuoteMatch[1]
+ quote = `'`
} else {
- result.WriteString(candidate)
+ // Try unquoted
+ unquotedRe := regexp.MustCompile(fmt.Sprintf(`\s%s\s*=\s*([^\s>]+)`, attr))
+ unquotedMatch := unquotedRe.FindStringSubmatch(match)
+ if len(unquotedMatch) >= 2 {
+ urlValue = unquotedMatch[1]
+ quote = ""
+ } else {
+ // Attribute not found
+ return match
+ }
}
+ }
- descriptorStart := position
- for position < len(value) && value[position] != ',' {
- position++
- }
- result.WriteString(value[descriptorStart:position])
+ // Skip data: URLs, blob: URLs, and already proxied URLs
+ if strings.HasPrefix(urlValue, "data:") ||
+ strings.HasPrefix(urlValue, "blob:") ||
+ strings.HasPrefix(urlValue, "/api/") ||
+ strings.HasPrefix(urlValue, "#") {
+ return match
}
- return result.String(), changed
- })
-}
-func rewriteAttributeValue(content, tag, attr string, rewrite func(string) (string, bool)) string {
- tagRe := regexp.MustCompile(`(?i)<` + regexp.QuoteMeta(tag) + `\b[^>]*>`)
- patterns := []*regexp.Regexp{
- regexp.MustCompile(`(?i)\s+` + regexp.QuoteMeta(attr) + `\s*=\s*"([^"]*)"`),
- regexp.MustCompile(`(?i)\s+` + regexp.QuoteMeta(attr) + `\s*=\s*'([^']*)'`),
- regexp.MustCompile(`(?i)\s+` + regexp.QuoteMeta(attr) + `\s*=\s*([^\s>]+)`),
- }
+ rewriteCount++
+ // if tag == "script" || tag == "link" {
+ // log.Printf("[%s Rewrite] Rewriting %s %d: %s", strings.ToUpper(tag), attr, rewriteCount, urlValue)
+ // }
- return tagRe.ReplaceAllStringFunc(content, func(match string) string {
- for index, pattern := range patterns {
- location := pattern.FindStringSubmatchIndex(match)
- if len(location) < 4 {
- continue
- }
- valueStart, valueEnd := location[2], location[3]
- rewritten, changed := rewrite(match[valueStart:valueEnd])
- if !changed {
- return match
- }
- if index == len(patterns)-1 {
- rewritten = `"` + rewritten + `"`
- }
- return match[:valueStart] + rewritten + match[valueEnd:]
+ // Resolve relative URLs
+ resolvedURL := resolveURL(urlValue, baseURL)
+
+ // Create proxied URL with base64 encoding
+ proxiedURL := fmt.Sprintf("/api/webpage/resource?url_b64=%s&referer_b64=%s",
+ base64.StdEncoding.EncodeToString([]byte(resolvedURL)),
+ base64.StdEncoding.EncodeToString([]byte(baseURL)))
+
+ // Replace the URL in the match
+ // Use regex to replace attribute value more reliably
+ if quote != "" {
+ // Quoted value - replace using regex for more flexibility
+ attrPattern := regexp.MustCompile(`(` + attr + `)\s*=\s*` + regexp.QuoteMeta(quote) + regexp.QuoteMeta(urlValue) + regexp.QuoteMeta(quote))
+ replacement := fmt.Sprintf(`%s=%s%s%s`, attr, quote, proxiedURL, quote)
+ return attrPattern.ReplaceAllString(match, replacement)
+ } else {
+ // Unquoted value - match until whitespace or > character
+ // We need to capture the delimiter (space or >) to preserve it
+ attrPattern := regexp.MustCompile(`(` + attr + `)\s*=\s*` + regexp.QuoteMeta(urlValue) + `([\s>])`)
+ replacement := fmt.Sprintf(`%s="%s"$2`, attr, proxiedURL)
+ return attrPattern.ReplaceAllString(match, replacement)
}
- return match
})
-}
-
-func proxyWebpageResourceURL(value, baseURL string) (string, bool) {
- value = strings.TrimSpace(html.UnescapeString(value))
- lowerValue := strings.ToLower(value)
- if value == "" || strings.HasPrefix(lowerValue, "data:") ||
- strings.HasPrefix(lowerValue, "blob:") || strings.HasPrefix(value, "#") ||
- strings.HasPrefix(value, "/api/") || strings.Contains(value, "/api/webpage/resource?") {
- return value, false
- }
- resolvedURL := resolveURL(value, baseURL)
- parsedURL, err := url.Parse(resolvedURL)
- if err != nil || (parsedURL.Scheme != "http" && parsedURL.Scheme != "https") {
- return value, false
- }
- return fmt.Sprintf("/api/webpage/resource?url_b64=%s&referer_b64=%s",
- base64.StdEncoding.EncodeToString([]byte(resolvedURL)),
- base64.StdEncoding.EncodeToString([]byte(baseURL))), true
-}
+ // if matchCount > 0 && (tag == "script" || tag == "link") {
+ // log.Printf("[%s Rewrite] Found %d %s tags, rewrote %d %s attributes", strings.ToUpper(tag), matchCount, tag, rewriteCount, attr)
+ // }
-func isHTMLSpace(value byte) bool {
- return value == ' ' || value == '\t' || value == '\n' || value == '\r' || value == '\f'
+ return result
}
// rewriteLinkHref rewrites href attributes in link tags