Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 0 additions & 9 deletions internal/handlers/core/full_text.go
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,6 @@ import (
"bytes"
"context"
"fmt"
"html"
"io"
"net/http"
"net/url"
Expand Down Expand Up @@ -121,22 +120,14 @@ func (h *Handler) FetchFullArticleContentContext(ctx context.Context, articleURL
}
extracted, extractErr := readability.FromReader(strings.NewReader(page), base)
var output bytes.Buffer
leadImageURL := ""
if extractErr == nil {
leadImageURL = extracted.ImageURL()
extractErr = extracted.RenderHTML(&output)
}
content := output.String()
if extractErr != nil || strings.TrimSpace(content) == "" {
// Explicit semantic article containers are a useful fallback for short pages.
content, _ = doc.Find("article,main,[role=main]").First().Html()
}
if !strings.Contains(strings.ToLower(content), "<img") {
if imageURL, err := base.Parse(html.UnescapeString(strings.TrimSpace(leadImageURL))); err == nil &&
(imageURL.Scheme == "http" || imageURL.Scheme == "https") {
content = `<p><img src="` + html.EscapeString(imageURL.String()) + `" alt=""></p>` + content
}
}
content = textutil.PrepareArticleContent(content, base.String())
if strings.TrimSpace(content) == "" {
return "", fmt.Errorf("no readable article content")
Expand Down
30 changes: 0 additions & 30 deletions internal/handlers/core/full_text_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -85,36 +85,6 @@ func TestFullTextReadabilityAndInvalidResponses(t *testing.T) {
}
}

func TestFullTextReadabilityRestoresLeadImageWithoutDuplicates(t *testing.T) {
h := fullTextHandler(t)
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.Header().Set("Content-Type", "text/html; charset=utf-8")
body := strings.Repeat("A long article sentence, with useful detail and punctuation. ", 30)
if r.URL.Path == "/existing" {
fmt.Fprintf(w, `<html><head><meta property="og:image" content="/og.jpg"></head><body><article><h1>Article</h1><p>%s</p><img src="/body.jpg"></article></body></html>`, body)
return
}
fmt.Fprintf(w, `<html><head><meta property="og:image" content="/lead.jpg"></head><body><article><h1>Article</h1><p>%s</p></article></body></html>`, body)
}))
defer server.Close()

content, err := h.FetchFullArticleContentContext(context.Background(), server.URL+"/lead", nil)
if err != nil {
t.Fatal(err)
}
if !strings.Contains(content, `src="`+server.URL+`/lead.jpg"`) || !strings.Contains(content, `referrerpolicy="no-referrer"`) {
t.Fatalf("lead image was not restored safely: %s", content)
}

content, err = h.FetchFullArticleContentContext(context.Background(), server.URL+"/existing", nil)
if err != nil {
t.Fatal(err)
}
if !strings.Contains(content, server.URL+"/body.jpg") || strings.Contains(content, server.URL+"/og.jpg") {
t.Fatalf("existing article image was duplicated or replaced: %s", content)
}
}

func TestFullTextNoSelectorMatchIsAnError(t *testing.T) {
h := fullTextHandler(t)
id, err := h.DB.AddFeed(&models.Feed{Title: "f", URL: "https://example.org"})
Expand Down
52 changes: 0 additions & 52 deletions internal/handlers/media/media_handlers_test.go
Original file line number Diff line number Diff line change
@@ -1,11 +1,8 @@
package media

import (
"encoding/base64"
"net/http"
"net/http/httptest"
"regexp"
"strings"
"testing"

"MrRSS/internal/database"
Expand Down Expand Up @@ -180,55 +177,6 @@ func TestProxyImagesInHTML_RelativeURLs(t *testing.T) {
}
}

func TestRewriteHTMLContent_ResponsiveImageCandidates(t *testing.T) {
baseURL := "https://example.com/news/article"
htmlContent := `<picture>
<source srcSet="/_next/image?url=%2Fhero.jpg&amp;w=1280&amp;q=75 1x, https://cdn.example.com/hero.jpg 2x">
<img src="/fallback.jpg" data-srcset="images/small.jpg 320w, images/large.jpg 1280w">
</picture>`

result := string(rewriteHTMLContent([]byte(htmlContent), baseURL))
for _, descriptor := range []string{" 1x", " 2x", " 320w", " 1280w"} {
if !strings.Contains(result, descriptor) {
t.Errorf("missing srcset descriptor %q in %s", descriptor, result)
}
}
if strings.Contains(result, `srcSet="/_next/image`) || strings.Contains(result, `data-srcset="images/`) {
t.Fatalf("responsive image candidates were not proxied: %s", result)
}

encodedURLs := regexp.MustCompile(`url_b64=([A-Za-z0-9+/=]+)`).FindAllStringSubmatch(result, -1)
var decodedURLs []string
for _, match := range encodedURLs {
decoded, err := base64.StdEncoding.DecodeString(match[1])
if err != nil {
t.Fatalf("decode proxied URL: %v", err)
}
decodedURLs = append(decodedURLs, string(decoded))
}
joinedURLs := strings.Join(decodedURLs, "\n")
for _, expected := range []string{
"https://example.com/_next/image?url=%2Fhero.jpg&w=1280&q=75",
"https://cdn.example.com/hero.jpg",
"https://example.com/news/images/small.jpg",
"https://example.com/news/images/large.jpg",
} {
if !strings.Contains(joinedURLs, expected) {
t.Errorf("missing decoded candidate %q in %s", expected, joinedURLs)
}
}
if strings.Contains(joinedURLs, "&amp;") {
t.Fatalf("HTML entities leaked into proxied URLs: %s", joinedURLs)
}
}

func TestRewriteSrcsetAttribute_SkipsNonHTTPAndProxiedCandidates(t *testing.T) {
content := `<img srcset="data:image/png;base64,AAAA 1x, blob:https://example.com/id 2x, #poster 320w, /api/webpage/resource?url_b64=abc 640w">`
if got := rewriteSrcsetAttribute(content, "img", "srcset", "https://example.com/article"); got != content {
t.Fatalf("special srcset candidates changed:\nwant: %s\n got: %s", content, got)
}
}

func contains(s, substr string) bool {
return len(s) >= len(substr) && (s == substr || len(substr) == 0 ||
(len(s) > 0 && findInString(s, substr)))
Expand Down
156 changes: 70 additions & 86 deletions internal/handlers/media/media_proxy.go
Original file line number Diff line number Diff line change
Expand Up @@ -823,8 +823,6 @@ func rewriteHTMLContent(bodyBytes []byte, baseURL string) []byte {

// Then rewrite img src attributes (now including the converted lazy images)
content = rewriteAttribute(content, "img", "src", baseURL)
content = rewriteSrcsetAttribute(content, "img", "srcset", baseURL)
content = rewriteSrcsetAttribute(content, "img", "data-srcset", baseURL)

// Rewrite iframe src attributes
content = rewriteAttribute(content, "iframe", "src", baseURL)
Expand All @@ -838,8 +836,6 @@ func rewriteHTMLContent(bodyBytes []byte, baseURL string) []byte {

// Rewrite source src attributes (for video/audio)
content = rewriteAttribute(content, "source", "src", baseURL)
content = rewriteSrcsetAttribute(content, "source", "srcset", baseURL)
content = rewriteSrcsetAttribute(content, "source", "data-srcset", baseURL)

// Rewrite track src attributes
content = rewriteAttribute(content, "track", "src", baseURL)
Expand Down Expand Up @@ -1059,101 +1055,89 @@ func parseHTMLAttributes(tag string) []htmlAttribute {
return attrs
}

// rewriteAttribute rewrites a specific URL attribute in HTML tags.
// rewriteAttribute rewrites a specific attribute in HTML tags
func rewriteAttribute(content, tag, attr, baseURL string) string {
return rewriteAttributeValue(content, tag, attr, func(value string) (string, bool) {
return proxyWebpageResourceURL(value, baseURL)
})
}
// Match all tags first
tagRe := regexp.MustCompile(fmt.Sprintf(`<%s[^>]*>`, tag))

// rewriteSrcsetAttribute proxies every candidate URL while preserving its
// density or width descriptor (for example, 2x or 640w).
func rewriteSrcsetAttribute(content, tag, attr, baseURL string) string {
return rewriteAttributeValue(content, tag, attr, func(value string) (string, bool) {
var result strings.Builder
changed := false
for position := 0; position < len(value); {
prefixStart := position
for position < len(value) && (isHTMLSpace(value[position]) || value[position] == ',') {
position++
}
result.WriteString(value[prefixStart:position])
if position >= len(value) {
break
}
matchCount := 0
rewriteCount := 0

urlStart := position
isDataURL := strings.HasPrefix(strings.ToLower(value[position:]), "data:")
for position < len(value) && !isHTMLSpace(value[position]) && (isDataURL || value[position] != ',') {
position++
}
candidate := value[urlStart:position]
if proxied, ok := proxyWebpageResourceURL(candidate, baseURL); ok {
result.WriteString(proxied)
changed = true
result := tagRe.ReplaceAllStringFunc(content, func(match string) string {
matchCount++
// Try to find the attribute with double quotes
doubleQuoteRe := regexp.MustCompile(fmt.Sprintf(`\s%s\s*=\s*"([^"]*)"`, attr))
doubleQuoteMatch := doubleQuoteRe.FindStringSubmatch(match)

var urlValue, quote string

if len(doubleQuoteMatch) >= 2 {
// Found double-quoted attribute
urlValue = doubleQuoteMatch[1]
quote = `"`
} else {
// Try single quotes
singleQuoteRe := regexp.MustCompile(fmt.Sprintf(`\s%s\s*=\s*'([^']*)'`, attr))
singleQuoteMatch := singleQuoteRe.FindStringSubmatch(match)
if len(singleQuoteMatch) >= 2 {
urlValue = singleQuoteMatch[1]
quote = `'`
} else {
result.WriteString(candidate)
// Try unquoted
unquotedRe := regexp.MustCompile(fmt.Sprintf(`\s%s\s*=\s*([^\s>]+)`, attr))
unquotedMatch := unquotedRe.FindStringSubmatch(match)
if len(unquotedMatch) >= 2 {
urlValue = unquotedMatch[1]
quote = ""
} else {
// Attribute not found
return match
}
}
}

descriptorStart := position
for position < len(value) && value[position] != ',' {
position++
}
result.WriteString(value[descriptorStart:position])
// Skip data: URLs, blob: URLs, and already proxied URLs
if strings.HasPrefix(urlValue, "data:") ||
strings.HasPrefix(urlValue, "blob:") ||
strings.HasPrefix(urlValue, "/api/") ||
strings.HasPrefix(urlValue, "#") {
return match
}
return result.String(), changed
})
}

func rewriteAttributeValue(content, tag, attr string, rewrite func(string) (string, bool)) string {
tagRe := regexp.MustCompile(`(?i)<` + regexp.QuoteMeta(tag) + `\b[^>]*>`)
patterns := []*regexp.Regexp{
regexp.MustCompile(`(?i)\s+` + regexp.QuoteMeta(attr) + `\s*=\s*"([^"]*)"`),
regexp.MustCompile(`(?i)\s+` + regexp.QuoteMeta(attr) + `\s*=\s*'([^']*)'`),
regexp.MustCompile(`(?i)\s+` + regexp.QuoteMeta(attr) + `\s*=\s*([^\s>]+)`),
}
rewriteCount++
// if tag == "script" || tag == "link" {
// log.Printf("[%s Rewrite] Rewriting %s %d: %s", strings.ToUpper(tag), attr, rewriteCount, urlValue)
// }

return tagRe.ReplaceAllStringFunc(content, func(match string) string {
for index, pattern := range patterns {
location := pattern.FindStringSubmatchIndex(match)
if len(location) < 4 {
continue
}
valueStart, valueEnd := location[2], location[3]
rewritten, changed := rewrite(match[valueStart:valueEnd])
if !changed {
return match
}
if index == len(patterns)-1 {
rewritten = `"` + rewritten + `"`
}
return match[:valueStart] + rewritten + match[valueEnd:]
// Resolve relative URLs
resolvedURL := resolveURL(urlValue, baseURL)

// Create proxied URL with base64 encoding
proxiedURL := fmt.Sprintf("/api/webpage/resource?url_b64=%s&referer_b64=%s",
base64.StdEncoding.EncodeToString([]byte(resolvedURL)),
base64.StdEncoding.EncodeToString([]byte(baseURL)))

// Replace the URL in the match
// Use regex to replace attribute value more reliably
if quote != "" {
// Quoted value - replace using regex for more flexibility
attrPattern := regexp.MustCompile(`(` + attr + `)\s*=\s*` + regexp.QuoteMeta(quote) + regexp.QuoteMeta(urlValue) + regexp.QuoteMeta(quote))
replacement := fmt.Sprintf(`%s=%s%s%s`, attr, quote, proxiedURL, quote)
return attrPattern.ReplaceAllString(match, replacement)
} else {
// Unquoted value - match until whitespace or > character
// We need to capture the delimiter (space or >) to preserve it
attrPattern := regexp.MustCompile(`(` + attr + `)\s*=\s*` + regexp.QuoteMeta(urlValue) + `([\s>])`)
replacement := fmt.Sprintf(`%s="%s"$2`, attr, proxiedURL)
return attrPattern.ReplaceAllString(match, replacement)
}
return match
})
}

func proxyWebpageResourceURL(value, baseURL string) (string, bool) {
value = strings.TrimSpace(html.UnescapeString(value))
lowerValue := strings.ToLower(value)
if value == "" || strings.HasPrefix(lowerValue, "data:") ||
strings.HasPrefix(lowerValue, "blob:") || strings.HasPrefix(value, "#") ||
strings.HasPrefix(value, "/api/") || strings.Contains(value, "/api/webpage/resource?") {
return value, false
}

resolvedURL := resolveURL(value, baseURL)
parsedURL, err := url.Parse(resolvedURL)
if err != nil || (parsedURL.Scheme != "http" && parsedURL.Scheme != "https") {
return value, false
}
return fmt.Sprintf("/api/webpage/resource?url_b64=%s&referer_b64=%s",
base64.StdEncoding.EncodeToString([]byte(resolvedURL)),
base64.StdEncoding.EncodeToString([]byte(baseURL))), true
}
// if matchCount > 0 && (tag == "script" || tag == "link") {
// log.Printf("[%s Rewrite] Found %d %s tags, rewrote %d %s attributes", strings.ToUpper(tag), matchCount, tag, rewriteCount, attr)
// }

func isHTMLSpace(value byte) bool {
return value == ' ' || value == '\t' || value == '\n' || value == '\r' || value == '\f'
return result
}

// rewriteLinkHref rewrites href attributes in link tags
Expand Down
Loading