From 16d3487c18114f2482bc948d68e8b40b56fd6ee8 Mon Sep 17 00:00:00 2001 From: Tsutomu Katsube Date: Sat, 26 Sep 2026 17:10:30 +0900 Subject: [PATCH 1/5] Include the query in the cache ID GitHub: follow-up GH-24 Because if URL has any query parameters, the cache ID may be conflicted. The cache ID replaces `/`, `:`, `?`, and `*` with `-`. They may appear in a URL without percent-encoding: * Host: alphanumerics and `- . _ ~ ! $ & ' ( ) * + , ; =` (RFC 3986 3.2.2) * Path: the host characters and `: @` (`/` is the separator) (RFC 3986 3.3) * Query: the path characters and `/ ?` (RFC 3986 3.4) However, they can't be used in file names: * POSIX: `/` * Windows: `< > : " / \ | ? *` ("Naming Files, Paths, and Namespaces" below) See also: * RFC 3986 "Uniform Resource Identifier (URI): Generic Syntax": https://www.rfc-editor.org/rfc/rfc3986 * "Naming Files, Paths, and Namespaces" (Microsoft Learn): https://learn.microsoft.com/en-us/windows/win32/fileio/naming-a-file Co-authored-by: Sutou Kouhei --- lib/remote_input.rb | 6 ++++-- test/test-remote-input.rb | 2 +- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/lib/remote_input.rb b/lib/remote_input.rb index b0fe0e9..ae278b4 100644 --- a/lib/remote_input.rb +++ b/lib/remote_input.rb @@ -56,8 +56,10 @@ def path def cache_path return @cache_path if @cache_path dirname = File.dirname(normalize_path).delete_suffix("/") - cache_id = "#{@url.host}#{dirname}".tr("/", "-") - @cache_path = CachePath.new(cache_id) + cache_id = "#{@url.host}#{dirname}" + query = @url.query + cache_id += "-#{query}" if query and not query.empty? + @cache_path = CachePath.new(cache_id.tr("/:?*", "-")) end def normalize_path diff --git a/test/test-remote-input.rb b/test/test-remote-input.rb index 0b71b3a..c367b26 100644 --- a/test/test-remote-input.rb +++ b/test/test-remote-input.rb @@ -61,7 +61,7 @@ def test_open_with_block_raised data("no path", ["/example.com/data", "https://example.com"]) data("root", ["/example.com/data", "https://example.com/"]) data("file", ["/example.com/file", "https://example.com/file"]) - data("query", ["/example.com/file", "https://example.com/file?a=b"]) + data("query", ["/example.com-a=b/file", "https://example.com/file?a=b"]) data("directory", ["/example.com-a/data", "https://example.com/a/"]) data("nested file", ["/example.com-a/file", "https://example.com/a/file"]) data("deeply nested", ["/example.com-a-b/file", "https://example.com/a/b/file"]) From bf0f9ab7995031002b46b6f538465953f6bdcc45 Mon Sep 17 00:00:00 2001 From: Tsutomu Katsube Date: Sun, 27 Sep 2026 06:59:05 +0900 Subject: [PATCH 2/5] Use allow list style instead of deny list style Because deny list style is difficult to maintain. The cache ID keeps only the URL unreserved characters (RFC 3986 2.3) and `=` for query readability. Co-authored-by: Sutou Kouhei --- lib/remote_input.rb | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/lib/remote_input.rb b/lib/remote_input.rb index ae278b4..7ba45ac 100644 --- a/lib/remote_input.rb +++ b/lib/remote_input.rb @@ -59,7 +59,7 @@ def cache_path cache_id = "#{@url.host}#{dirname}" query = @url.query cache_id += "-#{query}" if query and not query.empty? - @cache_path = CachePath.new(cache_id.tr("/:?*", "-")) + @cache_path = CachePath.new(cache_id.tr("^0-9A-Za-z._~=-", "-")) end def normalize_path From 5ab366bcb217ef6579b303c4478bf6d0d3b0de7e Mon Sep 17 00:00:00 2001 From: Tsutomu Katsube Date: Sun, 27 Sep 2026 07:36:07 +0900 Subject: [PATCH 3/5] Extract the allow list into a variable --- lib/remote_input.rb | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/lib/remote_input.rb b/lib/remote_input.rb index 7ba45ac..842cc64 100644 --- a/lib/remote_input.rb +++ b/lib/remote_input.rb @@ -59,7 +59,8 @@ def cache_path cache_id = "#{@url.host}#{dirname}" query = @url.query cache_id += "-#{query}" if query and not query.empty? - @cache_path = CachePath.new(cache_id.tr("^0-9A-Za-z._~=-", "-")) + allow_list = "0-9A-Za-z._~=-" + @cache_path = CachePath.new(cache_id.tr("^#{allow_list}", "-")) end def normalize_path From 6f5376f215cca6a63a65557d06fb5667f4a9b1f6 Mon Sep 17 00:00:00 2001 From: Tsutomu Katsube Date: Sun, 27 Sep 2026 07:46:23 +0900 Subject: [PATCH 4/5] test: confirm that a disallowed character is replaced --- test/test-remote-input.rb | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/test/test-remote-input.rb b/test/test-remote-input.rb index c367b26..a92429f 100644 --- a/test/test-remote-input.rb +++ b/test/test-remote-input.rb @@ -61,7 +61,7 @@ def test_open_with_block_raised data("no path", ["/example.com/data", "https://example.com"]) data("root", ["/example.com/data", "https://example.com/"]) data("file", ["/example.com/file", "https://example.com/file"]) - data("query", ["/example.com-a=b/file", "https://example.com/file?a=b"]) + data("query", ["/example.com-a=-/file", "https://example.com/file?a=/"]) data("directory", ["/example.com-a/data", "https://example.com/a/"]) data("nested file", ["/example.com-a/file", "https://example.com/a/file"]) data("deeply nested", ["/example.com-a-b/file", "https://example.com/a/b/file"]) From 0fe7cc531d053b3ab2a184290626932b68bc2622 Mon Sep 17 00:00:00 2001 From: Tsutomu Katsube Date: Mon, 28 Sep 2026 06:28:49 +0900 Subject: [PATCH 5/5] Use "+" as the query separator in the cache ID Because "+" isn't in the allow list and can be used in a file name. Co-authored-by: Sutou Kouhei --- lib/remote_input.rb | 10 +++++++--- test/test-remote-input.rb | 2 +- 2 files changed, 8 insertions(+), 4 deletions(-) diff --git a/lib/remote_input.rb b/lib/remote_input.rb index 842cc64..122d597 100644 --- a/lib/remote_input.rb +++ b/lib/remote_input.rb @@ -56,11 +56,15 @@ def path def cache_path return @cache_path if @cache_path dirname = File.dirname(normalize_path).delete_suffix("/") - cache_id = "#{@url.host}#{dirname}" + cache_id = to_cache_id("#{@url.host}#{dirname}") query = @url.query - cache_id += "-#{query}" if query and not query.empty? + cache_id += "+#{to_cache_id(query)}" if query and not query.empty? + @cache_path = CachePath.new(cache_id) + end + + def to_cache_id(s) allow_list = "0-9A-Za-z._~=-" - @cache_path = CachePath.new(cache_id.tr("^#{allow_list}", "-")) + s.tr("^#{allow_list}", "-") end def normalize_path diff --git a/test/test-remote-input.rb b/test/test-remote-input.rb index a92429f..d115aa7 100644 --- a/test/test-remote-input.rb +++ b/test/test-remote-input.rb @@ -61,7 +61,7 @@ def test_open_with_block_raised data("no path", ["/example.com/data", "https://example.com"]) data("root", ["/example.com/data", "https://example.com/"]) data("file", ["/example.com/file", "https://example.com/file"]) - data("query", ["/example.com-a=-/file", "https://example.com/file?a=/"]) + data("query", ["/example.com+a=-/file", "https://example.com/file?a=+"]) data("directory", ["/example.com-a/data", "https://example.com/a/"]) data("nested file", ["/example.com-a/file", "https://example.com/a/file"]) data("deeply nested", ["/example.com-a-b/file", "https://example.com/a/b/file"])