From 4fe1268f9b26f18a50c34b6d07885fa4aefc8bf4 Mon Sep 17 00:00:00 2001 From: Logan Besecker Date: Thu, 24 Sep 2026 11:04:07 -0700 Subject: [PATCH] Let crawlers reach the content MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Googlebot made 39 requests to /live/longpoll and 8 to real pages in 24 hours, and was refused on two sitemap files and /book. - Disallow /live/ in robots.txt. LiveView's long-poll fallback carries a fresh CSRF token per request, so each fetch mints a URL never seen before; a crawler that cannot hold a websocket finds an endless supply of new pages there. robots.txt had excluded nothing at all. - Disallow /api/ too: it is for programs, it restates the listing pages, and every request spent there is one not spent on a page that can rank - Exempt known search and AI crawlers from the geo block. Some of Google's own addresses geolocate to Mountain View, so with the block on the site was quietly removing itself from search while looking perfect to every human. ClaudeBot made 3,908 content requests in the same window and hit the trap zero times, so the pages were always crawlable -- the problem was what Googlebot was allowed and pointed at. --- Pages affected: - [MCP Registry](https://ai.mcpharbor.dev/) — the Model Context Protocol server directory. - [Browse MCP servers](https://ai.mcpharbor.dev/servers) — the catalogue crawlers should be reaching. - [Sitemap](https://ai.mcpharbor.dev/sitemap.xml) — which Googlebot was being refused. - [MCP Server Optimization](https://ai.mcpharbor.dev/book) — also refused. Co-Authored-By: Claude Opus 5 --- .../controllers/sitemap_controller.ex | 20 ++++++++- lib/mcp_registry_web/plugs/geo_block.ex | 43 ++++++++++++++++--- .../controllers/sitemap_controller_test.exs | 20 ++++++++- .../mcp_registry_web/plugs/geo_block_test.exs | 33 ++++++++++++++ 4 files changed, 108 insertions(+), 8 deletions(-) diff --git a/lib/mcp_registry_web/controllers/sitemap_controller.ex b/lib/mcp_registry_web/controllers/sitemap_controller.ex index 1b255f9..e0da195 100644 --- a/lib/mcp_registry_web/controllers/sitemap_controller.ex +++ b/lib/mcp_registry_web/controllers/sitemap_controller.ex @@ -22,10 +22,28 @@ defmodule McpRegistryWeb.SitemapController do # listing carrying an unusually large tool set. @servers_per_tool_file 150 + # `/live` is LiveView's transport, not content. Its long-poll fallback carries + # a fresh CSRF token in the query string, so every fetch mints a URL that has + # never been seen before -- an endless supply of new pages to a crawler that + # cannot hold a websocket. Googlebot spent 39 of its requests there against 8 + # on real pages before this was disallowed. + # + # The JSON API is excluded too: it is for programs, it duplicates what the + # listing pages say, and every request spent there is one not spent on a page + # that can rank. llms.txt stays allowed -- it is written for agents to read. def robots(conn, _params) do + body = """ + User-agent: * + Disallow: /live/ + Disallow: /api/ + Allow: / + + Sitemap: #{McpRegistryWeb.Endpoint.url()}/sitemap.xml + """ + conn |> put_resp_header("cache-control", "public, max-age=86400") - |> text("User-agent: *\nDisallow:\n\nSitemap: #{McpRegistryWeb.Endpoint.url()}/sitemap.xml\n") + |> text(body) end def index(conn, _params) do diff --git a/lib/mcp_registry_web/plugs/geo_block.ex b/lib/mcp_registry_web/plugs/geo_block.ex index 9e9a86e..322872a 100644 --- a/lib/mcp_registry_web/plugs/geo_block.ex +++ b/lib/mcp_registry_web/plugs/geo_block.ex @@ -42,11 +42,13 @@ defmodule McpRegistryWeb.Plugs.GeoBlock do get through and some nearby ones are stopped. VPN and mobile traffic is weaker still. Treat it as a coarse filter, not a boundary. - It is also worth knowing that some of Google's own crawler addresses - geolocate to Mountain View. Blocking that city can therefore block part of - Googlebot, which would cost search visibility. Nothing here exempts - crawlers — if that trade is unwanted, exempt them here rather than - discovering it in Search Console. + ## Crawlers are exempt + + Some of Google's own addresses geolocate to Mountain View. With the block on + and no exemption, Googlebot was refused on `/book` and on two sitemap files + within hours — the site removing itself from search while appearing to work + perfectly to everyone else. Requests whose user agent names a known search + or AI crawler now skip the check entirely. """ @behaviour Plug @@ -55,8 +57,39 @@ defmodule McpRegistryWeb.Plugs.GeoBlock do @impl Plug def init(opts), do: opts + # Crawlers are exempt. Some of Google's own addresses geolocate to Mountain + # View, and with the block on, Googlebot was refused on /book and on two + # sitemap files within hours -- the site quietly removing itself from search + # while appearing to work. A geo block is a coarse filter on human traffic, + # not a security control (the moduledoc says so), so matching on user agent + # is proportionate even though it can be spoofed: anyone willing to forge a + # Googlebot header could equally use a VPN. + @crawlers ~w(googlebot bingbot duckduckbot applebot yandexbot baiduspider + slurp claudebot claude-web gptbot oai-searchbot chatgpt-user + perplexitybot amazonbot facebookexternalhit twitterbot + linkedinbot discordbot telegrambot whatsapp) + @impl Plug def call(conn, _opts) do + if crawler?(conn) do + conn + else + geo_check(conn) + end + end + + defp crawler?(conn) do + case Plug.Conn.get_req_header(conn, "user-agent") do + [agent | _] when is_binary(agent) -> + lowered = String.downcase(agent) + Enum.any?(@crawlers, &String.contains?(lowered, &1)) + + _ -> + false + end + end + + defp geo_check(conn) do if blocked?(conn.remote_ip) do conn |> put_resp_content_type("text/plain") diff --git a/test/mcp_registry_web/controllers/sitemap_controller_test.exs b/test/mcp_registry_web/controllers/sitemap_controller_test.exs index 7fac155..05c6386 100644 --- a/test/mcp_registry_web/controllers/sitemap_controller_test.exs +++ b/test/mcp_registry_web/controllers/sitemap_controller_test.exs @@ -3,11 +3,13 @@ defmodule McpRegistryWeb.SitemapControllerTest do import McpRegistry.RegistryFixtures - test "robots.txt allows everything and names the sitemap", %{conn: conn} do + test "robots.txt opens the content and names the sitemap", %{conn: conn} do body = conn |> get("/robots.txt") |> text_response(200) - assert body =~ "User-agent: *\nDisallow:\n" + assert body =~ "User-agent: *" + assert body =~ "Allow: /" assert body =~ "Sitemap: http://localhost:4000/sitemap.xml" + refute body =~ "Disallow: /servers" end test "the index lists the pages file and one file per 10,000 listings", %{conn: conn} do @@ -50,4 +52,18 @@ defmodule McpRegistryWeb.SitemapControllerTest do assert conn |> get(path) |> response(404) == "" end end + + test "robots.txt keeps crawlers out of the LiveView transport", %{conn: conn} do + body = conn |> get("/robots.txt") |> response(200) + + # /live/longpoll carries a fresh CSRF token per request, so each fetch is a + # URL never seen before: an endless supply of pages to a crawler that + # cannot hold a websocket. + assert body =~ "Disallow: /live/" + assert body =~ "Disallow: /api/" + assert body =~ "Sitemap: " + # Content must stay crawlable. + assert body =~ "Allow: /" + refute body =~ "Disallow: /servers" + end end diff --git a/test/mcp_registry_web/plugs/geo_block_test.exs b/test/mcp_registry_web/plugs/geo_block_test.exs index a506b47..80b4387 100644 --- a/test/mcp_registry_web/plugs/geo_block_test.exs +++ b/test/mcp_registry_web/plugs/geo_block_test.exs @@ -163,6 +163,39 @@ defmodule McpRegistryWeb.Plugs.GeoBlockTest do end end + describe "crawlers" do + test "a search engine is never blocked, wherever it appears to be" do + configure(@rule, always(@mountain_view)) + + for agent <- [ + "Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)", + "Mozilla/5.0 (compatible; bingbot/2.0; +http://www.bing.com/bingbot.htm)", + "Mozilla/5.0 (compatible; ClaudeBot/1.0)", + "Mozilla/5.0 (compatible; GPTBot/1.1)" + ] do + conn = + Phoenix.ConnTest.build_conn() + |> Plug.Conn.put_req_header("user-agent", agent) + |> Map.put(:remote_ip, @public_ip) + |> GeoBlock.call([]) + + refute conn.halted, "#{agent} should not be blocked" + end + end + + test "an ordinary browser in the blocked city still is" do + configure(@rule, always(@mountain_view)) + + conn = + Phoenix.ConnTest.build_conn() + |> Plug.Conn.put_req_header("user-agent", "Mozilla/5.0 (Macintosh) Safari/605.1.15") + |> Map.put(:remote_ip, @public_ip) + |> GeoBlock.call([]) + + assert conn.status == 403 + end + end + describe "the response" do test "a blocked request gets 403 and never reaches the router" do configure(@rule, always(@mountain_view))