diff --git a/lib/mcp_registry_web/controllers/sitemap_controller.ex b/lib/mcp_registry_web/controllers/sitemap_controller.ex index 1b255f9..e0da195 100644 --- a/lib/mcp_registry_web/controllers/sitemap_controller.ex +++ b/lib/mcp_registry_web/controllers/sitemap_controller.ex @@ -22,10 +22,28 @@ defmodule McpRegistryWeb.SitemapController do # listing carrying an unusually large tool set. @servers_per_tool_file 150 + # `/live` is LiveView's transport, not content. Its long-poll fallback carries + # a fresh CSRF token in the query string, so every fetch mints a URL that has + # never been seen before -- an endless supply of new pages to a crawler that + # cannot hold a websocket. Googlebot spent 39 of its requests there against 8 + # on real pages before this was disallowed. + # + # The JSON API is excluded too: it is for programs, it duplicates what the + # listing pages say, and every request spent there is one not spent on a page + # that can rank. llms.txt stays allowed -- it is written for agents to read. def robots(conn, _params) do + body = """ + User-agent: * + Disallow: /live/ + Disallow: /api/ + Allow: / + + Sitemap: #{McpRegistryWeb.Endpoint.url()}/sitemap.xml + """ + conn |> put_resp_header("cache-control", "public, max-age=86400") - |> text("User-agent: *\nDisallow:\n\nSitemap: #{McpRegistryWeb.Endpoint.url()}/sitemap.xml\n") + |> text(body) end def index(conn, _params) do diff --git a/lib/mcp_registry_web/plugs/geo_block.ex b/lib/mcp_registry_web/plugs/geo_block.ex index 9e9a86e..322872a 100644 --- a/lib/mcp_registry_web/plugs/geo_block.ex +++ b/lib/mcp_registry_web/plugs/geo_block.ex @@ -42,11 +42,13 @@ defmodule McpRegistryWeb.Plugs.GeoBlock do get through and some nearby ones are stopped. VPN and mobile traffic is weaker still. Treat it as a coarse filter, not a boundary. - It is also worth knowing that some of Google's own crawler addresses - geolocate to Mountain View. Blocking that city can therefore block part of - Googlebot, which would cost search visibility. Nothing here exempts - crawlers — if that trade is unwanted, exempt them here rather than - discovering it in Search Console. + ## Crawlers are exempt + + Some of Google's own addresses geolocate to Mountain View. With the block on + and no exemption, Googlebot was refused on `/book` and on two sitemap files + within hours — the site removing itself from search while appearing to work + perfectly to everyone else. Requests whose user agent names a known search + or AI crawler now skip the check entirely. """ @behaviour Plug @@ -55,8 +57,39 @@ defmodule McpRegistryWeb.Plugs.GeoBlock do @impl Plug def init(opts), do: opts + # Crawlers are exempt. Some of Google's own addresses geolocate to Mountain + # View, and with the block on, Googlebot was refused on /book and on two + # sitemap files within hours -- the site quietly removing itself from search + # while appearing to work. A geo block is a coarse filter on human traffic, + # not a security control (the moduledoc says so), so matching on user agent + # is proportionate even though it can be spoofed: anyone willing to forge a + # Googlebot header could equally use a VPN. + @crawlers ~w(googlebot bingbot duckduckbot applebot yandexbot baiduspider + slurp claudebot claude-web gptbot oai-searchbot chatgpt-user + perplexitybot amazonbot facebookexternalhit twitterbot + linkedinbot discordbot telegrambot whatsapp) + @impl Plug def call(conn, _opts) do + if crawler?(conn) do + conn + else + geo_check(conn) + end + end + + defp crawler?(conn) do + case Plug.Conn.get_req_header(conn, "user-agent") do + [agent | _] when is_binary(agent) -> + lowered = String.downcase(agent) + Enum.any?(@crawlers, &String.contains?(lowered, &1)) + + _ -> + false + end + end + + defp geo_check(conn) do if blocked?(conn.remote_ip) do conn |> put_resp_content_type("text/plain") diff --git a/test/mcp_registry_web/controllers/sitemap_controller_test.exs b/test/mcp_registry_web/controllers/sitemap_controller_test.exs index 7fac155..05c6386 100644 --- a/test/mcp_registry_web/controllers/sitemap_controller_test.exs +++ b/test/mcp_registry_web/controllers/sitemap_controller_test.exs @@ -3,11 +3,13 @@ defmodule McpRegistryWeb.SitemapControllerTest do import McpRegistry.RegistryFixtures - test "robots.txt allows everything and names the sitemap", %{conn: conn} do + test "robots.txt opens the content and names the sitemap", %{conn: conn} do body = conn |> get("/robots.txt") |> text_response(200) - assert body =~ "User-agent: *\nDisallow:\n" + assert body =~ "User-agent: *" + assert body =~ "Allow: /" assert body =~ "Sitemap: http://localhost:4000/sitemap.xml" + refute body =~ "Disallow: /servers" end test "the index lists the pages file and one file per 10,000 listings", %{conn: conn} do @@ -50,4 +52,18 @@ defmodule McpRegistryWeb.SitemapControllerTest do assert conn |> get(path) |> response(404) == "" end end + + test "robots.txt keeps crawlers out of the LiveView transport", %{conn: conn} do + body = conn |> get("/robots.txt") |> response(200) + + # /live/longpoll carries a fresh CSRF token per request, so each fetch is a + # URL never seen before: an endless supply of pages to a crawler that + # cannot hold a websocket. + assert body =~ "Disallow: /live/" + assert body =~ "Disallow: /api/" + assert body =~ "Sitemap: " + # Content must stay crawlable. + assert body =~ "Allow: /" + refute body =~ "Disallow: /servers" + end end diff --git a/test/mcp_registry_web/plugs/geo_block_test.exs b/test/mcp_registry_web/plugs/geo_block_test.exs index a506b47..80b4387 100644 --- a/test/mcp_registry_web/plugs/geo_block_test.exs +++ b/test/mcp_registry_web/plugs/geo_block_test.exs @@ -163,6 +163,39 @@ defmodule McpRegistryWeb.Plugs.GeoBlockTest do end end + describe "crawlers" do + test "a search engine is never blocked, wherever it appears to be" do + configure(@rule, always(@mountain_view)) + + for agent <- [ + "Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)", + "Mozilla/5.0 (compatible; bingbot/2.0; +http://www.bing.com/bingbot.htm)", + "Mozilla/5.0 (compatible; ClaudeBot/1.0)", + "Mozilla/5.0 (compatible; GPTBot/1.1)" + ] do + conn = + Phoenix.ConnTest.build_conn() + |> Plug.Conn.put_req_header("user-agent", agent) + |> Map.put(:remote_ip, @public_ip) + |> GeoBlock.call([]) + + refute conn.halted, "#{agent} should not be blocked" + end + end + + test "an ordinary browser in the blocked city still is" do + configure(@rule, always(@mountain_view)) + + conn = + Phoenix.ConnTest.build_conn() + |> Plug.Conn.put_req_header("user-agent", "Mozilla/5.0 (Macintosh) Safari/605.1.15") + |> Map.put(:remote_ip, @public_ip) + |> GeoBlock.call([]) + + assert conn.status == 403 + end + end + describe "the response" do test "a blocked request gets 403 and never reaches the router" do configure(@rule, always(@mountain_view))