{
  "entities": [
    {
      "id": 6,
      "slug": "claude-searchbot",
      "vendor": "Anthropic",
      "name": "Claude-SearchBot",
      "kind": "search_bot",
      "purpose": "search",
      "ua_token": "Claude-SearchBot",
      "ua_pattern": "Claude-SearchBot",
      "robots_token": "Claude-SearchBot",
      "ip_list_url": "https://claude.com/crawling/bots.json",
      "docs_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Navigates the web to improve Claude search result quality.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 7,
      "slug": "claude-user",
      "vendor": "Anthropic",
      "name": "Claude-User",
      "kind": "fetcher",
      "purpose": "user_fetch",
      "ua_token": "Claude-User",
      "ua_pattern": "Claude-User",
      "robots_token": "Claude-User",
      "ip_list_url": "https://claude.com/crawling/bots.json",
      "docs_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Retrieves a page when a Claude user directs Claude to access that URL.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 5,
      "slug": "claudebot",
      "vendor": "Anthropic",
      "name": "ClaudeBot",
      "kind": "crawler",
      "purpose": "training",
      "ua_token": "ClaudeBot",
      "ua_pattern": "ClaudeBot",
      "robots_token": "ClaudeBot",
      "ip_list_url": "https://claude.com/crawling/bots.json",
      "docs_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Collects web content that may contribute to Anthropic model training.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 14,
      "slug": "applebot",
      "vendor": "Apple",
      "name": "Applebot",
      "kind": "crawler",
      "purpose": "search",
      "ua_token": "Applebot",
      "ua_pattern": "Applebot(?!-Extended)",
      "robots_token": "Applebot",
      "ip_list_url": "https://search.developer.apple.com/applebot.json",
      "docs_url": "https://support.apple.com/en-us/119829",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Crawls the web to power Siri, Spotlight and Safari search features.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 15,
      "slug": "applebot-extended",
      "vendor": "Apple",
      "name": "Applebot-Extended",
      "kind": "policy_token",
      "purpose": "training",
      "ua_token": null,
      "ua_pattern": null,
      "robots_token": "Applebot-Extended",
      "ip_list_url": null,
      "docs_url": "https://support.apple.com/en-us/119829",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Robots.txt token that controls whether content already crawled by Applebot may be used to train Apple foundation models. Vendor documentation states Applebot-Extended does not crawl web pages: it is a policy token, not an observable crawler.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 12,
      "slug": "google-extended",
      "vendor": "Google",
      "name": "Google-Extended",
      "kind": "policy_token",
      "purpose": "training",
      "ua_token": null,
      "ua_pattern": null,
      "robots_token": "Google-Extended",
      "ip_list_url": null,
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Robots.txt token that controls whether content already crawled by Googlebot may be used to train Gemini models and ground Gemini features. Vendor documentation states Google-Extended is not a separate crawler and uses no distinct user agent string: it cannot be observed directly in request logs, only declared for in robots.txt.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 11,
      "slug": "googleother",
      "vendor": "Google",
      "name": "GoogleOther",
      "kind": "crawler",
      "purpose": "mixed",
      "ua_token": "GoogleOther",
      "ua_pattern": "GoogleOther",
      "robots_token": "GoogleOther",
      "ip_list_url": "https://developers.google.com/static/crawling/ipranges/common-crawlers.json",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Generic crawler used by various Google product teams to fetch publicly accessible content outside Search.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 10,
      "slug": "googlebot",
      "vendor": "Google",
      "name": "Googlebot",
      "kind": "crawler",
      "purpose": "search",
      "ua_token": "Googlebot",
      "ua_pattern": "Googlebot",
      "robots_token": "Googlebot",
      "ip_list_url": "https://developers.google.com/static/crawling/ipranges/common-crawlers.json",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Common crawler for Google Search, Discover, Images, Video and News.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 13,
      "slug": "bingbot",
      "vendor": "Microsoft",
      "name": "Bingbot",
      "kind": "crawler",
      "purpose": "search",
      "ua_token": "bingbot",
      "ua_pattern": "bingbot",
      "robots_token": "bingbot",
      "ip_list_url": "https://www.bing.com/toolbox/bingbot.json",
      "docs_url": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Crawls the web for Bing Search and, per Microsoft, feeds Copilot and Bing Chat surfaces. Microsoft documents a separate opt-out from generative-AI training use via the noarchive robots meta tag or X-Robots-Tag header rather than a distinct Copilot user agent; no Copilot-specific crawler token is named in Microsoft's public documentation as of 2026-09-13.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 3,
      "slug": "chatgpt-user",
      "vendor": "OpenAI",
      "name": "ChatGPT-User",
      "kind": "fetcher",
      "purpose": "user_fetch",
      "ua_token": "ChatGPT-User",
      "ua_pattern": "ChatGPT-User",
      "robots_token": "ChatGPT-User",
      "ip_list_url": "https://openai.com/chatgpt-user.json",
      "docs_url": "https://platform.openai.com/docs/bots",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Fetches a page only when a ChatGPT or Custom GPT user directs it to that URL. Vendor documentation states robots.txt rules may not apply to this fetcher.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 1,
      "slug": "gptbot",
      "vendor": "OpenAI",
      "name": "GPTBot",
      "kind": "crawler",
      "purpose": "training",
      "ua_token": "GPTBot",
      "ua_pattern": "GPTBot",
      "robots_token": "GPTBot",
      "ip_list_url": "https://openai.com/gptbot.json",
      "docs_url": "https://platform.openai.com/docs/bots",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Crawls public web content for OpenAI model training.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 4,
      "slug": "oai-adsbot",
      "vendor": "OpenAI",
      "name": "OAI-AdsBot",
      "kind": "ads_bot",
      "purpose": "ads",
      "ua_token": "OAI-AdsBot",
      "ua_pattern": "OAI-AdsBot",
      "robots_token": "OAI-AdsBot",
      "ip_list_url": "https://openai.com/adsbot.json",
      "docs_url": "https://platform.openai.com/docs/bots",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Checks ad landing page quality and relevance for ChatGPT ads. Vendor documentation states it only visits pages submitted for ads review.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 2,
      "slug": "oai-searchbot",
      "vendor": "OpenAI",
      "name": "OAI-SearchBot",
      "kind": "search_bot",
      "purpose": "search",
      "ua_token": "OAI-SearchBot",
      "ua_pattern": "OAI-SearchBot",
      "robots_token": "OAI-SearchBot",
      "ip_list_url": "https://openai.com/searchbot.json",
      "docs_url": "https://platform.openai.com/docs/bots",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Surfaces web pages in ChatGPT search results.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 9,
      "slug": "perplexity-user",
      "vendor": "Perplexity",
      "name": "Perplexity-User",
      "kind": "fetcher",
      "purpose": "user_fetch",
      "ua_token": "Perplexity-User",
      "ua_pattern": "Perplexity-User",
      "robots_token": "Perplexity-User",
      "ip_list_url": "https://www.perplexity.ai/perplexity-user.json",
      "docs_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Fetches a page when a Perplexity user asks a question that requires that page. Vendor documentation states this fetcher generally ignores robots.txt rules.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 8,
      "slug": "perplexitybot",
      "vendor": "Perplexity",
      "name": "PerplexityBot",
      "kind": "search_bot",
      "purpose": "search",
      "ua_token": "PerplexityBot",
      "ua_pattern": "PerplexityBot",
      "robots_token": "PerplexityBot",
      "ip_list_url": "https://www.perplexity.ai/perplexitybot.json",
      "docs_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Surfaces and links websites in Perplexity search results.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    }
  ],
  "definition": "The Crawlers index of Rattlesnakes By Mail lists every automated agent and every robots.txt policy token that Rattlesnakes By Mail documents, one row per entity. An entity is one named agent, or one named robots.txt token, that a vendor publishes and documents separately, and each entity holds its own record at its own URL. Rattlesnakes By Mail covers six vendors: Anthropic, OpenAI, Google, Microsoft, Apple and Perplexity. Fifteen rows cover crawlers, fetchers, search bots, one ads bot and two robots.txt policy tokens."
}