{
  "claims": [
    {
      "id": 34,
      "entity_slug": "claude-searchbot",
      "entity_name": "Claude-SearchBot",
      "field": "cites_sources",
      "value": "not_documented",
      "statement": "Anthropic does not publish whether Claude search responses cite pages crawled by Claude-SearchBot.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 29,
      "entity_slug": "claude-searchbot",
      "entity_name": "Claude-SearchBot",
      "field": "honours_crawl_delay",
      "value": "yes",
      "statement": "Claude-SearchBot respects the Crawl-delay directive in robots.txt when crawling domains.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 30,
      "entity_slug": "claude-searchbot",
      "entity_name": "Claude-SearchBot",
      "field": "publishes_ip_list",
      "value": "yes",
      "statement": "Anthropic publishes the IP address ranges used by Claude-SearchBot as a JSON file at https://claude.com/crawling/bots.json.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 32,
      "entity_slug": "claude-searchbot",
      "entity_name": "Claude-SearchBot",
      "field": "purpose_documented",
      "value": "search",
      "statement": "Claude-SearchBot analyses online content to improve the relevance and accuracy of Claude search responses.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 35,
      "entity_slug": "claude-searchbot",
      "entity_name": "Claude-SearchBot",
      "field": "reads_llms_txt",
      "value": "not_documented",
      "statement": "Anthropic does not publish whether Claude-SearchBot reads llms.txt files.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 28,
      "entity_slug": "claude-searchbot",
      "entity_name": "Claude-SearchBot",
      "field": "respects_robots_txt",
      "value": "yes",
      "statement": "Claude-SearchBot honours industry standard robots.txt directives that signal do not crawl.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 33,
      "entity_slug": "claude-searchbot",
      "entity_name": "Claude-SearchBot",
      "field": "separate_search_and_training_tokens",
      "value": "yes",
      "statement": "Anthropic separates search and training controls into two robots.txt tokens: Claude-SearchBot for search and ClaudeBot for training.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 36,
      "entity_slug": "claude-searchbot",
      "entity_name": "Claude-SearchBot",
      "field": "user_agent_string_full",
      "value": "not_documented",
      "statement": "Anthropic does not publish a full user agent string for Claude-SearchBot.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 31,
      "entity_slug": "claude-searchbot",
      "entity_name": "Claude-SearchBot",
      "field": "verifiable_by_rdns",
      "value": "not_documented",
      "statement": "Anthropic does not publish a reverse-DNS verification method for Claude-SearchBot.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 44,
      "entity_slug": "claude-user",
      "entity_name": "Claude-User",
      "field": "cites_sources",
      "value": "not_documented",
      "statement": "Anthropic does not publish whether Claude responses cite pages fetched by Claude-User.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 39,
      "entity_slug": "claude-user",
      "entity_name": "Claude-User",
      "field": "honours_crawl_delay",
      "value": "yes",
      "statement": "Claude-User respects the Crawl-delay directive in robots.txt when fetching pages.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 40,
      "entity_slug": "claude-user",
      "entity_name": "Claude-User",
      "field": "publishes_ip_list",
      "value": "yes",
      "statement": "Anthropic publishes the IP address ranges used by Claude-User as a JSON file at https://claude.com/crawling/bots.json.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 42,
      "entity_slug": "claude-user",
      "entity_name": "Claude-User",
      "field": "purpose_documented",
      "value": "user_fetch",
      "statement": "Claude-User accesses websites in response to questions that individual users ask Claude.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 45,
      "entity_slug": "claude-user",
      "entity_name": "Claude-User",
      "field": "reads_llms_txt",
      "value": "not_documented",
      "statement": "Anthropic does not publish whether Claude-User reads llms.txt files.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 37,
      "entity_slug": "claude-user",
      "entity_name": "Claude-User",
      "field": "respects_robots_txt",
      "value": "yes",
      "statement": "Claude-User honours industry standard robots.txt directives that signal do not crawl.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 43,
      "entity_slug": "claude-user",
      "entity_name": "Claude-User",
      "field": "separate_search_and_training_tokens",
      "value": "not_documented",
      "statement": "Anthropic does not document Claude-User as a search or training opt-out token.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 46,
      "entity_slug": "claude-user",
      "entity_name": "Claude-User",
      "field": "user_agent_string_full",
      "value": "not_documented",
      "statement": "Anthropic does not publish a full user agent string for Claude-User.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 38,
      "entity_slug": "claude-user",
      "entity_name": "Claude-User",
      "field": "user_fetch_exception_documented",
      "value": "no",
      "statement": "Anthropic applies the same robots.txt statement to Claude-User as to ClaudeBot and Claude-SearchBot and documents no exception for user-initiated fetches.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 41,
      "entity_slug": "claude-user",
      "entity_name": "Claude-User",
      "field": "verifiable_by_rdns",
      "value": "not_documented",
      "statement": "Anthropic does not publish a reverse-DNS verification method for Claude-User.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 53,
      "entity_slug": "claudebot",
      "entity_name": "ClaudeBot",
      "field": "anti_circumvention_stance",
      "value": "documented",
      "statement": "ClaudeBot respects anti-circumvention technologies and does not bypass CAPTCHAs on crawled sites.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 48,
      "entity_slug": "claudebot",
      "entity_name": "ClaudeBot",
      "field": "honours_crawl_delay",
      "value": "yes",
      "statement": "Anthropic supports the non-standard robots.txt Crawl-delay extension for ClaudeBot.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 143,
      "entity_slug": "claudebot",
      "entity_name": "ClaudeBot",
      "field": "observed_fetches_json",
      "value": "yes",
      "statement": "ClaudeBot made 177 JSON requests to Rattlesnakes By Mail between 2026-09-13T20:00Z and 2026-09-15T09:00Z, covering the .json twin of 170 base paths. The .json URLs of Rattlesnakes By Mail appear in no entry of the 171 URL sitemap.",
      "confidence": "high",
      "verified_at": "2026-09-15",
      "status": "current"
    },
    {
      "id": 142,
      "entity_slug": "claudebot",
      "entity_name": "ClaudeBot",
      "field": "observed_fetches_markdown",
      "value": "yes",
      "statement": "ClaudeBot fetched the .md twin of 170 base paths on Rattlesnakes By Mail between 2026-09-13T20:00Z and 2026-09-15T09:00Z, 170 Markdown requests of 542 ClaudeBot requests in that window. The .md URLs of Rattlesnakes By Mail appear in no entry of the 171 URL sitemap.",
      "confidence": "high",
      "verified_at": "2026-09-15",
      "status": "current"
    },
    {
      "id": 144,
      "entity_slug": "claudebot",
      "entity_name": "ClaudeBot",
      "field": "observed_first_request_path",
      "value": "/sitemap.xml",
      "statement": "ClaudeBot's first request to Rattlesnakes By Mail between 2026-09-13T20:00Z and 2026-09-15T09:00Z was /sitemap.xml at 2026-09-14T19:53:45Z. ClaudeBot fetched /robots.txt on the second request to Rattlesnakes By Mail.",
      "confidence": "high",
      "verified_at": "2026-09-15",
      "status": "current"
    },
    {
      "id": 145,
      "entity_slug": "claudebot",
      "entity_name": "ClaudeBot",
      "field": "observed_format_pass_order",
      "value": "html_json_md",
      "statement": "ClaudeBot fetched Rattlesnakes By Mail in three format passes between 2026-09-13T20:00Z and 2026-09-15T09:00Z, HTML from 2026-09-14T20:52Z to 2026-09-14T20:57Z, then JSON from 2026-09-14T21:37Z to 2026-09-14T21:56Z, then Markdown from 2026-09-14T21:37Z to 2026-09-14T21:57Z. The same order holds on 170 of 170 base paths that ClaudeBot fetched in all three formats on Rattlesnakes By Mail.",
      "confidence": "high",
      "verified_at": "2026-09-15",
      "status": "current"
    },
    {
      "id": 49,
      "entity_slug": "claudebot",
      "entity_name": "ClaudeBot",
      "field": "publishes_ip_list",
      "value": "yes",
      "statement": "Anthropic publishes the IP address ranges used by ClaudeBot as a JSON file at https://claude.com/crawling/bots.json.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 51,
      "entity_slug": "claudebot",
      "entity_name": "ClaudeBot",
      "field": "purpose_documented",
      "value": "training",
      "statement": "ClaudeBot collects web content for training Anthropic generative AI models.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 47,
      "entity_slug": "claudebot",
      "entity_name": "ClaudeBot",
      "field": "respects_robots_txt",
      "value": "yes",
      "statement": "ClaudeBot honours industry standard robots.txt directives that signal do not crawl.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 52,
      "entity_slug": "claudebot",
      "entity_name": "ClaudeBot",
      "field": "separate_search_and_training_tokens",
      "value": "yes",
      "statement": "Anthropic separates training and search controls into two robots.txt tokens: ClaudeBot for training and Claude-SearchBot for search.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 54,
      "entity_slug": "claudebot",
      "entity_name": "ClaudeBot",
      "field": "user_agent_string_full",
      "value": "not_documented",
      "statement": "Anthropic does not publish a full user agent string for ClaudeBot.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 50,
      "entity_slug": "claudebot",
      "entity_name": "ClaudeBot",
      "field": "verifiable_by_rdns",
      "value": "not_documented",
      "statement": "Anthropic does not publish a reverse-DNS verification method for ClaudeBot.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 15,
      "entity_slug": "applebot",
      "entity_name": "Applebot",
      "field": "cites_sources",
      "value": "yes",
      "statement": "Apple includes links to source websites in Siri and Search answers to broad world knowledge questions.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 18,
      "entity_slug": "applebot",
      "entity_name": "Applebot",
      "field": "fallback_to_googlebot_rules",
      "value": "yes",
      "statement": "Applebot follows Googlebot robots.txt instructions when a robots.txt file mentions Googlebot but not Applebot.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 10,
      "entity_slug": "applebot",
      "entity_name": "Applebot",
      "field": "honours_crawl_delay",
      "value": "no",
      "statement": "Applebot does not follow the robots.txt Crawl-delay directive.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 150,
      "entity_slug": "applebot",
      "entity_name": "Applebot",
      "field": "observed_paths_fetched",
      "value": "robots_txt_and_home_only",
      "statement": "Applebot made 6 requests to Rattlesnakes By Mail between 2026-09-13T20:00Z and 2026-09-15T09:00Z, to 2 distinct paths, /robots.txt and the home page. Applebot's 6 requests to Rattlesnakes By Mail run from 2026-09-14T12:22:23Z to 2026-09-14T12:59:26Z and alternate the two paths three times.",
      "confidence": "medium",
      "verified_at": "2026-09-15",
      "status": "current"
    },
    {
      "id": 11,
      "entity_slug": "applebot",
      "entity_name": "Applebot",
      "field": "publishes_ip_list",
      "value": "yes",
      "statement": "Apple publishes the IP address CIDR ranges used by Applebot as a JSON file at https://search.developer.apple.com/applebot.json.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 13,
      "entity_slug": "applebot",
      "entity_name": "Applebot",
      "field": "purpose_documented",
      "value": "mixed",
      "statement": "Applebot crawls content for Apple search results and for training Apple foundation models.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 16,
      "entity_slug": "applebot",
      "entity_name": "Applebot",
      "field": "reads_llms_txt",
      "value": "not_documented",
      "statement": "Apple does not publish whether Applebot reads llms.txt files.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 9,
      "entity_slug": "applebot",
      "entity_name": "Applebot",
      "field": "respects_robots_txt",
      "value": "yes",
      "statement": "Applebot respects standard robots.txt directives targeted at Applebot in general search crawls.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 14,
      "entity_slug": "applebot",
      "entity_name": "Applebot",
      "field": "separate_search_and_training_tokens",
      "value": "partial",
      "statement": "Apple uses one crawler token, Applebot, for search crawling and for training crawling, and a second robots.txt token, Applebot-Extended, that opts content out of training use.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 17,
      "entity_slug": "applebot",
      "entity_name": "Applebot",
      "field": "user_agent_string_full",
      "value": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.4 Safari/605.1.15 (Applebot/0.1; +http://www.apple.com/go/applebot)",
      "statement": "Applebot sends the desktop user agent string Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.4 Safari/605.1.15 (Applebot/0.1; +http://www.apple.com/go/applebot).",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 12,
      "entity_slug": "applebot",
      "entity_name": "Applebot",
      "field": "verifiable_by_rdns",
      "value": "yes",
      "statement": "Applebot traffic is verifiable by reverse DNS lookup in the applebot.apple.com domain.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 2,
      "entity_slug": "applebot-extended",
      "entity_name": "Applebot-Extended",
      "field": "honours_crawl_delay",
      "value": "n/a",
      "statement": "Apple does not publish a Crawl-delay behaviour for Applebot-Extended.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 3,
      "entity_slug": "applebot-extended",
      "entity_name": "Applebot-Extended",
      "field": "publishes_ip_list",
      "value": "no",
      "statement": "Apple publishes no IP list specific to Applebot-Extended and documents one shared crawl identity for Applebot.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 5,
      "entity_slug": "applebot-extended",
      "entity_name": "Applebot-Extended",
      "field": "purpose_documented",
      "value": "policy_only",
      "statement": "Applebot-Extended controls whether website content is used to train Apple general purpose foundation models.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 1,
      "entity_slug": "applebot-extended",
      "entity_name": "Applebot-Extended",
      "field": "respects_robots_txt",
      "value": "n/a",
      "statement": "Applebot-Extended is a robots.txt token that publishers disallow to opt out of generative model training.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 8,
      "entity_slug": "applebot-extended",
      "entity_name": "Applebot-Extended",
      "field": "robots_token_differs_from_ua",
      "value": "yes",
      "statement": "Applebot-Extended is a robots.txt control token and does not correspond to a separate crawler user agent string.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 6,
      "entity_slug": "applebot-extended",
      "entity_name": "Applebot-Extended",
      "field": "separate_search_and_training_tokens",
      "value": "yes",
      "statement": "Webpages that disallow Applebot-Extended remain eligible for inclusion in Apple search results.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 4,
      "entity_slug": "applebot-extended",
      "entity_name": "Applebot-Extended",
      "field": "verifiable_by_rdns",
      "value": "n/a",
      "statement": "Apple does not publish a reverse-DNS verification method specific to Applebot-Extended.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 7,
      "entity_slug": "applebot-extended",
      "entity_name": "Applebot-Extended",
      "field": "what_it_controls",
      "value": "AI training use of content, not search indexing",
      "statement": "Applebot-Extended controls the AI training use of crawled content and does not control search indexing.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 56,
      "entity_slug": "google-extended",
      "entity_name": "Google-Extended",
      "field": "honours_crawl_delay",
      "value": "n/a",
      "statement": "Google does not publish any Crawl-delay behaviour for the Google-Extended token.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 57,
      "entity_slug": "google-extended",
      "entity_name": "Google-Extended",
      "field": "publishes_ip_list",
      "value": "no",
      "statement": "Google publishes no IP list for Google-Extended because Google-Extended has no separate HTTP request user agent string.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 59,
      "entity_slug": "google-extended",
      "entity_name": "Google-Extended",
      "field": "purpose_documented",
      "value": "policy_only",
      "statement": "Google-Extended is a standalone product token that controls whether Google uses crawled content to train Gemini models.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 55,
      "entity_slug": "google-extended",
      "entity_name": "Google-Extended",
      "field": "respects_robots_txt",
      "value": "n/a",
      "statement": "Google-Extended is a robots.txt token that publishers set to control use of crawled content for Gemini training.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 62,
      "entity_slug": "google-extended",
      "entity_name": "Google-Extended",
      "field": "robots_token_differs_from_ua",
      "value": "yes",
      "statement": "Google-Extended is a robots.txt control token and does not correspond to a crawler user agent string.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 60,
      "entity_slug": "google-extended",
      "entity_name": "Google-Extended",
      "field": "separate_search_and_training_tokens",
      "value": "yes",
      "statement": "Google-Extended controls Gemini training use only and does not affect inclusion in Google Search.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 58,
      "entity_slug": "google-extended",
      "entity_name": "Google-Extended",
      "field": "verifiable_by_rdns",
      "value": "n/a",
      "statement": "Google does not publish a reverse-DNS verification method for Google-Extended.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 61,
      "entity_slug": "google-extended",
      "entity_name": "Google-Extended",
      "field": "what_it_controls",
      "value": "Gemini model training and grounding, not Search inclusion",
      "statement": "Google-Extended controls AI training and grounding in Google systems and does not control Google Search inclusion.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 72,
      "entity_slug": "googleother",
      "entity_name": "GoogleOther",
      "field": "honours_crawl_delay",
      "value": "no",
      "statement": "Google does not support the robots.txt Crawl-delay directive for GoogleOther.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 147,
      "entity_slug": "googleother",
      "entity_name": "GoogleOther",
      "field": "observed_fetches_robots_txt",
      "value": "no",
      "statement": "GoogleOther made 93 requests to Rattlesnakes By Mail between 2026-09-13T20:00Z and 2026-09-15T09:00Z and fetched /robots.txt on none of those requests. GoogleOther's first request to Rattlesnakes By Mail was the home page at 2026-09-14T12:03:30Z.",
      "confidence": "high",
      "verified_at": "2026-09-15",
      "status": "current"
    },
    {
      "id": 146,
      "entity_slug": "googleother",
      "entity_name": "GoogleOther",
      "field": "observed_requested_mcp_endpoint",
      "value": "yes",
      "statement": "GoogleOther requested the /mcp endpoint of Rattlesnakes By Mail twice between 2026-09-13T20:00Z and 2026-09-15T09:00Z, at 2026-09-14T15:07:30Z and at 2026-09-14T18:27:49Z. Rattlesnakes By Mail returned status 405 to both GoogleOther requests for /mcp.",
      "confidence": "high",
      "verified_at": "2026-09-15",
      "status": "current"
    },
    {
      "id": 73,
      "entity_slug": "googleother",
      "entity_name": "GoogleOther",
      "field": "publishes_ip_list",
      "value": "yes",
      "statement": "Google publishes the IP address ranges used by GoogleOther at https://developers.google.com/static/crawling/ipranges/common-crawlers.json.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 75,
      "entity_slug": "googleother",
      "entity_name": "GoogleOther",
      "field": "purpose_documented",
      "value": "mixed",
      "statement": "GoogleOther is a generic crawler that Google product teams use to fetch publicly accessible content from sites.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 71,
      "entity_slug": "googleother",
      "entity_name": "GoogleOther",
      "field": "respects_robots_txt",
      "value": "yes",
      "statement": "GoogleOther obeys robots.txt rules when crawling automatically.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 77,
      "entity_slug": "googleother",
      "entity_name": "GoogleOther",
      "field": "robots_txt_token",
      "value": "GoogleOther",
      "statement": "The robots.txt token for GoogleOther is GoogleOther.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 76,
      "entity_slug": "googleother",
      "entity_name": "GoogleOther",
      "field": "separate_search_and_training_tokens",
      "value": "not_documented",
      "statement": "Google does not document whether content fetched by GoogleOther is used for Gemini training.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 74,
      "entity_slug": "googleother",
      "entity_name": "GoogleOther",
      "field": "verifiable_by_rdns",
      "value": "yes",
      "statement": "GoogleOther requests are verifiable by reverse DNS lookup resolving to googlebot.com, google.com, or googleusercontent.com.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 69,
      "entity_slug": "googlebot",
      "entity_name": "Googlebot",
      "field": "cites_sources",
      "value": "yes",
      "statement": "Google AI Overviews and AI Mode return AI generated responses with links to supporting websites.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 64,
      "entity_slug": "googlebot",
      "entity_name": "Googlebot",
      "field": "honours_crawl_delay",
      "value": "no",
      "statement": "Google does not support the robots.txt Crawl-delay directive for Googlebot.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 149,
      "entity_slug": "googlebot",
      "entity_name": "Googlebot",
      "field": "observed_verified_request_ratio_census",
      "value": "0.991",
      "statement": "Googlebot requests to Rattlesnakes By Mail matched Google's published IP ranges on 108 of 109 requests between 2026-09-13T20:00Z and 2026-09-15T09:00Z. Googlebot's first request to Rattlesnakes By Mail in that window was /robots.txt at 2026-09-14T11:53:00Z.",
      "confidence": "high",
      "verified_at": "2026-09-15",
      "status": "current"
    },
    {
      "id": 65,
      "entity_slug": "googlebot",
      "entity_name": "Googlebot",
      "field": "publishes_ip_list",
      "value": "yes",
      "statement": "Google publishes the IP address ranges used by Googlebot at https://developers.google.com/static/crawling/ipranges/common-crawlers.json.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 67,
      "entity_slug": "googlebot",
      "entity_name": "Googlebot",
      "field": "purpose_documented",
      "value": "search",
      "statement": "Googlebot crawls pages for Google Search, Google Images, Google Video, Google News, and Discover.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 132,
      "entity_slug": "googlebot",
      "entity_name": "Googlebot",
      "field": "reads_llms_txt",
      "value": "no",
      "statement": "Google Search does not use llms.txt files.",
      "confidence": "high",
      "verified_at": "2026-09-14",
      "status": "current"
    },
    {
      "id": 63,
      "entity_slug": "googlebot",
      "entity_name": "Googlebot",
      "field": "respects_robots_txt",
      "value": "yes",
      "statement": "Googlebot obeys robots.txt rules when crawling automatically.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 68,
      "entity_slug": "googlebot",
      "entity_name": "Googlebot",
      "field": "separate_search_and_training_tokens",
      "value": "yes",
      "statement": "Google separates Search crawling under the Googlebot token from Gemini training under the Google-Extended token.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 70,
      "entity_slug": "googlebot",
      "entity_name": "Googlebot",
      "field": "user_agent_string_full",
      "value": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Googlebot/2.1; +http://www.google.com/bot.html)",
      "statement": "Googlebot sends the user agent string Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Googlebot/2.1; +http://www.google.com/bot.html).",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 66,
      "entity_slug": "googlebot",
      "entity_name": "Googlebot",
      "field": "verifiable_by_rdns",
      "value": "yes",
      "statement": "Googlebot requests are verifiable by reverse DNS lookup resolving to googlebot.com, google.com, or googleusercontent.com.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 127,
      "entity_slug": "bingbot",
      "entity_name": "Bingbot",
      "field": "cites_sources",
      "value": "yes",
      "statement": "Copilot Search in Bing delivers clearly cited sources alongside generative answers to support publishers and content owners.",
      "confidence": "high",
      "verified_at": "2026-09-14",
      "status": "current"
    },
    {
      "id": 131,
      "entity_slug": "bingbot",
      "entity_name": "Bingbot",
      "field": "follows_noarchive",
      "value": "yes",
      "statement": "Bingbot excludes noarchive-tagged content from Copilot and Chat answers and from generative AI training data.",
      "confidence": "high",
      "verified_at": "2026-09-14",
      "status": "current"
    },
    {
      "id": 122,
      "entity_slug": "bingbot",
      "entity_name": "Bingbot",
      "field": "honours_crawl_delay",
      "value": "yes",
      "statement": "Microsoft states that a crawl-delay directive in robots.txt always takes precedence over Bing Webmaster Tools crawl control settings for Bingbot.",
      "confidence": "high",
      "verified_at": "2026-09-14",
      "status": "current"
    },
    {
      "id": 130,
      "entity_slug": "bingbot",
      "entity_name": "Bingbot",
      "field": "opt_out_mechanism",
      "value": "yes",
      "statement": "Microsoft lets publishers block Bingbot content from Copilot answers and AI training using noarchive and nocache tags.",
      "confidence": "high",
      "verified_at": "2026-09-14",
      "status": "current"
    },
    {
      "id": 123,
      "entity_slug": "bingbot",
      "entity_name": "Bingbot",
      "field": "publishes_ip_list",
      "value": "yes",
      "statement": "Microsoft publishes Bingbot's IP address ranges as a JSON file at bing.com/toolbox/bingbot.json for verification purposes.",
      "confidence": "high",
      "verified_at": "2026-09-14",
      "status": "current"
    },
    {
      "id": 125,
      "entity_slug": "bingbot",
      "entity_name": "Bingbot",
      "field": "purpose_documented",
      "value": "yes",
      "statement": "Bingbot serves as Microsoft's standard crawler and handles most of Bing's daily web crawling.",
      "confidence": "high",
      "verified_at": "2026-09-14",
      "status": "current"
    },
    {
      "id": 128,
      "entity_slug": "bingbot",
      "entity_name": "Bingbot",
      "field": "reads_llms_txt",
      "value": "not_documented",
      "statement": "Microsoft does not publish whether Bingbot reads llms.txt files.",
      "confidence": null,
      "verified_at": "2026-09-14",
      "status": "current"
    },
    {
      "id": 121,
      "entity_slug": "bingbot",
      "entity_name": "Bingbot",
      "field": "respects_robots_txt",
      "value": "yes",
      "statement": "Bingbot honors robots.txt directives and other Microsoft-supported control mechanisms that content owners set for crawling.",
      "confidence": "high",
      "verified_at": "2026-09-14",
      "status": "current"
    },
    {
      "id": 126,
      "entity_slug": "bingbot",
      "entity_name": "Bingbot",
      "field": "separate_search_and_training_tokens",
      "value": "no",
      "statement": "Microsoft controls Bingbot's use in AI training through meta tags rather than publishing a separate training-specific crawler token.",
      "confidence": "high",
      "verified_at": "2026-09-14",
      "status": "current"
    },
    {
      "id": 129,
      "entity_slug": "bingbot",
      "entity_name": "Bingbot",
      "field": "user_agent_string_full",
      "value": "Mozilla/5.0 (compatible; bingbot/2.0; +http://www.bing.com/bingbot.htm)",
      "statement": "Bingbot sends a user agent string containing bingbot/2.0 and a link to Microsoft's bingbot.htm page.",
      "confidence": "high",
      "verified_at": "2026-09-14",
      "status": "current"
    },
    {
      "id": 124,
      "entity_slug": "bingbot",
      "entity_name": "Bingbot",
      "field": "verifiable_by_rdns",
      "value": "yes",
      "statement": "Microsoft verifies Bingbot traffic through reverse DNS lookups that resolve to hostnames ending in search.msn.com.",
      "confidence": "high",
      "verified_at": "2026-09-14",
      "status": "current"
    },
    {
      "id": 25,
      "entity_slug": "chatgpt-user",
      "entity_name": "ChatGPT-User",
      "field": "cites_sources",
      "value": "not_documented",
      "statement": "OpenAI does not publish whether ChatGPT responses cite pages fetched by ChatGPT-User.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 20,
      "entity_slug": "chatgpt-user",
      "entity_name": "ChatGPT-User",
      "field": "honours_crawl_delay",
      "value": "not_documented",
      "statement": "OpenAI does not publish whether ChatGPT-User honours the robots.txt Crawl-delay directive.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 21,
      "entity_slug": "chatgpt-user",
      "entity_name": "ChatGPT-User",
      "field": "publishes_ip_list",
      "value": "yes",
      "statement": "OpenAI publishes the IP address ranges used by ChatGPT-User as a JSON file at https://openai.com/chatgpt-user.json.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 23,
      "entity_slug": "chatgpt-user",
      "entity_name": "ChatGPT-User",
      "field": "purpose_documented",
      "value": "user_fetch",
      "statement": "ChatGPT-User visits web pages for user actions in ChatGPT and in Custom GPTs.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 26,
      "entity_slug": "chatgpt-user",
      "entity_name": "ChatGPT-User",
      "field": "reads_llms_txt",
      "value": "not_documented",
      "statement": "OpenAI does not publish whether ChatGPT-User reads llms.txt files.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 19,
      "entity_slug": "chatgpt-user",
      "entity_name": "ChatGPT-User",
      "field": "respects_robots_txt",
      "value": "partial",
      "statement": "ChatGPT-User fetches pages in response to user actions in ChatGPT, and OpenAI states that robots.txt rules may not apply to ChatGPT-User fetches.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 24,
      "entity_slug": "chatgpt-user",
      "entity_name": "ChatGPT-User",
      "field": "separate_search_and_training_tokens",
      "value": "not_documented",
      "statement": "OpenAI does not document ChatGPT-User as a search or training opt-out token.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 27,
      "entity_slug": "chatgpt-user",
      "entity_name": "ChatGPT-User",
      "field": "user_agent_string_full",
      "value": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; ChatGPT-User/1.0; +https://openai.com/bot",
      "statement": "ChatGPT-User sends the user agent string Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; ChatGPT-User/1.0; +https://openai.com/bot.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 22,
      "entity_slug": "chatgpt-user",
      "entity_name": "ChatGPT-User",
      "field": "verifiable_by_rdns",
      "value": "not_documented",
      "statement": "OpenAI does not publish a reverse-DNS verification method for ChatGPT-User.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 79,
      "entity_slug": "gptbot",
      "entity_name": "GPTBot",
      "field": "honours_crawl_delay",
      "value": "not_documented",
      "statement": "OpenAI does not publish whether GPTBot honours the robots.txt Crawl-delay directive.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 80,
      "entity_slug": "gptbot",
      "entity_name": "GPTBot",
      "field": "publishes_ip_list",
      "value": "yes",
      "statement": "OpenAI publishes the IP address ranges used by GPTBot as a JSON file at https://openai.com/gptbot.json.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 82,
      "entity_slug": "gptbot",
      "entity_name": "GPTBot",
      "field": "purpose_documented",
      "value": "training",
      "statement": "GPTBot crawls content that may be used to train OpenAI generative AI foundation models.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 78,
      "entity_slug": "gptbot",
      "entity_name": "GPTBot",
      "field": "respects_robots_txt",
      "value": "yes",
      "statement": "GPTBot honours robots.txt rules that webmasters set for the GPTBot token.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 85,
      "entity_slug": "gptbot",
      "entity_name": "GPTBot",
      "field": "robots_txt_fetch_marker",
      "value": "yes",
      "statement": "OpenAI may add a robots.txt marker to the GPTBot user agent string when GPTBot fetches robots.txt files.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 83,
      "entity_slug": "gptbot",
      "entity_name": "GPTBot",
      "field": "separate_search_and_training_tokens",
      "value": "yes",
      "statement": "OpenAI separates search and training controls into two robots.txt tokens: OAI-SearchBot for search and GPTBot for training.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 84,
      "entity_slug": "gptbot",
      "entity_name": "GPTBot",
      "field": "user_agent_string_full",
      "value": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
      "statement": "GPTBot sends the user agent string Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 81,
      "entity_slug": "gptbot",
      "entity_name": "GPTBot",
      "field": "verifiable_by_rdns",
      "value": "not_documented",
      "statement": "OpenAI does not publish a reverse-DNS verification method for GPTBot.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 87,
      "entity_slug": "oai-adsbot",
      "entity_name": "OAI-AdsBot",
      "field": "honours_crawl_delay",
      "value": "not_documented",
      "statement": "OpenAI does not publish whether OAI-AdsBot honours the robots.txt Crawl-delay directive.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 88,
      "entity_slug": "oai-adsbot",
      "entity_name": "OAI-AdsBot",
      "field": "publishes_ip_list",
      "value": "yes",
      "statement": "OpenAI publishes the IP address ranges used by OAI-AdsBot as a JSON file at https://openai.com/adsbot.json.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 90,
      "entity_slug": "oai-adsbot",
      "entity_name": "OAI-AdsBot",
      "field": "purpose_documented",
      "value": "ads",
      "statement": "OAI-AdsBot visits landing pages submitted as ads on ChatGPT to check compliance with OpenAI policies.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 86,
      "entity_slug": "oai-adsbot",
      "entity_name": "OAI-AdsBot",
      "field": "respects_robots_txt",
      "value": "not_documented",
      "statement": "OpenAI does not publish whether OAI-AdsBot honours robots.txt disallow rules.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 93,
      "entity_slug": "oai-adsbot",
      "entity_name": "OAI-AdsBot",
      "field": "scope_of_crawling",
      "value": "limited to submitted ad landing pages",
      "statement": "OAI-AdsBot visits only pages submitted as ads on ChatGPT.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 91,
      "entity_slug": "oai-adsbot",
      "entity_name": "OAI-AdsBot",
      "field": "separate_search_and_training_tokens",
      "value": "not_documented",
      "statement": "OpenAI does not document OAI-AdsBot as part of a search and training token split.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 92,
      "entity_slug": "oai-adsbot",
      "entity_name": "OAI-AdsBot",
      "field": "user_agent_string_full",
      "value": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; OAI-AdsBot/1.0; +https://openai.com/adsbot",
      "statement": "OAI-AdsBot sends the user agent string Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; OAI-AdsBot/1.0; +https://openai.com/adsbot.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 89,
      "entity_slug": "oai-adsbot",
      "entity_name": "OAI-AdsBot",
      "field": "verifiable_by_rdns",
      "value": "not_documented",
      "statement": "OpenAI does not publish a reverse-DNS verification method for OAI-AdsBot.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 100,
      "entity_slug": "oai-searchbot",
      "entity_name": "OAI-SearchBot",
      "field": "cites_sources",
      "value": "yes",
      "statement": "OAI-SearchBot surfaces crawled websites as links in ChatGPT search answers.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 95,
      "entity_slug": "oai-searchbot",
      "entity_name": "OAI-SearchBot",
      "field": "honours_crawl_delay",
      "value": "not_documented",
      "statement": "OpenAI does not publish whether OAI-SearchBot honours the robots.txt Crawl-delay directive.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 148,
      "entity_slug": "oai-searchbot",
      "entity_name": "OAI-SearchBot",
      "field": "observed_arrival_before_sitemap",
      "value": "yes",
      "statement": "OAI-SearchBot first requested Rattlesnakes By Mail at 2026-09-13T21:41:14Z, before the Rattlesnakes By Mail sitemap was submitted to Bing Webmaster Tools shortly before 2026-09-15T08:38Z and to Google Search Console at 2026-09-15T09:00Z. OAI-SearchBot made 3 requests to Rattlesnakes By Mail between 2026-09-13T20:00Z and 2026-09-15T09:00Z, all 3 to /robots.txt.",
      "confidence": "medium",
      "verified_at": "2026-09-15",
      "status": "current"
    },
    {
      "id": 96,
      "entity_slug": "oai-searchbot",
      "entity_name": "OAI-SearchBot",
      "field": "publishes_ip_list",
      "value": "yes",
      "statement": "OpenAI publishes the IP address ranges used by OAI-SearchBot as a JSON file at https://openai.com/searchbot.json.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 98,
      "entity_slug": "oai-searchbot",
      "entity_name": "OAI-SearchBot",
      "field": "purpose_documented",
      "value": "search",
      "statement": "OAI-SearchBot surfaces websites in search results in ChatGPT search features.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 101,
      "entity_slug": "oai-searchbot",
      "entity_name": "OAI-SearchBot",
      "field": "reads_llms_txt",
      "value": "not_documented",
      "statement": "OpenAI does not publish whether OAI-SearchBot reads llms.txt files.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 94,
      "entity_slug": "oai-searchbot",
      "entity_name": "OAI-SearchBot",
      "field": "respects_robots_txt",
      "value": "yes",
      "statement": "OAI-SearchBot honours robots.txt opt-outs set for the OAI-SearchBot token.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 99,
      "entity_slug": "oai-searchbot",
      "entity_name": "OAI-SearchBot",
      "field": "separate_search_and_training_tokens",
      "value": "yes",
      "statement": "OpenAI separates search and training controls into two robots.txt tokens: OAI-SearchBot for search and GPTBot for training.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 102,
      "entity_slug": "oai-searchbot",
      "entity_name": "OAI-SearchBot",
      "field": "user_agent_string_full",
      "value": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
      "statement": "OAI-SearchBot sends the user agent string Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 97,
      "entity_slug": "oai-searchbot",
      "entity_name": "OAI-SearchBot",
      "field": "verifiable_by_rdns",
      "value": "not_documented",
      "statement": "OpenAI does not publish a reverse-DNS verification method for OAI-SearchBot.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 109,
      "entity_slug": "perplexity-user",
      "entity_name": "Perplexity-User",
      "field": "cites_sources",
      "value": "yes",
      "statement": "Perplexity includes a link to each page that Perplexity-User visits in the Perplexity response.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 104,
      "entity_slug": "perplexity-user",
      "entity_name": "Perplexity-User",
      "field": "honours_crawl_delay",
      "value": "not_documented",
      "statement": "Perplexity does not publish whether Perplexity-User honours the robots.txt Crawl-delay directive.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 105,
      "entity_slug": "perplexity-user",
      "entity_name": "Perplexity-User",
      "field": "publishes_ip_list",
      "value": "yes",
      "statement": "Perplexity publishes the IP address ranges used by Perplexity-User as a JSON file at https://www.perplexity.ai/perplexity-user.json.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 107,
      "entity_slug": "perplexity-user",
      "entity_name": "Perplexity-User",
      "field": "purpose_documented",
      "value": "user_fetch",
      "statement": "Perplexity-User visits web pages to answer questions that users ask Perplexity.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 110,
      "entity_slug": "perplexity-user",
      "entity_name": "Perplexity-User",
      "field": "reads_llms_txt",
      "value": "not_documented",
      "statement": "Perplexity does not publish whether Perplexity-User reads llms.txt files.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 103,
      "entity_slug": "perplexity-user",
      "entity_name": "Perplexity-User",
      "field": "respects_robots_txt",
      "value": "no",
      "statement": "Perplexity-User ignores robots.txt rules because a user requests each Perplexity-User fetch.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 108,
      "entity_slug": "perplexity-user",
      "entity_name": "Perplexity-User",
      "field": "separate_search_and_training_tokens",
      "value": "not_documented",
      "statement": "Perplexity does not document Perplexity-User as a search or training opt-out token.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 111,
      "entity_slug": "perplexity-user",
      "entity_name": "Perplexity-User",
      "field": "user_agent_string_full",
      "value": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Perplexity-User/1.0; +https://perplexity.ai/perplexity-user)",
      "statement": "Perplexity-User sends the user agent string Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Perplexity-User/1.0; +https://perplexity.ai/perplexity-user).",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 106,
      "entity_slug": "perplexity-user",
      "entity_name": "Perplexity-User",
      "field": "verifiable_by_rdns",
      "value": "not_documented",
      "statement": "Perplexity does not publish a reverse-DNS verification method for Perplexity-User.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 118,
      "entity_slug": "perplexitybot",
      "entity_name": "PerplexityBot",
      "field": "cites_sources",
      "value": "yes",
      "statement": "Perplexity links to the websites that PerplexityBot surfaces in Perplexity search results.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 113,
      "entity_slug": "perplexitybot",
      "entity_name": "PerplexityBot",
      "field": "honours_crawl_delay",
      "value": "not_documented",
      "statement": "Perplexity does not publish whether PerplexityBot honours the robots.txt Crawl-delay directive.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 114,
      "entity_slug": "perplexitybot",
      "entity_name": "PerplexityBot",
      "field": "publishes_ip_list",
      "value": "yes",
      "statement": "Perplexity publishes the IP address ranges used by PerplexityBot as a JSON file at https://www.perplexity.ai/perplexitybot.json.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 116,
      "entity_slug": "perplexitybot",
      "entity_name": "PerplexityBot",
      "field": "purpose_documented",
      "value": "search",
      "statement": "PerplexityBot surfaces and links websites in Perplexity search results.",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 119,
      "entity_slug": "perplexitybot",
      "entity_name": "PerplexityBot",
      "field": "reads_llms_txt",
      "value": "not_documented",
      "statement": "Perplexity does not publish whether PerplexityBot reads llms.txt files.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 112,
      "entity_slug": "perplexitybot",
      "entity_name": "PerplexityBot",
      "field": "respects_robots_txt",
      "value": "yes",
      "statement": "PerplexityBot honours robots.txt rules that webmasters set for the PerplexityBot token.",
      "confidence": "medium",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 117,
      "entity_slug": "perplexitybot",
      "entity_name": "PerplexityBot",
      "field": "separate_search_and_training_tokens",
      "value": "not_documented",
      "statement": "Perplexity does not document a separate training crawler token alongside PerplexityBot.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 120,
      "entity_slug": "perplexitybot",
      "entity_name": "PerplexityBot",
      "field": "user_agent_string_full",
      "value": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
      "statement": "PerplexityBot sends the user agent string Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot).",
      "confidence": "high",
      "verified_at": "2026-09-13",
      "status": "current"
    },
    {
      "id": 115,
      "entity_slug": "perplexitybot",
      "entity_name": "PerplexityBot",
      "field": "verifiable_by_rdns",
      "value": "not_documented",
      "statement": "Perplexity does not publish a reverse-DNS verification method for PerplexityBot.",
      "confidence": null,
      "verified_at": "2026-09-13",
      "status": "current"
    }
  ],
  "definition": "The Claims ledger of Rattlesnakes By Mail holds every fact published on Rattlesnakes By Mail, one claim per row. A claim is one atomic dated statement about one entity, carrying a verbatim vendor quote of 300 characters or fewer, a source URL, a verification method and a confidence. A published claim is never edited. A correction creates a new claim that supersedes the older claim, both claims stay addressable at their own URLs, and every supersession appears at /changes."
}