[
  {
    "id": 1,
    "entity_id": 15,
    "slug": "applebot-extended/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "n/a",
    "statement": "Applebot-Extended is a robots.txt token that publishers disallow to opt out of generative model training.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": "\"Web publishers can opt-out from having their content used to train generative foundation models by disallowing Applebot-Extended in the robots.txt file.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 2,
    "entity_id": 15,
    "slug": "applebot-extended/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "n/a",
    "statement": "Apple does not publish a Crawl-delay behaviour for Applebot-Extended.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 3,
    "entity_id": 15,
    "slug": "applebot-extended/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "no",
    "statement": "Apple publishes no IP list specific to Applebot-Extended and documents one shared crawl identity for Applebot.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": "\"Traffic coming from Applebot is generally identified by using reverse DNS in the *.applebot.apple.com domain\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 4,
    "entity_id": 15,
    "slug": "applebot-extended/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "n/a",
    "statement": "Apple does not publish a reverse-DNS verification method specific to Applebot-Extended.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 5,
    "entity_id": 15,
    "slug": "applebot-extended/purpose_documented",
    "field": "purpose_documented",
    "value": "policy_only",
    "statement": "Applebot-Extended controls whether website content is used to train Apple general purpose foundation models.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": "\"opt out of their website content being used to train Apple's general purpose foundation models powering generative AI features\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 6,
    "entity_id": 15,
    "slug": "applebot-extended/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "yes",
    "statement": "Webpages that disallow Applebot-Extended remain eligible for inclusion in Apple search results.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": "\"Webpages that disallow Applebot-Extended can still be included in search results.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 7,
    "entity_id": 15,
    "slug": "applebot-extended/what_it_controls",
    "field": "what_it_controls",
    "value": "AI training use of content, not search indexing",
    "statement": "Applebot-Extended controls the AI training use of crawled content and does not control search indexing.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": "\"Web publishers can opt-out from having their content used to train generative foundation models by disallowing Applebot-Extended in the robots.txt file. Applebot crawled data may be used to provide additional context and up-to-date content when AI models are used to generate output ...\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 8,
    "entity_id": 15,
    "slug": "applebot-extended/robots_token_differs_from_ua",
    "field": "robots_token_differs_from_ua",
    "value": "yes",
    "statement": "Applebot-Extended is a robots.txt control token and does not correspond to a separate crawler user agent string.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 9,
    "entity_id": 14,
    "slug": "applebot/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "yes",
    "statement": "Applebot respects standard robots.txt directives targeted at Applebot in general search crawls.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": "\"Applebot respects standard robots.txt directives in general search crawls that are targeted at Applebot.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 10,
    "entity_id": 14,
    "slug": "applebot/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "no",
    "statement": "Applebot does not follow the robots.txt Crawl-delay directive.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": "\"Applebot does not follow crawl-delay. Applebot is engineered for efficiency and will adjust to minimize the impact on site owners.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 11,
    "entity_id": 14,
    "slug": "applebot/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "yes",
    "statement": "Apple publishes the IP address CIDR ranges used by Applebot as a JSON file at https://search.developer.apple.com/applebot.json.",
    "evidence_url": "https://search.developer.apple.com/applebot.json",
    "evidence_quote": "\"Another way is to match the IP address with a CIDR prefix contained in the following JSON file: Applebot IP CIDRs.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-31"
  },
  {
    "id": 12,
    "entity_id": 14,
    "slug": "applebot/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "yes",
    "statement": "Applebot traffic is verifiable by reverse DNS lookup in the applebot.apple.com domain.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": "\"Traffic coming from Applebot is generally identified by using reverse DNS in the *.applebot.apple.com domain.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 13,
    "entity_id": 14,
    "slug": "applebot/purpose_documented",
    "field": "purpose_documented",
    "value": "mixed",
    "statement": "Applebot crawls content for Apple search results and for training Apple foundation models.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": "\"Enabling Applebot in robots.txt allows website content to appear in search results for Apple users ... The data crawled by Applebot may also be used to help train Apple foundation models powering generative AI features across Apple products.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 14,
    "entity_id": 14,
    "slug": "applebot/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "partial",
    "statement": "Apple uses one crawler token, Applebot, for search crawling and for training crawling, and a second robots.txt token, Applebot-Extended, that opts content out of training use.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": "\"Web publishers can opt-out from having their content used to train generative foundation models by disallowing Applebot-Extended in the robots.txt file.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 15,
    "entity_id": 14,
    "slug": "applebot/cites_sources",
    "field": "cites_sources",
    "value": "yes",
    "statement": "Apple includes links to source websites in Siri and Search answers to broad world knowledge questions.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": "\"answering broad world knowledge questions in Siri and Search that may include links to sources and websites used to help\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 16,
    "entity_id": 14,
    "slug": "applebot/reads_llms_txt",
    "field": "reads_llms_txt",
    "value": "not_documented",
    "statement": "Apple does not publish whether Applebot reads llms.txt files.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 17,
    "entity_id": 14,
    "slug": "applebot/user_agent_string_full",
    "field": "user_agent_string_full",
    "value": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.4 Safari/605.1.15 (Applebot/0.1; +http://www.apple.com/go/applebot)",
    "statement": "Applebot sends the desktop user agent string Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.4 Safari/605.1.15 (Applebot/0.1; +http://www.apple.com/go/applebot).",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 18,
    "entity_id": 14,
    "slug": "applebot/fallback_to_googlebot_rules",
    "field": "fallback_to_googlebot_rules",
    "value": "yes",
    "statement": "Applebot follows Googlebot robots.txt instructions when a robots.txt file mentions Googlebot but not Applebot.",
    "evidence_url": "https://support.apple.com/en-us/119829",
    "evidence_quote": "\"If robots instructions don't mention Applebot but mention Googlebot, the Apple robot will follow Googlebot instructions.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-04"
  },
  {
    "id": 19,
    "entity_id": 3,
    "slug": "chatgpt-user/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "partial",
    "statement": "ChatGPT-User fetches pages in response to user actions in ChatGPT, and OpenAI states that robots.txt rules may not apply to ChatGPT-User fetches.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"ChatGPT-User is not used for crawling the web in an automatic fashion. Because these actions are initiated by a user, robots.txt rules may not apply.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 20,
    "entity_id": 3,
    "slug": "chatgpt-user/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "not_documented",
    "statement": "OpenAI does not publish whether ChatGPT-User honours the robots.txt Crawl-delay directive.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 21,
    "entity_id": 3,
    "slug": "chatgpt-user/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "yes",
    "statement": "OpenAI publishes the IP address ranges used by ChatGPT-User as a JSON file at https://openai.com/chatgpt-user.json.",
    "evidence_url": "https://openai.com/chatgpt-user.json",
    "evidence_quote": "\"Published IP addresses: https://openai.com/chatgpt-user.json\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-11"
  },
  {
    "id": 22,
    "entity_id": 3,
    "slug": "chatgpt-user/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "not_documented",
    "statement": "OpenAI does not publish a reverse-DNS verification method for ChatGPT-User.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 23,
    "entity_id": 3,
    "slug": "chatgpt-user/purpose_documented",
    "field": "purpose_documented",
    "value": "user_fetch",
    "statement": "ChatGPT-User visits web pages for user actions in ChatGPT and in Custom GPTs.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"OpenAI also uses ChatGPT-User for certain user actions in ChatGPT and Custom GPTs. When users ask ChatGPT or a CustomGPT a question, it may visit a web page with a ChatGPT-User agent.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 24,
    "entity_id": 3,
    "slug": "chatgpt-user/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "not_documented",
    "statement": "OpenAI does not document ChatGPT-User as a search or training opt-out token.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"ChatGPT-User is not used to determine whether content may appear in Search. Please use OAI-SearchBot in robots.txt for managing Search opt outs and automatic crawl.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 25,
    "entity_id": 3,
    "slug": "chatgpt-user/cites_sources",
    "field": "cites_sources",
    "value": "not_documented",
    "statement": "OpenAI does not publish whether ChatGPT responses cite pages fetched by ChatGPT-User.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 26,
    "entity_id": 3,
    "slug": "chatgpt-user/reads_llms_txt",
    "field": "reads_llms_txt",
    "value": "not_documented",
    "statement": "OpenAI does not publish whether ChatGPT-User reads llms.txt files.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 27,
    "entity_id": 3,
    "slug": "chatgpt-user/user_agent_string_full",
    "field": "user_agent_string_full",
    "value": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; ChatGPT-User/1.0; +https://openai.com/bot",
    "statement": "ChatGPT-User sends the user agent string Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; ChatGPT-User/1.0; +https://openai.com/bot.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"Full user-agent string: Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; ChatGPT-User/1.0; +https://openai.com/bot\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 28,
    "entity_id": 6,
    "slug": "claude-searchbot/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "yes",
    "statement": "Claude-SearchBot honours industry standard robots.txt directives that signal do not crawl.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"Anthropic's Bots respect \"do not crawl\" signals by honoring industry standard directives in robots.txt.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 29,
    "entity_id": 6,
    "slug": "claude-searchbot/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "yes",
    "statement": "Claude-SearchBot respects the Crawl-delay directive in robots.txt when crawling domains.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"We aim for minimal disruption by being thoughtful about how quickly we crawl the same domains and respecting Crawl-delay where appropriate.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 30,
    "entity_id": 6,
    "slug": "claude-searchbot/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "yes",
    "statement": "Anthropic publishes the IP address ranges used by Claude-SearchBot as a JSON file at https://claude.com/crawling/bots.json.",
    "evidence_url": "https://claude.com/crawling/bots.json",
    "evidence_quote": "\"If a crawler has a source IP address on this list, it indicates that the crawler is coming from Anthropic.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-08-18"
  },
  {
    "id": 31,
    "entity_id": 6,
    "slug": "claude-searchbot/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "not_documented",
    "statement": "Anthropic does not publish a reverse-DNS verification method for Claude-SearchBot.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 32,
    "entity_id": 6,
    "slug": "claude-searchbot/purpose_documented",
    "field": "purpose_documented",
    "value": "search",
    "statement": "Claude-SearchBot analyses online content to improve the relevance and accuracy of Claude search responses.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"Claude-SearchBot navigates the web to improve search result quality for users. It analyzes online content specifically to enhance the relevance and accuracy of search responses.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 33,
    "entity_id": 6,
    "slug": "claude-searchbot/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "yes",
    "statement": "Anthropic separates search and training controls into two robots.txt tokens: Claude-SearchBot for search and ClaudeBot for training.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"Disabling Claude-SearchBot on your site prevents our system from indexing your content for search optimization ... \"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 34,
    "entity_id": 6,
    "slug": "claude-searchbot/cites_sources",
    "field": "cites_sources",
    "value": "not_documented",
    "statement": "Anthropic does not publish whether Claude search responses cite pages crawled by Claude-SearchBot.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 35,
    "entity_id": 6,
    "slug": "claude-searchbot/reads_llms_txt",
    "field": "reads_llms_txt",
    "value": "not_documented",
    "statement": "Anthropic does not publish whether Claude-SearchBot reads llms.txt files.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 36,
    "entity_id": 6,
    "slug": "claude-searchbot/user_agent_string_full",
    "field": "user_agent_string_full",
    "value": "not_documented",
    "statement": "Anthropic does not publish a full user agent string for Claude-SearchBot.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 37,
    "entity_id": 7,
    "slug": "claude-user/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "yes",
    "statement": "Claude-User honours industry standard robots.txt directives that signal do not crawl.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"Anthropic's Bots respect \"do not crawl\" signals by honoring industry standard directives in robots.txt.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 38,
    "entity_id": 7,
    "slug": "claude-user/user_fetch_exception_documented",
    "field": "user_fetch_exception_documented",
    "value": "no",
    "statement": "Anthropic applies the same robots.txt statement to Claude-User as to ClaudeBot and Claude-SearchBot and documents no exception for user-initiated fetches.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"Anthropic's Bots respect \"do not crawl\" signals by honoring industry standard directives in robots.txt.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 39,
    "entity_id": 7,
    "slug": "claude-user/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "yes",
    "statement": "Claude-User respects the Crawl-delay directive in robots.txt when fetching pages.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"We aim for minimal disruption by being thoughtful about how quickly we crawl the same domains and respecting Crawl-delay where appropriate.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 40,
    "entity_id": 7,
    "slug": "claude-user/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "yes",
    "statement": "Anthropic publishes the IP address ranges used by Claude-User as a JSON file at https://claude.com/crawling/bots.json.",
    "evidence_url": "https://claude.com/crawling/bots.json",
    "evidence_quote": "\"If a crawler has a source IP address on this list, it indicates that the crawler is coming from Anthropic.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-08-18"
  },
  {
    "id": 41,
    "entity_id": 7,
    "slug": "claude-user/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "not_documented",
    "statement": "Anthropic does not publish a reverse-DNS verification method for Claude-User.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 42,
    "entity_id": 7,
    "slug": "claude-user/purpose_documented",
    "field": "purpose_documented",
    "value": "user_fetch",
    "statement": "Claude-User accesses websites in response to questions that individual users ask Claude.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"Claude-User supports Claude AI users. When individuals ask questions to Claude, it may access websites using a Claude-User agent.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 43,
    "entity_id": 7,
    "slug": "claude-user/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "not_documented",
    "statement": "Anthropic does not document Claude-User as a search or training opt-out token.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"Claude-User allows site owners to control which sites can be accessed through these user-initiated requests.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 44,
    "entity_id": 7,
    "slug": "claude-user/cites_sources",
    "field": "cites_sources",
    "value": "not_documented",
    "statement": "Anthropic does not publish whether Claude responses cite pages fetched by Claude-User.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 45,
    "entity_id": 7,
    "slug": "claude-user/reads_llms_txt",
    "field": "reads_llms_txt",
    "value": "not_documented",
    "statement": "Anthropic does not publish whether Claude-User reads llms.txt files.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 46,
    "entity_id": 7,
    "slug": "claude-user/user_agent_string_full",
    "field": "user_agent_string_full",
    "value": "not_documented",
    "statement": "Anthropic does not publish a full user agent string for Claude-User.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 47,
    "entity_id": 5,
    "slug": "claudebot/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "yes",
    "statement": "ClaudeBot honours industry standard robots.txt directives that signal do not crawl.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"Anthropic's Bots respect \"do not crawl\" signals by honoring industry standard directives in robots.txt.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 48,
    "entity_id": 5,
    "slug": "claudebot/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "yes",
    "statement": "Anthropic supports the non-standard robots.txt Crawl-delay extension for ClaudeBot.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"To limit crawling activity, we support the non-standard Crawl-delay extension to robots.txt. An example of this might be: User-agent: ClaudeBot Crawl-delay: 1\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 49,
    "entity_id": 5,
    "slug": "claudebot/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "yes",
    "statement": "Anthropic publishes the IP address ranges used by ClaudeBot as a JSON file at https://claude.com/crawling/bots.json.",
    "evidence_url": "https://claude.com/crawling/bots.json",
    "evidence_quote": "\"If a crawler has a source IP address on this list, it indicates that the crawler is coming from Anthropic.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-08-18"
  },
  {
    "id": 50,
    "entity_id": 5,
    "slug": "claudebot/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "not_documented",
    "statement": "Anthropic does not publish a reverse-DNS verification method for ClaudeBot.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"Alternate methods like blocking IP address(es) from which Anthropic Bots operates may not work correctly or persistently guarantee an opt-out\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 51,
    "entity_id": 5,
    "slug": "claudebot/purpose_documented",
    "field": "purpose_documented",
    "value": "training",
    "statement": "ClaudeBot collects web content for training Anthropic generative AI models.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"ClaudeBot helps enhance the utility and safety of our generative AI models by collecting web content that could potentially contribute to their training.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 52,
    "entity_id": 5,
    "slug": "claudebot/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "yes",
    "statement": "Anthropic separates training and search controls into two robots.txt tokens: ClaudeBot for training and Claude-SearchBot for search.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"ClaudeBot ... When a site restricts ClaudeBot access, it signals that the site's future materials should be excluded from our AI model training datasets.\" next to \"Claude-SearchBot navigates the web to improve search result quality for users.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 53,
    "entity_id": 5,
    "slug": "claudebot/anti_circumvention_stance",
    "field": "anti_circumvention_stance",
    "value": "documented",
    "statement": "ClaudeBot respects anti-circumvention technologies and does not bypass CAPTCHAs on crawled sites.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"Anthropic's Bots respect anti-circumvention technologies (e.g., we will not attempt to bypass CAPTCHAs for the sites we crawl.)\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 54,
    "entity_id": 5,
    "slug": "claudebot/user_agent_string_full",
    "field": "user_agent_string_full",
    "value": "not_documented",
    "statement": "Anthropic does not publish a full user agent string for ClaudeBot.",
    "evidence_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
    "evidence_quote": "\"User-agent: ClaudeBot\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-04-07"
  },
  {
    "id": 55,
    "entity_id": 12,
    "slug": "google-extended/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "n/a",
    "statement": "Google-Extended is a robots.txt token that publishers set to control use of crawled content for Gemini training.",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": "\"Google-Extended is a standalone product token that web publishers can use to manage whether content Google crawls from their sites may be used for training future generations of Gemini models\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 56,
    "entity_id": 12,
    "slug": "google-extended/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "n/a",
    "statement": "Google does not publish any Crawl-delay behaviour for the Google-Extended token.",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 57,
    "entity_id": 12,
    "slug": "google-extended/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "no",
    "statement": "Google publishes no IP list for Google-Extended because Google-Extended has no separate HTTP request user agent string.",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": "\"Google-Extended doesn't have a separate HTTP request user agent string. Crawling is done with existing Google user agent strings; the robots.txt user-agent token is used in a control capacity.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 58,
    "entity_id": 12,
    "slug": "google-extended/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "n/a",
    "statement": "Google does not publish a reverse-DNS verification method for Google-Extended.",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": "\"Google-Extended doesn't have a separate HTTP request user agent string.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 59,
    "entity_id": 12,
    "slug": "google-extended/purpose_documented",
    "field": "purpose_documented",
    "value": "policy_only",
    "statement": "Google-Extended is a standalone product token that controls whether Google uses crawled content to train Gemini models.",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": "\"Google-Extended is a standalone product token that web publishers can use to manage whether content Google crawls from their sites may be used for training future generations of Gemini models, that power Gemini Apps and the Vertex AI API.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 60,
    "entity_id": 12,
    "slug": "google-extended/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "yes",
    "statement": "Google-Extended controls Gemini training use only and does not affect inclusion in Google Search.",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": "\"Google-Extended does not impact a site's inclusion in Google Search nor is it used as a ranking signal in Google Search.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 61,
    "entity_id": 12,
    "slug": "google-extended/what_it_controls",
    "field": "what_it_controls",
    "value": "Gemini model training and grounding, not Search inclusion",
    "statement": "Google-Extended controls AI training and grounding in Google systems and does not control Google Search inclusion.",
    "evidence_url": "https://developers.google.com/search/docs/appearance/ai-features",
    "evidence_quote": "\"To limit AI training and grounding in some of Google's other systems, read more about Google-Extended.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2025-12-10"
  },
  {
    "id": 62,
    "entity_id": 12,
    "slug": "google-extended/robots_token_differs_from_ua",
    "field": "robots_token_differs_from_ua",
    "value": "yes",
    "statement": "Google-Extended is a robots.txt control token and does not correspond to a crawler user agent string.",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": "\"Google-Extended doesn't have a separate HTTP request user agent string. Crawling is done with existing Google user agent strings; the robots.txt user-agent token is used in a control capacity.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 63,
    "entity_id": 10,
    "slug": "googlebot/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "yes",
    "statement": "Googlebot obeys robots.txt rules when crawling automatically.",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": "\"They always obey robots.txt rules when crawling automatically.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 64,
    "entity_id": 10,
    "slug": "googlebot/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "no",
    "statement": "Google does not support the robots.txt Crawl-delay directive for Googlebot.",
    "evidence_url": "https://developers.google.com/search/blog/2019/07/a-note-on-unsupported-rules-in-robotstxt",
    "evidence_quote": "\"Why isn't a code handler for other rules like crawl-delay included in the code?\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 65,
    "entity_id": 10,
    "slug": "googlebot/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "yes",
    "statement": "Google publishes the IP address ranges used by Googlebot at https://developers.google.com/static/crawling/ipranges/common-crawlers.json.",
    "evidence_url": "https://developers.google.com/static/crawling/ipranges/common-crawlers.json",
    "evidence_quote": "\"match the crawler's IP address to the lists of Google crawlers' and fetchers' IP ranges\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-11"
  },
  {
    "id": 66,
    "entity_id": 10,
    "slug": "googlebot/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "yes",
    "statement": "Googlebot requests are verifiable by reverse DNS lookup resolving to googlebot.com, google.com, or googleusercontent.com.",
    "evidence_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/verifying-googlebot",
    "evidence_quote": "\"Run a reverse DNS lookup on the accessing IP address from your logs, using the host command\" and \"Verify that the domain name is either googlebot.com, google.com, or googleusercontent.com.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-03-20"
  },
  {
    "id": 67,
    "entity_id": 10,
    "slug": "googlebot/purpose_documented",
    "field": "purpose_documented",
    "value": "search",
    "statement": "Googlebot crawls pages for Google Search, Google Images, Google Video, Google News, and Discover.",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": "\"Crawling preferences for Googlebot affect Google Search (including Discover and all Google Search features), as well as other products such as Google Images, Google Video, Google News, and Discover.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 68,
    "entity_id": 10,
    "slug": "googlebot/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "yes",
    "statement": "Google separates Search crawling under the Googlebot token from Gemini training under the Google-Extended token.",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": "\"Google-Extended is a standalone product token that web publishers can use to manage whether content Google crawls from their sites may be used for training future generations of Gemini models\" and \"Google-Extended does not impact a site's inclusion in Google Search\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 69,
    "entity_id": 10,
    "slug": "googlebot/cites_sources",
    "field": "cites_sources",
    "value": "yes",
    "statement": "Google AI Overviews and AI Mode return AI generated responses with links to supporting websites.",
    "evidence_url": "https://developers.google.com/search/docs/appearance/ai-features",
    "evidence_quote": "\"you can spend more time discovering ... get a comprehensive AI-powered response with links to supporting websites\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2025-12-10"
  },
  {
    "id": 70,
    "entity_id": 10,
    "slug": "googlebot/user_agent_string_full",
    "field": "user_agent_string_full",
    "value": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Googlebot/2.1; +http://www.google.com/bot.html)",
    "statement": "Googlebot sends the user agent string Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Googlebot/2.1; +http://www.google.com/bot.html).",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 71,
    "entity_id": 11,
    "slug": "googleother/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "yes",
    "statement": "GoogleOther obeys robots.txt rules when crawling automatically.",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": "\"They always obey robots.txt rules when crawling automatically.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 72,
    "entity_id": 11,
    "slug": "googleother/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "no",
    "statement": "Google does not support the robots.txt Crawl-delay directive for GoogleOther.",
    "evidence_url": "https://developers.google.com/search/blog/2019/07/a-note-on-unsupported-rules-in-robotstxt",
    "evidence_quote": "\"Why isn't a code handler for other rules like crawl-delay included in the code?\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 73,
    "entity_id": 11,
    "slug": "googleother/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "yes",
    "statement": "Google publishes the IP address ranges used by GoogleOther at https://developers.google.com/static/crawling/ipranges/common-crawlers.json.",
    "evidence_url": "https://developers.google.com/static/crawling/ipranges/common-crawlers.json",
    "evidence_quote": "\"match the crawler's IP address to the lists of Google crawlers' and fetchers' IP ranges\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-09-11"
  },
  {
    "id": 74,
    "entity_id": 11,
    "slug": "googleother/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "yes",
    "statement": "GoogleOther requests are verifiable by reverse DNS lookup resolving to googlebot.com, google.com, or googleusercontent.com.",
    "evidence_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/verifying-googlebot",
    "evidence_quote": "\"Verify that the domain name is either googlebot.com, google.com, or googleusercontent.com.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-03-20"
  },
  {
    "id": 75,
    "entity_id": 11,
    "slug": "googleother/purpose_documented",
    "field": "purpose_documented",
    "value": "mixed",
    "statement": "GoogleOther is a generic crawler that Google product teams use to fetch publicly accessible content from sites.",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": "\"GoogleOther is the generic crawler that may be used by various product teams for fetching publicly accessible content from sites.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 76,
    "entity_id": 11,
    "slug": "googleother/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "not_documented",
    "statement": "Google does not document whether content fetched by GoogleOther is used for Gemini training.",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": "\"GoogleOther is the generic crawler that may be used by various product teams\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 77,
    "entity_id": 11,
    "slug": "googleother/robots_txt_token",
    "field": "robots_txt_token",
    "value": "GoogleOther",
    "statement": "The robots.txt token for GoogleOther is GoogleOther.",
    "evidence_url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers",
    "evidence_quote": "\"The robots.txt token is GoogleOther.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-07-14"
  },
  {
    "id": 78,
    "entity_id": 1,
    "slug": "gptbot/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "yes",
    "statement": "GPTBot honours robots.txt rules that webmasters set for the GPTBot token.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"OpenAI uses OAI-SearchBot and GPTBot robots.txt tags to enable webmasters to manage how their sites and content work with AI.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 79,
    "entity_id": 1,
    "slug": "gptbot/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "not_documented",
    "statement": "OpenAI does not publish whether GPTBot honours the robots.txt Crawl-delay directive.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 80,
    "entity_id": 1,
    "slug": "gptbot/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "yes",
    "statement": "OpenAI publishes the IP address ranges used by GPTBot as a JSON file at https://openai.com/gptbot.json.",
    "evidence_url": "https://openai.com/gptbot.json",
    "evidence_quote": "\"Published IP addresses: https://openai.com/gptbot.json\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2025-10-30"
  },
  {
    "id": 81,
    "entity_id": 1,
    "slug": "gptbot/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "not_documented",
    "statement": "OpenAI does not publish a reverse-DNS verification method for GPTBot.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 82,
    "entity_id": 1,
    "slug": "gptbot/purpose_documented",
    "field": "purpose_documented",
    "value": "training",
    "statement": "GPTBot crawls content that may be used to train OpenAI generative AI foundation models.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"GPTBot is used to make our generative AI foundation models more useful and safe. It is used to crawl content that may be used in training our generative AI foundation models.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 83,
    "entity_id": 1,
    "slug": "gptbot/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "yes",
    "statement": "OpenAI separates search and training controls into two robots.txt tokens: OAI-SearchBot for search and GPTBot for training.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"a webmaster can allow OAI-SearchBot in order to appear in search results while disallowing GPTBot to indicate that crawled content should not be used for training OpenAI's generative AI foundation models.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 84,
    "entity_id": 1,
    "slug": "gptbot/user_agent_string_full",
    "field": "user_agent_string_full",
    "value": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
    "statement": "GPTBot sends the user agent string Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"Example user-agent string (the version number may change): Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 85,
    "entity_id": 1,
    "slug": "gptbot/robots_txt_fetch_marker",
    "field": "robots_txt_fetch_marker",
    "value": "yes",
    "statement": "OpenAI may add a robots.txt marker to the GPTBot user agent string when GPTBot fetches robots.txt files.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"When fetching robots.txt files, we may add a robots.txt marker to the user-agent string to help site owners distinguish those requests from requests for other resources\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 86,
    "entity_id": 4,
    "slug": "oai-adsbot/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "not_documented",
    "statement": "OpenAI does not publish whether OAI-AdsBot honours robots.txt disallow rules.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 87,
    "entity_id": 4,
    "slug": "oai-adsbot/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "not_documented",
    "statement": "OpenAI does not publish whether OAI-AdsBot honours the robots.txt Crawl-delay directive.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 88,
    "entity_id": 4,
    "slug": "oai-adsbot/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "yes",
    "statement": "OpenAI publishes the IP address ranges used by OAI-AdsBot as a JSON file at https://openai.com/adsbot.json.",
    "evidence_url": "https://openai.com/adsbot.json",
    "evidence_quote": "\"Published IP addresses: https://openai.com/adsbot.json\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-05-12"
  },
  {
    "id": 89,
    "entity_id": 4,
    "slug": "oai-adsbot/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "not_documented",
    "statement": "OpenAI does not publish a reverse-DNS verification method for OAI-AdsBot.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 90,
    "entity_id": 4,
    "slug": "oai-adsbot/purpose_documented",
    "field": "purpose_documented",
    "value": "ads",
    "statement": "OAI-AdsBot visits landing pages submitted as ads on ChatGPT to check compliance with OpenAI policies.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"OAI-AdsBot is used to validate the safety of web pages submitted as ads on ChatGPT. When you submit an ad, OpenAI may visit the landing page to ensure it complies with our policies.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 91,
    "entity_id": 4,
    "slug": "oai-adsbot/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "not_documented",
    "statement": "OpenAI does not document OAI-AdsBot as part of a search and training token split.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"OAI-AdsBot only visits pages submitted as ads, and the data collected by OAI-AdsBot is not used to train generative AI foundation models.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 92,
    "entity_id": 4,
    "slug": "oai-adsbot/user_agent_string_full",
    "field": "user_agent_string_full",
    "value": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; OAI-AdsBot/1.0; +https://openai.com/adsbot",
    "statement": "OAI-AdsBot sends the user agent string Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; OAI-AdsBot/1.0; +https://openai.com/adsbot.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"Full user-agent string: Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; OAI-AdsBot/1.0; +https://openai.com/adsbot\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 93,
    "entity_id": 4,
    "slug": "oai-adsbot/scope_of_crawling",
    "field": "scope_of_crawling",
    "value": "limited to submitted ad landing pages",
    "statement": "OAI-AdsBot visits only pages submitted as ads on ChatGPT.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"OAI-AdsBot only visits pages submitted as ads, and the data collected by OAI-AdsBot is not used to train generative AI foundation models.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 94,
    "entity_id": 2,
    "slug": "oai-searchbot/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "yes",
    "statement": "OAI-SearchBot honours robots.txt opt-outs set for the OAI-SearchBot token.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"Sites that are opted out of OAI-SearchBot will not be shown in ChatGPT search answers, though can still appear as navigational links.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 95,
    "entity_id": 2,
    "slug": "oai-searchbot/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "not_documented",
    "statement": "OpenAI does not publish whether OAI-SearchBot honours the robots.txt Crawl-delay directive.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 96,
    "entity_id": 2,
    "slug": "oai-searchbot/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "yes",
    "statement": "OpenAI publishes the IP address ranges used by OAI-SearchBot as a JSON file at https://openai.com/searchbot.json.",
    "evidence_url": "https://openai.com/searchbot.json",
    "evidence_quote": "\"Published IP addresses: https://openai.com/searchbot.json\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-02"
  },
  {
    "id": 97,
    "entity_id": 2,
    "slug": "oai-searchbot/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "not_documented",
    "statement": "OpenAI does not publish a reverse-DNS verification method for OAI-SearchBot.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 98,
    "entity_id": 2,
    "slug": "oai-searchbot/purpose_documented",
    "field": "purpose_documented",
    "value": "search",
    "statement": "OAI-SearchBot surfaces websites in search results in ChatGPT search features.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"OAI-SearchBot is used to surface websites in search results in ChatGPT's search features.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 99,
    "entity_id": 2,
    "slug": "oai-searchbot/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "yes",
    "statement": "OpenAI separates search and training controls into two robots.txt tokens: OAI-SearchBot for search and GPTBot for training.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"a webmaster can allow OAI-SearchBot in order to appear in search results while disallowing GPTBot to indicate that crawled content should not be used for training OpenAI's generative AI foundation models.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 100,
    "entity_id": 2,
    "slug": "oai-searchbot/cites_sources",
    "field": "cites_sources",
    "value": "yes",
    "statement": "OAI-SearchBot surfaces crawled websites as links in ChatGPT search answers.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"OAI-SearchBot is used to surface websites in search results in ChatGPT's search features. Sites that are opted out of OAI-SearchBot will not be shown in ChatGPT search answers, though can still appear as navigational links.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 101,
    "entity_id": 2,
    "slug": "oai-searchbot/reads_llms_txt",
    "field": "reads_llms_txt",
    "value": "not_documented",
    "statement": "OpenAI does not publish whether OAI-SearchBot reads llms.txt files.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 102,
    "entity_id": 2,
    "slug": "oai-searchbot/user_agent_string_full",
    "field": "user_agent_string_full",
    "value": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
    "statement": "OAI-SearchBot sends the user agent string Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot.",
    "evidence_url": "https://developers.openai.com/api/docs/bots",
    "evidence_quote": "\"Example user-agent string (the version number may change): Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": null
  },
  {
    "id": 103,
    "entity_id": 9,
    "slug": "perplexity-user/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "no",
    "statement": "Perplexity-User ignores robots.txt rules because a user requests each Perplexity-User fetch.",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": "\"Since a user requested the fetch, this fetcher generally ignores robots.txt rules.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 104,
    "entity_id": 9,
    "slug": "perplexity-user/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "not_documented",
    "statement": "Perplexity does not publish whether Perplexity-User honours the robots.txt Crawl-delay directive.",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 105,
    "entity_id": 9,
    "slug": "perplexity-user/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "yes",
    "statement": "Perplexity publishes the IP address ranges used by Perplexity-User as a JSON file at https://www.perplexity.ai/perplexity-user.json.",
    "evidence_url": "https://www.perplexity.ai/perplexity-user.json",
    "evidence_quote": "\"Published IP addresses: https://www.perplexity.com/perplexity-user.json\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2025-10-17"
  },
  {
    "id": 106,
    "entity_id": 9,
    "slug": "perplexity-user/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "not_documented",
    "statement": "Perplexity does not publish a reverse-DNS verification method for Perplexity-User.",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 107,
    "entity_id": 9,
    "slug": "perplexity-user/purpose_documented",
    "field": "purpose_documented",
    "value": "user_fetch",
    "statement": "Perplexity-User visits web pages to answer questions that users ask Perplexity.",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": "\"Perplexity-User supports user actions within Perplexity. When users ask Perplexity a question, it might visit a web page to help provide an accurate answer and include a link to the page in its response.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 108,
    "entity_id": 9,
    "slug": "perplexity-user/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "not_documented",
    "statement": "Perplexity does not document Perplexity-User as a search or training opt-out token.",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": "\"It is not used for web crawling or to collect content for training AI foundation models.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 109,
    "entity_id": 9,
    "slug": "perplexity-user/cites_sources",
    "field": "cites_sources",
    "value": "yes",
    "statement": "Perplexity includes a link to each page that Perplexity-User visits in the Perplexity response.",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": "\"When users ask Perplexity a question, it might visit a web page to help provide an accurate answer and include a link to the page in its response.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 110,
    "entity_id": 9,
    "slug": "perplexity-user/reads_llms_txt",
    "field": "reads_llms_txt",
    "value": "not_documented",
    "statement": "Perplexity does not publish whether Perplexity-User reads llms.txt files.",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 111,
    "entity_id": 9,
    "slug": "perplexity-user/user_agent_string_full",
    "field": "user_agent_string_full",
    "value": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Perplexity-User/1.0; +https://perplexity.ai/perplexity-user)",
    "statement": "Perplexity-User sends the user agent string Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Perplexity-User/1.0; +https://perplexity.ai/perplexity-user).",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": "\"Full user Agent: Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Perplexity-User/1.0; +https://perplexity.ai/perplexity-user)\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 112,
    "entity_id": 8,
    "slug": "perplexitybot/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "yes",
    "statement": "PerplexityBot honours robots.txt rules that webmasters set for the PerplexityBot token.",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": "\"Webmasters can use the following robots.txt tags to manage how their sites and content interact with Perplexity. Each setting works independently, and it may take up to 24 hours for our systems to reflect changes.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "medium",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 113,
    "entity_id": 8,
    "slug": "perplexitybot/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "not_documented",
    "statement": "Perplexity does not publish whether PerplexityBot honours the robots.txt Crawl-delay directive.",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 114,
    "entity_id": 8,
    "slug": "perplexitybot/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "yes",
    "statement": "Perplexity publishes the IP address ranges used by PerplexityBot as a JSON file at https://www.perplexity.ai/perplexitybot.json.",
    "evidence_url": "https://www.perplexity.ai/perplexitybot.json",
    "evidence_quote": "\"Published IP addresses: https://www.perplexity.com/perplexitybot.json\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2025-02-07"
  },
  {
    "id": 115,
    "entity_id": 8,
    "slug": "perplexitybot/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "not_documented",
    "statement": "Perplexity does not publish a reverse-DNS verification method for PerplexityBot.",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 116,
    "entity_id": 8,
    "slug": "perplexitybot/purpose_documented",
    "field": "purpose_documented",
    "value": "search",
    "statement": "PerplexityBot surfaces and links websites in Perplexity search results.",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": "\"PerplexityBot is designed to surface and link websites in search results on Perplexity. It is not used to crawl content for AI foundation models.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 117,
    "entity_id": 8,
    "slug": "perplexitybot/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "not_documented",
    "statement": "Perplexity does not document a separate training crawler token alongside PerplexityBot.",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": "\"It is not used to crawl content for AI foundation models.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 118,
    "entity_id": 8,
    "slug": "perplexitybot/cites_sources",
    "field": "cites_sources",
    "value": "yes",
    "statement": "Perplexity links to the websites that PerplexityBot surfaces in Perplexity search results.",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": "\"PerplexityBot is designed to surface and link websites in search results on Perplexity.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 119,
    "entity_id": 8,
    "slug": "perplexitybot/reads_llms_txt",
    "field": "reads_llms_txt",
    "value": "not_documented",
    "statement": "Perplexity does not publish whether PerplexityBot reads llms.txt files.",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 120,
    "entity_id": 8,
    "slug": "perplexitybot/user_agent_string_full",
    "field": "user_agent_string_full",
    "value": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
    "statement": "PerplexityBot sends the user agent string Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot).",
    "evidence_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
    "evidence_quote": "\"Full user Agent: Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-13",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:15:09Z",
    "updated_at": "2026-09-14T00:15:09Z",
    "evidence_date": "2026-01-29"
  },
  {
    "id": 121,
    "entity_id": 13,
    "slug": "bingbot/respects_robots_txt",
    "field": "respects_robots_txt",
    "value": "yes",
    "statement": "Bingbot honors robots.txt directives and other Microsoft-supported control mechanisms that content owners set for crawling.",
    "evidence_url": "https://www.bing.com/webmasters/help/ai-performance-9f8e7d6c",
    "evidence_quote": "\"Bing respects all content owner preferences expressed through robots.txt and other supported control mechanisms.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-14",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:32:17Z",
    "updated_at": "2026-09-14T00:32:17Z",
    "evidence_date": null
  },
  {
    "id": 122,
    "entity_id": 13,
    "slug": "bingbot/honours_crawl_delay",
    "field": "honours_crawl_delay",
    "value": "yes",
    "statement": "Microsoft states that a crawl-delay directive in robots.txt always takes precedence over Bing Webmaster Tools crawl control settings for Bingbot.",
    "evidence_url": "https://www.bing.com/webmasters/help/crawl-control-55a30303",
    "evidence_quote": "\"If we find a crawl-delay directive in your robots.txt file, then it will always take precedence over the information from this feature.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-14",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:32:17Z",
    "updated_at": "2026-09-14T00:32:17Z",
    "evidence_date": null
  },
  {
    "id": 123,
    "entity_id": 13,
    "slug": "bingbot/publishes_ip_list",
    "field": "publishes_ip_list",
    "value": "yes",
    "statement": "Microsoft publishes Bingbot's IP address ranges as a JSON file at bing.com/toolbox/bingbot.json for verification purposes.",
    "evidence_url": "https://www.bing.com/toolbox/bingbot.json",
    "evidence_quote": "(none, JSON data file)",
    "method": "vendor_doc",
    "verified_at": "2026-09-14",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:32:17Z",
    "updated_at": "2026-09-14T00:32:17Z",
    "evidence_date": "2024-01-03"
  },
  {
    "id": 124,
    "entity_id": 13,
    "slug": "bingbot/verifiable_by_rdns",
    "field": "verifiable_by_rdns",
    "value": "yes",
    "statement": "Microsoft verifies Bingbot traffic through reverse DNS lookups that resolve to hostnames ending in search.msn.com.",
    "evidence_url": "https://www.bing.com/webmasters/help/how-to-verify-bingbot-3905dc26",
    "evidence_quote": "\"Perform a reverse DNS lookup using the IP address from the logs to verify that it resolves to a name that end with search.msn.com\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-14",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:32:17Z",
    "updated_at": "2026-09-14T00:32:17Z",
    "evidence_date": null
  },
  {
    "id": 125,
    "entity_id": 13,
    "slug": "bingbot/purpose_documented",
    "field": "purpose_documented",
    "value": "yes",
    "statement": "Bingbot serves as Microsoft's standard crawler and handles most of Bing's daily web crawling.",
    "evidence_url": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
    "evidence_quote": "\"Bingbot is our standard crawler and handles most of our crawling needs each day.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-14",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:32:17Z",
    "updated_at": "2026-09-14T00:32:17Z",
    "evidence_date": null
  },
  {
    "id": 126,
    "entity_id": 13,
    "slug": "bingbot/separate_search_and_training_tokens",
    "field": "separate_search_and_training_tokens",
    "value": "no",
    "statement": "Microsoft controls Bingbot's use in AI training through meta tags rather than publishing a separate training-specific crawler token.",
    "evidence_url": "https://www.bing.com/webmasters/help/robots-meta-tags-and-attributes-that-bing-supports-5198d240",
    "evidence_quote": "\"Do not use the content for training Microsoft's generative AI foundation models.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-14",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:32:17Z",
    "updated_at": "2026-09-14T00:32:17Z",
    "evidence_date": null
  },
  {
    "id": 127,
    "entity_id": 13,
    "slug": "bingbot/cites_sources",
    "field": "cites_sources",
    "value": "yes",
    "statement": "Copilot Search in Bing delivers clearly cited sources alongside generative answers to support publishers and content owners.",
    "evidence_url": "https://blogs.bing.com/search/April-2025/Introducing-Copilot-Search-in-Bing",
    "evidence_quote": "\"We've set out to deliver helpful, clearly cited sources, plus rich, relevant data, images and videos from your favorite publishers and content owners to inspire you to explore further.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-14",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:32:17Z",
    "updated_at": "2026-09-14T00:32:17Z",
    "evidence_date": "2025-04-04"
  },
  {
    "id": 128,
    "entity_id": 13,
    "slug": "bingbot/reads_llms_txt",
    "field": "reads_llms_txt",
    "value": "not_documented",
    "statement": "Microsoft does not publish whether Bingbot reads llms.txt files.",
    "evidence_url": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
    "evidence_quote": null,
    "method": "vendor_doc",
    "verified_at": "2026-09-14",
    "confidence": null,
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:32:17Z",
    "updated_at": "2026-09-14T00:32:17Z",
    "evidence_date": null
  },
  {
    "id": 129,
    "entity_id": 13,
    "slug": "bingbot/user_agent_string_full",
    "field": "user_agent_string_full",
    "value": "Mozilla/5.0 (compatible; bingbot/2.0; +http://www.bing.com/bingbot.htm)",
    "statement": "Bingbot sends a user agent string containing bingbot/2.0 and a link to Microsoft's bingbot.htm page.",
    "evidence_url": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
    "evidence_quote": "\"Mozilla/5.0 (compatible; bingbot/2.0; +http://www.bing.com/bingbot.htm) W.X.Y.Z Safari/537.36\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-14",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:32:17Z",
    "updated_at": "2026-09-14T00:32:17Z",
    "evidence_date": null
  },
  {
    "id": 130,
    "entity_id": 13,
    "slug": "bingbot/opt_out_mechanism",
    "field": "opt_out_mechanism",
    "value": "yes",
    "statement": "Microsoft lets publishers block Bingbot content from Copilot answers and AI training using noarchive and nocache tags.",
    "evidence_url": "https://www.bing.com/webmasters/help/robots-meta-tags-and-attributes-that-bing-supports-5198d240",
    "evidence_quote": "\"Do not link in Chat and Copilot. Do not use the content for training Microsoft's generative AI foundation models.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-14",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:32:17Z",
    "updated_at": "2026-09-14T00:32:17Z",
    "evidence_date": null
  },
  {
    "id": 131,
    "entity_id": 13,
    "slug": "bingbot/follows_noarchive",
    "field": "follows_noarchive",
    "value": "yes",
    "statement": "Bingbot excludes noarchive-tagged content from Copilot and Chat answers and from generative AI training data.",
    "evidence_url": "https://www.bing.com/webmasters/help/robots-meta-tags-and-attributes-that-bing-supports-5198d240",
    "evidence_quote": "\"Do not link in Chat and Copilot. Do not use the content for training Microsoft's generative AI foundation models.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-14",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:32:17Z",
    "updated_at": "2026-09-14T00:32:17Z",
    "evidence_date": null
  },
  {
    "id": 132,
    "entity_id": 10,
    "slug": "googlebot/reads_llms_txt",
    "field": "reads_llms_txt",
    "value": "no",
    "statement": "Google Search does not use llms.txt files.",
    "evidence_url": "https://developers.google.com/search/docs/fundamentals/ai-optimization-guide",
    "evidence_quote": "\"LLMS.txt files and other \"special\" markup: You don't need to create new machine readable files, AI text files, markup, or Markdown to appear in Google Search (including its generative AI capabilities), as Google Search itself doesn't use them.\"",
    "method": "vendor_doc",
    "verified_at": "2026-09-14",
    "confidence": "high",
    "status": "current",
    "supersedes_id": null,
    "created_at": "2026-09-14T00:32:29Z",
    "updated_at": "2026-09-14T00:32:29Z",
    "evidence_date": "2026-07-10"
  }
]