{
  "entities": [
    {
      "id": 6,
      "slug": "claude-searchbot",
      "vendor": "Anthropic",
      "name": "Claude-SearchBot",
      "kind": "search_bot",
      "purpose": "search",
      "ua_token": "Claude-SearchBot",
      "ua_pattern": "Claude-SearchBot",
      "robots_token": "Claude-SearchBot",
      "ip_list_url": "https://claude.com/crawling/bots.json",
      "docs_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Navigates the web to improve Claude search result quality.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 7,
      "slug": "claude-user",
      "vendor": "Anthropic",
      "name": "Claude-User",
      "kind": "fetcher",
      "purpose": "user_fetch",
      "ua_token": "Claude-User",
      "ua_pattern": "Claude-User",
      "robots_token": "Claude-User",
      "ip_list_url": "https://claude.com/crawling/bots.json",
      "docs_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Retrieves a page when a Claude user directs Claude to access that URL.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 5,
      "slug": "claudebot",
      "vendor": "Anthropic",
      "name": "ClaudeBot",
      "kind": "crawler",
      "purpose": "training",
      "ua_token": "ClaudeBot",
      "ua_pattern": "ClaudeBot",
      "robots_token": "ClaudeBot",
      "ip_list_url": "https://claude.com/crawling/bots.json",
      "docs_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-content-from-the-web-and-how-can-site-owners-block-the-crawler",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Collects web content that may contribute to Anthropic model training.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 14,
      "slug": "applebot",
      "vendor": "Apple",
      "name": "Applebot",
      "kind": "crawler",
      "purpose": "search",
      "ua_token": "Applebot",
      "ua_pattern": "Applebot(?!-Extended)",
      "robots_token": "Applebot",
      "ip_list_url": "https://search.developer.apple.com/applebot.json",
      "docs_url": "https://support.apple.com/en-us/119829",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Crawls the web to power Siri, Spotlight and Safari search features.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 15,
      "slug": "applebot-extended",
      "vendor": "Apple",
      "name": "Applebot-Extended",
      "kind": "policy_token",
      "purpose": "training",
      "ua_token": null,
      "ua_pattern": null,
      "robots_token": "Applebot-Extended",
      "ip_list_url": null,
      "docs_url": "https://support.apple.com/en-us/119829",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Robots.txt token that controls whether content already crawled by Applebot may be used to train Apple foundation models. Vendor documentation states Applebot-Extended does not crawl web pages: it is a policy token, not an observable crawler.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 12,
      "slug": "google-extended",
      "vendor": "Google",
      "name": "Google-Extended",
      "kind": "policy_token",
      "purpose": "training",
      "ua_token": null,
      "ua_pattern": null,
      "robots_token": "Google-Extended",
      "ip_list_url": null,
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Robots.txt token that controls whether content already crawled by Googlebot may be used to train Gemini models and ground Gemini features. Vendor documentation states Google-Extended is not a separate crawler and uses no distinct user agent string: it cannot be observed directly in request logs, only declared for in robots.txt.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 11,
      "slug": "googleother",
      "vendor": "Google",
      "name": "GoogleOther",
      "kind": "crawler",
      "purpose": "mixed",
      "ua_token": "GoogleOther",
      "ua_pattern": "GoogleOther",
      "robots_token": "GoogleOther",
      "ip_list_url": "https://developers.google.com/static/crawling/ipranges/common-crawlers.json",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Generic crawler used by various Google product teams to fetch publicly accessible content outside Search.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 10,
      "slug": "googlebot",
      "vendor": "Google",
      "name": "Googlebot",
      "kind": "crawler",
      "purpose": "search",
      "ua_token": "Googlebot",
      "ua_pattern": "Googlebot",
      "robots_token": "Googlebot",
      "ip_list_url": "https://developers.google.com/static/crawling/ipranges/common-crawlers.json",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Common crawler for Google Search, Discover, Images, Video and News.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 13,
      "slug": "bingbot",
      "vendor": "Microsoft",
      "name": "Bingbot",
      "kind": "crawler",
      "purpose": "search",
      "ua_token": "bingbot",
      "ua_pattern": "bingbot",
      "robots_token": "bingbot",
      "ip_list_url": "https://www.bing.com/toolbox/bingbot.json",
      "docs_url": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Crawls the web for Bing Search and, per Microsoft, feeds Copilot and Bing Chat surfaces. Microsoft documents a separate opt-out from generative-AI training use via the noarchive robots meta tag or X-Robots-Tag header rather than a distinct Copilot user agent; no Copilot-specific crawler token is named in Microsoft's public documentation as of 2026-09-13.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 3,
      "slug": "chatgpt-user",
      "vendor": "OpenAI",
      "name": "ChatGPT-User",
      "kind": "fetcher",
      "purpose": "user_fetch",
      "ua_token": "ChatGPT-User",
      "ua_pattern": "ChatGPT-User",
      "robots_token": "ChatGPT-User",
      "ip_list_url": "https://openai.com/chatgpt-user.json",
      "docs_url": "https://platform.openai.com/docs/bots",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Fetches a page only when a ChatGPT or Custom GPT user directs it to that URL. Vendor documentation states robots.txt rules may not apply to this fetcher.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 1,
      "slug": "gptbot",
      "vendor": "OpenAI",
      "name": "GPTBot",
      "kind": "crawler",
      "purpose": "training",
      "ua_token": "GPTBot",
      "ua_pattern": "GPTBot",
      "robots_token": "GPTBot",
      "ip_list_url": "https://openai.com/gptbot.json",
      "docs_url": "https://platform.openai.com/docs/bots",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Crawls public web content for OpenAI model training.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 4,
      "slug": "oai-adsbot",
      "vendor": "OpenAI",
      "name": "OAI-AdsBot",
      "kind": "ads_bot",
      "purpose": "ads",
      "ua_token": "OAI-AdsBot",
      "ua_pattern": "OAI-AdsBot",
      "robots_token": "OAI-AdsBot",
      "ip_list_url": "https://openai.com/adsbot.json",
      "docs_url": "https://platform.openai.com/docs/bots",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Checks ad landing page quality and relevance for ChatGPT ads. Vendor documentation states it only visits pages submitted for ads review.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 2,
      "slug": "oai-searchbot",
      "vendor": "OpenAI",
      "name": "OAI-SearchBot",
      "kind": "search_bot",
      "purpose": "search",
      "ua_token": "OAI-SearchBot",
      "ua_pattern": "OAI-SearchBot",
      "robots_token": "OAI-SearchBot",
      "ip_list_url": "https://openai.com/searchbot.json",
      "docs_url": "https://platform.openai.com/docs/bots",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Surfaces web pages in ChatGPT search results.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 9,
      "slug": "perplexity-user",
      "vendor": "Perplexity",
      "name": "Perplexity-User",
      "kind": "fetcher",
      "purpose": "user_fetch",
      "ua_token": "Perplexity-User",
      "ua_pattern": "Perplexity-User",
      "robots_token": "Perplexity-User",
      "ip_list_url": "https://www.perplexity.ai/perplexity-user.json",
      "docs_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Fetches a page when a Perplexity user asks a question that requires that page. Vendor documentation states this fetcher generally ignores robots.txt rules.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    },
    {
      "id": 8,
      "slug": "perplexitybot",
      "vendor": "Perplexity",
      "name": "PerplexityBot",
      "kind": "search_bot",
      "purpose": "search",
      "ua_token": "PerplexityBot",
      "ua_pattern": "PerplexityBot",
      "robots_token": "PerplexityBot",
      "ip_list_url": "https://www.perplexity.ai/perplexitybot.json",
      "docs_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
      "first_documented": "2026-09-13",
      "first_seen_here": null,
      "last_seen_here": null,
      "status": "active",
      "notes": "Surfaces and links websites in Perplexity search results.",
      "created_at": "2026-09-13T00:00:00Z",
      "updated_at": "2026-09-13T00:00:00Z"
    }
  ]
}