{
  "version": "1.0.0",
  "updated": "2026-09-01",
  "license": "CC BY 4.0",
  "crawlers": [
    {
      "token": "AhrefsBot",
      "operator": "Ahrefs",
      "purpose": "seo-tool",
      "docs_url": "https://ahrefs.com/robot",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Ahrefs' backlink-index crawler; the page states \"Obeys robots.txt: Yes\" and it honours Crawl-Delay, so blocking it only removes you from Ahrefs' link data and has no effect on Google or AI answers."
    },
    {
      "token": "AhrefsSiteAudit",
      "operator": "Ahrefs",
      "purpose": "seo-tool",
      "docs_url": "https://ahrefs.com/robot",
      "respects_robots": "partial",
      "confidence": "official",
      "notes": "Separate token from AhrefsBot, used only when someone runs a Site Audit; docs say \"Yes by default (website owners can request to disobey robots.txt on their sites)\" — verified owners can opt to have it crawl disallowed sections."
    },
    {
      "token": "AI2Bot",
      "operator": "Allen Institute for AI (Ai2)",
      "purpose": "ai-training",
      "docs_url": "https://allenai.org/crawler",
      "respects_robots": "undocumented",
      "confidence": "official",
      "notes": "Feeds open training datasets such as Dolma. Ai2's notice publishes the AI2Bot token but only says the UA string can be used to filter traffic - it makes no robots.txt commitment, so treat obedience as unverified."
    },
    {
      "token": "Amazonbot",
      "operator": "Amazon",
      "purpose": "ai-training",
      "docs_url": "https://developer.amazon.com/amazonbot",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Improves Amazon products and may train Amazon AI models. Caches robots.txt up to 30 days and does not support Crawl-delay."
    },
    {
      "token": "Amzn-SearchBot",
      "operator": "Amazon",
      "purpose": "ai-answers",
      "docs_url": "https://developer.amazon.com/amazonbot",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Powers Amazon search experiences such as Alexa; explicitly not used for generative AI training. If unnamed in robots.txt it follows the rules given to other search bots."
    },
    {
      "token": "Amzn-User",
      "operator": "Amazon",
      "purpose": "ai-agent",
      "docs_url": "https://developer.amazon.com/amazonbot",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Live fetch for user queries such as Alexa needing current info; documented as not used for generative AI training."
    },
    {
      "token": "Claude-SearchBot",
      "operator": "Anthropic",
      "purpose": "ai-answers",
      "docs_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Indexes for Claude's search results; Anthropic warns blocking it may reduce how often your site is surfaced in Claude answers."
    },
    {
      "token": "Claude-User",
      "operator": "Anthropic",
      "purpose": "ai-agent",
      "docs_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Fetches a page because a person asked Claude about it; unlike OpenAI's and Perplexity's user-initiated agents, Anthropic documents this one as controllable via robots.txt."
    },
    {
      "token": "ClaudeBot",
      "operator": "Anthropic",
      "purpose": "ai-training",
      "docs_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Training corpus only; Anthropic also honours the non-standard Crawl-delay directive."
    },
    {
      "token": "Applebot",
      "operator": "Apple",
      "purpose": "search",
      "docs_url": "https://support.apple.com/en-us/119829",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Apple documents Applebot for Spotlight, Siri and Safari search only and never mentions iMessage previews, so treat the common \"it doubles as the Messages preview fetcher\" claim as unverified; note the split — disallow Applebot-Extended to opt out of foundation-model training and use the nosnippet tag to opt out of AI answers, both of which leave you in Apple search."
    },
    {
      "token": "Applebot-Extended",
      "operator": "Apple",
      "purpose": "ai-training",
      "docs_url": "https://support.apple.com/en-us/119829",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "A control token that never crawls: it tells Apple not to use Applebot-collected data to train foundation models. Apple states pages disallowing it still appear in search results."
    },
    {
      "token": "barkrowler",
      "operator": "Babbar.tech",
      "purpose": "seo-tool",
      "docs_url": "https://www.babbar.tech/crawler",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Babbar documents the token lowercase and states \"We respect robots.txt file (using crawler-commons tool set) and disallow directives\", including Crawl-Delay; the HTTP header reads Barkrowler/x.y."
    },
    {
      "token": "Baiduspider",
      "operator": "Baidu",
      "purpose": "search",
      "docs_url": "https://help.baidu.com/question?prod_id=99&class=0&id=3001",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Blocking Baiduspider covers PC, mobile and general search, but Baidu documents that Baiduspider-cpro and Baiduspider-ads ignore robots.txt and Baiduspider-video does not support the rules."
    },
    {
      "token": "Bytespider",
      "operator": "ByteDance",
      "purpose": "ai-training",
      "docs_url": "https://darkvisitors.com/agents/bytespider",
      "respects_robots": "undocumented",
      "confidence": "unofficial",
      "notes": "Token comes from the crawler's own UA string in server logs and third-party trackers - ByteDance publishes no reachable documentation, IP list or robots.txt statement (its UA reference link zhanzhang.toutiao.com is unavailable outside China). Independent log analyses report it fetching robots.txt then crawling disallowed URLs, so block at the WAF."
    },
    {
      "token": "CCBot",
      "operator": "Common Crawl",
      "purpose": "ai-training",
      "docs_url": "https://commoncrawl.org/ccbot",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Nonprofit open archive that is a major upstream corpus for many LLMs, so blocking it is an indirect training opt-out across several vendors. Common Crawl warns of bots falsely claiming to be CCBot - verify by IP."
    },
    {
      "token": "DataForSeoBot",
      "operator": "DataForSEO",
      "purpose": "seo-tool",
      "docs_url": "https://dataforseo.com/dataforseo-bot",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Page states \"Obeys robots.txt: YES\" and documents both Disallow and Crawl-delay examples using this exact token (note the lowercase \"eo\" in SeoBot)."
    },
    {
      "token": "Diffbot",
      "operator": "Diffbot",
      "purpose": "other",
      "docs_url": "https://www.diffbot.com/docs/crawl/faq/robots-txt",
      "respects_robots": "partial",
      "confidence": "official",
      "notes": "Structured-data extraction feeding Diffbot's Knowledge Graph, which is resold to AI companies. Adheres to robots.txt by default including Crawl-delay, but Diffbot says instructions can be overridden under partner agreements."
    },
    {
      "token": "Diffbot-User",
      "operator": "Diffbot",
      "purpose": "ai-agent",
      "docs_url": "https://www.diffbot.com/docs/crawl/faq/robots-txt",
      "respects_robots": "partial",
      "confidence": "official",
      "notes": "Requests initiated by individual users of Diffbot software; single-URL extractions requested by a customer can proceed even where a block exists."
    },
    {
      "token": "Discordbot",
      "operator": "Discord",
      "purpose": "social-preview",
      "docs_url": "https://support.discord.com/hc/en-us/articles/42500550752919-About-Discord-Link-Previews-and-the-Discordbot",
      "respects_robots": "undocumented",
      "confidence": "official",
      "notes": "Discord's own article names Discordbot and the UA \"Mozilla/5.0 (compatible; Discordbot/2.0; +https://discordapp.com)\" and confirms it fetches only when a link is shared with \"no continuous scanning\", but says nothing about robots.txt — it is widely observed to request robots.txt, so assume a Disallow will break your Discord embeds; the page 403s to automated fetchers behind Cloudflare but loads in a browser."
    },
    {
      "token": "DuckAssistBot",
      "operator": "DuckDuckGo",
      "purpose": "ai-answers",
      "docs_url": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Real-time fetch for DuckDuckGo's cited AI answers; DuckDuckGo states the data never trains models and opting out does not affect organic rankings. Takes ~72h to take effect."
    },
    {
      "token": "DuckDuckBot",
      "operator": "DuckDuckGo",
      "purpose": "search",
      "docs_url": "https://duckduckgo.com/duckduckgo-help-pages/results/duckduckbot/",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "DDG's main results come from Bing, so DuckDuckBot is a supplementary crawler and blocking it does not remove you from DuckDuckGo results."
    },
    {
      "token": "AdsBot-Google",
      "operator": "Google",
      "purpose": "ads",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-special-case-crawlers",
      "respects_robots": "partial",
      "confidence": "official",
      "notes": "Ignores 'User-agent: *' and must be named explicitly; blocking it degrades Google Ads landing-page quality checks, which can hurt Quality Score if you run ads."
    },
    {
      "token": "AdsBot-Google-Mobile",
      "operator": "Google",
      "purpose": "ads",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-special-case-crawlers",
      "respects_robots": "partial",
      "confidence": "official",
      "notes": "Mobile ad-quality checker that ignores 'User-agent: *' and must be named explicitly; a retired iPhone variant used the same token."
    },
    {
      "token": "APIs-Google",
      "operator": "Google",
      "purpose": "other",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-special-case-crawlers",
      "respects_robots": "partial",
      "confidence": "official",
      "notes": "Ignores the wildcard 'User-agent: *' group and only obeys rules addressed to APIs-Google by name; it delivers push notification messages for Google APIs."
    },
    {
      "token": "FeedFetcher-Google",
      "operator": "Google",
      "purpose": "other",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-user-triggered-fetchers",
      "respects_robots": "no",
      "confidence": "unofficial",
      "notes": "Pulls RSS/Atom feeds for Google News and WebSub; user-triggered, so it ignores robots.txt and Google publishes no robots.txt token for it."
    },
    {
      "token": "Google-Agent",
      "operator": "Google",
      "purpose": "ai-agent",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-user-triggered-fetchers",
      "respects_robots": "no",
      "confidence": "unofficial",
      "notes": "Added March 2026 for Google-hosted agents acting on user request; Google lists it only as an HTTP user-agent with no robots.txt token and says user-triggered fetchers ignore robots.txt, and it is also experimenting with Web Bot Auth identity agent.bot.goog — note the older 'GoogleAgent-Mariner' name appears in no current Google documentation."
    },
    {
      "token": "Google-CloudVertexBot",
      "operator": "Google",
      "purpose": "ai-agent",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Only crawls sites when a site owner explicitly requests it while building Vertex AI Agents; it has no effect on Google Search, and it also matches rules written for the Googlebot token."
    },
    {
      "token": "Google-CWS",
      "operator": "Google",
      "purpose": "other",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-user-triggered-fetchers",
      "respects_robots": "no",
      "confidence": "unofficial",
      "notes": "Chrome Web Store fetcher for URLs listed in extension and theme metadata; ignores robots.txt and has no published robots.txt token."
    },
    {
      "token": "Google-Extended",
      "operator": "Google",
      "purpose": "ai-training",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Widely misunderstood: it is a control-only token with no crawler behind it, and it governs Gemini model training plus grounding in Gemini Apps and Vertex AI — it does NOT remove you from Google Search or from AI Overviews/AI Mode, which are controlled by Googlebot and nosnippet."
    },
    {
      "token": "Google-GeminiNotebook",
      "operator": "Google",
      "purpose": "ai-answers",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-user-triggered-fetchers",
      "respects_robots": "no",
      "confidence": "unofficial",
      "notes": "Renamed from Google-NotebookLM in July 2026 (old string supported until August 2026); fetches URLs a user pasted into Gemini Notebook, and as a user-triggered fetcher it ignores robots.txt and has no published robots.txt token."
    },
    {
      "token": "Google-InspectionTool",
      "operator": "Google",
      "purpose": "seo-tool",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Powers Search Console URL Inspection and the Rich Results Test; blocking it breaks your own debugging tools but has no effect on Search rankings or indexing."
    },
    {
      "token": "Google-Pinpoint",
      "operator": "Google",
      "purpose": "other",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-user-triggered-fetchers",
      "respects_robots": "no",
      "confidence": "unofficial",
      "notes": "Fetches URLs a Pinpoint user added to a personal research collection; ignores robots.txt and has no published robots.txt token."
    },
    {
      "token": "Google-Read-Aloud",
      "operator": "Google",
      "purpose": "other",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-user-triggered-fetchers",
      "respects_robots": "no",
      "confidence": "unofficial",
      "notes": "Fetches pages for text-to-speech on user request and ignores robots.txt; token taken from its HTTP user-agent since Google publishes no robots.txt token, and the deprecated form was 'google-speakr'."
    },
    {
      "token": "Google-Safety",
      "operator": "Google",
      "purpose": "other",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-special-case-crawlers",
      "respects_robots": "no",
      "confidence": "unofficial",
      "notes": "Google publishes no robots.txt token for this agent and documents that it ignores robots.txt entirely — this string is its HTTP user-agent, so a rule naming it will have no effect; it does abuse and malware scanning."
    },
    {
      "token": "Google-Site-Verification",
      "operator": "Google",
      "purpose": "seo-tool",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-user-triggered-fetchers",
      "respects_robots": "no",
      "confidence": "unofficial",
      "notes": "Fetches your Search Console verification token on demand; it ignores robots.txt by design, which is why site verification still works on heavily disallowed sites."
    },
    {
      "token": "Googlebot",
      "operator": "Google",
      "purpose": "search",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "The main Search index crawler, and also the crawler behind AI Overviews and AI Mode — blocking it removes you from Google Search entirely, so use nosnippet/max-snippet instead if AI summaries are the concern."
    },
    {
      "token": "Googlebot-Image",
      "operator": "Google",
      "purpose": "search",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Also matches rules written for the Googlebot token; affects Google Images, Discover, and any Search feature showing images, logos, or favicons."
    },
    {
      "token": "Googlebot-News",
      "operator": "Google",
      "purpose": "search",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Has no separate HTTP user-agent string — it crawls with normal Googlebot strings and this token exists only as a robots.txt control for Google News."
    },
    {
      "token": "Googlebot-Video",
      "operator": "Google",
      "purpose": "search",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Also matches rules written for the Googlebot token; controls video-related Search features."
    },
    {
      "token": "GoogleMessages",
      "operator": "Google",
      "purpose": "social-preview",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-user-triggered-fetchers",
      "respects_robots": "no",
      "confidence": "unofficial",
      "notes": "Generates link previews for URLs sent in Google Messages chats; ignores robots.txt, so blocking it only breaks your own link cards if it obeyed rules at all."
    },
    {
      "token": "GoogleOther",
      "operator": "Google",
      "purpose": "other",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Google's generic crawler for internal R&D and one-off crawls; Google states it affects no specific product, so blocking it is low-risk but it is not the AI training control (that is Google-Extended)."
    },
    {
      "token": "GoogleOther-Image",
      "operator": "Google",
      "purpose": "other",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Image-optimized variant of GoogleOther; also matches rules written for the GoogleOther token."
    },
    {
      "token": "GoogleOther-Video",
      "operator": "Google",
      "purpose": "other",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Video-optimized variant of GoogleOther; also matches rules written for the GoogleOther token."
    },
    {
      "token": "GoogleProducer",
      "operator": "Google",
      "purpose": "other",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-user-triggered-fetchers",
      "respects_robots": "no",
      "confidence": "unofficial",
      "notes": "Google Publisher Center fetcher for feeds a publisher explicitly submitted for Google News; ignores robots.txt and has no published robots.txt token."
    },
    {
      "token": "Mediapartners-Google",
      "operator": "Google",
      "purpose": "ads",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-special-case-crawlers",
      "respects_robots": "partial",
      "confidence": "official",
      "notes": "Ignores 'User-agent: *' and must be named explicitly; blocking it stops AdSense from serving relevant ads on the blocked pages."
    },
    {
      "token": "Storebot-Google",
      "operator": "Google",
      "purpose": "search",
      "docs_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Controls all Google Shopping surfaces including the Shopping tab — relevant if you sell anything, and it does not fall back to the Googlebot token."
    },
    {
      "token": "archive.org_bot",
      "operator": "Internet Archive",
      "purpose": "archive",
      "docs_url": "https://archive.org/details/archive.org_bot",
      "respects_robots": "partial",
      "confidence": "official",
      "notes": "Internet Archive publicly announced it stopped obeying robots.txt for US government/military sites and intends to do so more broadly, so treat a Disallow as a request rather than a guarantee."
    },
    {
      "token": "LinkedInBot",
      "operator": "LinkedIn (Microsoft)",
      "purpose": "social-preview",
      "docs_url": "https://www.linkedin.com/robots.txt",
      "respects_robots": "undocumented",
      "confidence": "unofficial",
      "notes": "Blocking this means LinkedIn posts linking to you show a bare URL with no image or headline and Post Inspector fails, but LinkedIn publishes no crawler documentation — the token is confirmed only by the \"User-agent: LinkedInBot\" group in LinkedIn's own robots.txt."
    },
    {
      "token": "MJ12bot",
      "operator": "Majestic (Majestic-12)",
      "purpose": "seo-tool",
      "docs_url": "https://mj12bot.com/",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Page states \"Obeys Robots.txt: Yes\" and \"Obeys Crawl Delay: Yes\"; it is a distributed volunteer crawler with no fixed IP range, so robots.txt or user-agent rules are the only reliable way to control it."
    },
    {
      "token": "search.marginalia.nu",
      "operator": "Marginalia Search",
      "purpose": "search",
      "docs_url": "https://about.marginalia-search.com/article/crawler",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "The token is the bare domain string search.marginalia.nu (there is no \"MarginaliaBot\"), and a full re-crawl only happens every 8-10 weeks."
    },
    {
      "token": "facebookexternalhit",
      "operator": "Meta",
      "purpose": "social-preview",
      "docs_url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers",
      "respects_robots": "partial",
      "confidence": "official",
      "notes": "Blocking this kills link preview cards on Facebook, Instagram, Threads and Messenger, and Meta explicitly reserves an exception: it \"might bypass robots.txt when performing security or integrity checks\" — separate from meta-externalagent (AI training) and meta-externalfetcher (user-requested AI fetches, which \"may bypass robots.txt\")."
    },
    {
      "token": "meta-externalads",
      "operator": "Meta",
      "purpose": "ads",
      "docs_url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Advertising and business products only; not an AI answer surface."
    },
    {
      "token": "meta-externalagent",
      "operator": "Meta",
      "purpose": "ai-training",
      "docs_url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Meta writes it as Meta-ExternalAgent in prose but uses lowercase in its robots.txt example (matching is case-insensitive). Trains foundation models and indexes content; robots.txt edits take up to 24h."
    },
    {
      "token": "meta-externalfetcher",
      "operator": "Meta",
      "purpose": "ai-agent",
      "docs_url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
      "respects_robots": "partial",
      "confidence": "official",
      "notes": "Fetches individual links at a user's request for agentic AI; Meta documents it may bypass robots.txt when user-initiated."
    },
    {
      "token": "meta-webindexer",
      "operator": "Meta",
      "purpose": "ai-answers",
      "docs_url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Newer agent that indexes specifically for Meta AI search quality; blocking it costs Meta AI answer visibility, not just training."
    },
    {
      "token": "WhatsApp",
      "operator": "Meta (WhatsApp)",
      "purpose": "social-preview",
      "docs_url": "https://developers.facebook.com/docs/whatsapp/link-previews/",
      "respects_robots": "undocumented",
      "confidence": "unofficial",
      "notes": "Meta documents the HTTP UA \"WhatsApp/2.x.x.x A|I|N\" and that WhatsApp fetches the URL on the sender's behalf while they type — before the message is even sent — but publishes no robots.txt token or compliance statement, so this line is inferred from the UA's product token and is unlikely to be honoured."
    },
    {
      "token": "bingbot",
      "operator": "Microsoft",
      "purpose": "search",
      "docs_url": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Microsoft ships no separate AI token: Copilot answers and Microsoft generative-AI training are controlled by meta tags - NOARCHIVE excludes you from Copilot answers and training, NOCACHE limits Copilot to URL/title/snippet but still permits training. robots.txt can only remove you from Bing Search entirely."
    },
    {
      "token": "BingPreview",
      "operator": "Microsoft Bing",
      "purpose": "other",
      "docs_url": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Still listed by Bing, but every UA string Bing shows for it now says bingbot/2.0, so a BingPreview rule almost certainly matches nothing - block bingbot instead."
    },
    {
      "token": "BingVideoPreview",
      "operator": "Microsoft Bing",
      "purpose": "other",
      "docs_url": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Fetches video previews shown in Bing only; identifies itself as BingVideoPreview/1.0."
    },
    {
      "token": "MicrosoftPreview",
      "operator": "Microsoft Bing",
      "purpose": "other",
      "docs_url": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Generates page snapshots for Microsoft products and is the one preview bot that still sends its own distinct UA (MicrosoftPreview/2.0)."
    },
    {
      "token": "adidxbot",
      "operator": "Microsoft Bing (Microsoft Advertising)",
      "purpose": "ads",
      "docs_url": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Bing's doc capitalizes it \"AdIdxBot\" but the UA product token is adidxbot; robots.txt matching is case-insensitive, and blocking it only affects ad quality checks, not search."
    },
    {
      "token": "MistralAI-Index",
      "operator": "Mistral AI",
      "purpose": "ai-answers",
      "docs_url": "https://docs.mistral.ai/robots/",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Builds Mistral's search index; Mistral states this content is not used for generative AI training of any kind. Block it and you drop out of Mistral's answer citations."
    },
    {
      "token": "MistralAI-Training",
      "operator": "Mistral AI",
      "purpose": "ai-training",
      "docs_url": "https://docs.mistral.ai/robots/",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Training datasets only; blocking it does not affect Mistral search or Le Chat/Vibe answers."
    },
    {
      "token": "MistralAI-User",
      "operator": "Mistral AI",
      "purpose": "ai-agent",
      "docs_url": "https://docs.mistral.ai/robots/",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "User-initiated fetch for citations; documented as governed by robots.txt and not used for training."
    },
    {
      "token": "MojeekBot",
      "operator": "Mojeek",
      "purpose": "search",
      "docs_url": "https://www.mojeek.com/bot.html",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Obeys the first record whose User-agent contains \"MojeekBot\", falls back to *, and does not support crawl-delay."
    },
    {
      "token": "dotbot",
      "operator": "Moz",
      "purpose": "seo-tool",
      "docs_url": "https://moz.com/help/moz-procedures/crawlers/dotbot",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Moz writes the token lowercase in its own examples (robots.txt matching is case-insensitive, so DotBot works too); it feeds the Moz Link Index behind Link Explorer and the Links API."
    },
    {
      "token": "rogerbot",
      "operator": "Moz",
      "purpose": "seo-tool",
      "docs_url": "https://moz.com/help/moz-procedures/crawlers/rogerbot",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "NOT retired — Moz's page was last modified 2025-05-05 and describes rogerbot in the present tense as the Moz Pro Site Crawl auditor that crawls campaign sites weekly, distinct from dotbot."
    },
    {
      "token": "Yeti",
      "operator": "Naver",
      "purpose": "search",
      "docs_url": "https://searchadvisor.naver.com/guide/seo-basic-robots",
      "respects_robots": "partial",
      "confidence": "official",
      "notes": "Naver's crawler token is Yeti, not \"NaverBot\"; Naver warns that special-purpose robots (ad info, link previews) may not fully follow robots.txt, and treats a 4xx robots.txt as allow-all."
    },
    {
      "token": "ChatGPT-User",
      "operator": "OpenAI",
      "purpose": "ai-agent",
      "docs_url": "https://developers.openai.com/api/docs/bots",
      "respects_robots": "partial",
      "confidence": "official",
      "notes": "User-triggered fetch; OpenAI states 'robots.txt rules may not apply' because a person initiated it, and it does not affect Search inclusion."
    },
    {
      "token": "GPTBot",
      "operator": "OpenAI",
      "purpose": "ai-training",
      "docs_url": "https://developers.openai.com/api/docs/bots",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Training only. Blocking GPTBot does NOT remove you from ChatGPT search answers - that is OAI-SearchBot."
    },
    {
      "token": "OAI-AdsBot",
      "operator": "OpenAI",
      "purpose": "ads",
      "docs_url": "https://developers.openai.com/api/docs/bots",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Only validates safety of landing pages submitted as ChatGPT ads; blocking it has no effect on organic ChatGPT answers."
    },
    {
      "token": "OAI-SearchBot",
      "operator": "OpenAI",
      "purpose": "ai-answers",
      "docs_url": "https://developers.openai.com/api/docs/bots",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "The one that decides ChatGPT Search visibility; OpenAI states opted-out sites will not appear in ChatGPT search results."
    },
    {
      "token": "Perplexity-User",
      "operator": "Perplexity",
      "purpose": "ai-agent",
      "docs_url": "https://docs.perplexity.ai/guides/bots",
      "respects_robots": "no",
      "confidence": "official",
      "notes": "Perplexity documents that this fetcher 'generally ignores robots.txt' because a user requested it. Disputed: Cloudflare and others have publicly accused Perplexity of stealth-crawling disallowed pages under undeclared user agents; Perplexity denies it. Enforce with WAF/IP rules, not robots.txt."
    },
    {
      "token": "PerplexityBot",
      "operator": "Perplexity",
      "purpose": "ai-answers",
      "docs_url": "https://docs.perplexity.ai/guides/bots",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Index for Perplexity answer citations, not model training; Perplexity states neither of its bots trains foundation models. Blocking removes you from Perplexity's cited sources."
    },
    {
      "token": "Pinterestbot",
      "operator": "Pinterest",
      "purpose": "social-preview",
      "docs_url": "https://help.pinterest.com/en/business/article/pinterestbot",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Use \"Pinterestbot\", not \"Pinterest\"; Pinterest says it respects robots.txt but honours Crawl-delay only up to 1 second, and blocking it means Pin titles, prices and images from your site go stale and broken-link Pins stop being cleaned up."
    },
    {
      "token": "ProRataInc",
      "operator": "ProRata.ai / Gist",
      "purpose": "ai-answers",
      "docs_url": "https://platform.gist.ai/docs/prorata-ai-crawler-bot",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Feeds Gist.ai's RAG answer engine (a publisher revenue-share model), not foundation-model training. Docs publish fixed IPs for verification."
    },
    {
      "token": "redditbot",
      "operator": "Reddit",
      "purpose": "social-preview",
      "docs_url": "https://www.reddit.com/robots.txt",
      "respects_robots": "undocumented",
      "confidence": "unofficial",
      "notes": "Weakest row in the set — Reddit publishes no crawler page and its robots.txt (linked here as the only live first-party artefact) does not mention redditbot at all; the token comes from the observed UA \"Mozilla/5.0 (compatible; redditbot/1.0; +http://www.reddit.com/feedback)\", and blocking it costs you the thumbnail on Reddit submissions."
    },
    {
      "token": "SemrushBot",
      "operator": "Semrush",
      "purpose": "seo-tool",
      "docs_url": "https://www.semrush.com/bot/",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "This token covers only the backlink/webgraph crawler — Semrush documents nine more separate tokens (SiteAuditBot, SemrushBot-BA, -SI, -SWA, -OCOB, -FT, -ESI, SplitSignalBot, RyteBot), and changes take \"up to one hour or 100 requests\" to take effect."
    },
    {
      "token": "SeznamBot",
      "operator": "Seznam.cz",
      "purpose": "search",
      "docs_url": "https://napoveda.seznam.cz/en/full-text-search/crawling-control/",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Seznam states it \"fully complies\" with robots.txt and prints the exact block recipe (User-agent: SeznamBot / Disallow: /); rule changes can take days to weeks to apply."
    },
    {
      "token": "Sitebulb",
      "operator": "Sitebulb",
      "purpose": "seo-tool",
      "docs_url": "https://support.sitebulb.com/en/articles/9993764-sitebulb-s-user-agent",
      "respects_robots": "partial",
      "confidence": "unofficial",
      "notes": "Sitebulb publishes HTTP user-agent strings but never a robots.txt token — only the Desktop string contains \"Sitebulb/1.1\"; the Smartphone string is a plain Chrome-Android UA with just \"+https://sitebulb.com\" appended, so this line will not reliably block mobile audits, and users can untick \"Respect Robots Directives\" anyway."
    },
    {
      "token": "Slackbot-LinkExpanding",
      "operator": "Slack (Salesforce)",
      "purpose": "social-preview",
      "docs_url": "https://api.slack.com/robots",
      "respects_robots": "no",
      "confidence": "official",
      "notes": "Putting this in robots.txt does nothing — Slack states \"We do not currently honor robots.txt files\" because it acts on behalf of a human, so you must ask Slack for a blocklist entry or block by user-agent at your edge; Slack-ImgProxy and plain Slackbot are separate agents."
    },
    {
      "token": "TelegramBot",
      "operator": "Telegram",
      "purpose": "social-preview",
      "docs_url": "https://telegram.org/blog/link-preview",
      "respects_robots": "undocumented",
      "confidence": "unofficial",
      "notes": "Telegram publishes no crawler documentation; the token comes from the observed UA \"TelegramBot (like TwitterBot)\" — and because that string also contains \"TwitterBot\", a rule aimed at Twitterbot may catch Telegram previews too if its fetcher reads robots.txt at all."
    },
    {
      "token": "BLEXBot",
      "operator": "WebMeUp / SEO PowerSuite (Link-Assistant.Com)",
      "purpose": "seo-tool",
      "docs_url": "https://webmeup.com/crawler/",
      "respects_robots": "undocumented",
      "confidence": "unofficial",
      "notes": "Lowest-confidence SEO row: the operator's crawler page is gone — webmeup-crawler.com now 301s to link-assistant.com/news/ and webmeup.com/crawler/ lands on a product page — so the token comes only from its self-identifying UA \"Mozilla/5.0 (compatible; BLEXBot/1.0; +http://webmeup.com/crawler/)\", and there is no current published robots.txt commitment."
    },
    {
      "token": "Twitterbot",
      "operator": "X (Twitter)",
      "purpose": "social-preview",
      "docs_url": "https://help.x.com/robots.txt",
      "respects_robots": "undocumented",
      "confidence": "unofficial",
      "notes": "Blocking this means links to your site posted on X render with no summary card, title or image — but X has retired its Cards developer docs (developer.x.com card guides now redirect to docs.x.com/overview), so the token is only confirmable from X's own robots.txt, which contains a \"User-agent: Twitterbot\" group; legacy Twitter docs said it obeys robots.txt but that page is no longer published."
    },
    {
      "token": "Slurp",
      "operator": "Yahoo",
      "purpose": "search",
      "docs_url": "https://help.yahoo.com/kb/SLN22600.html",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Yahoo web results are Bing-powered, so Slurp now crawls mainly for Yahoo Mobile Search and Yahoo News/Finance/Sports content; it obeys the first record whose User-agent contains \"Slurp\"."
    },
    {
      "token": "YandexBot",
      "operator": "Yandex",
      "purpose": "search",
      "docs_url": "https://yandex.com/support/webmaster/robot-workings/check-yandex-robots.html",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Yandex's own table marks YandexBot as honoring general robots.txt rules, while naming several siblings (YaDirectFetcher, YandexMetrika) that ignore robots.txt entirely."
    },
    {
      "token": "YouBot",
      "operator": "You.com",
      "purpose": "ai-answers",
      "docs_url": "https://you.com/docs/youbot",
      "respects_robots": "yes",
      "confidence": "official",
      "notes": "Indexes for You.com search and its search/research APIs that ground third-party LLMs; docs describe indexing, not training. Respects Crawl-delay; caches robots.txt for 30 minutes."
    }
  ]
}