{
 "version": "1.0",
 "updated": "2026-08-24",
 "license": "CC-BY-4.0",
 "source": "https://ipscanner.io/bot",
 "count": 153,
 "agents": [
  {
   "slug": "addsearchbot",
   "name": "AddSearchBot",
   "operator": null,
   "operator_url": null,
   "category": "search",
   "user_agent": "AddSearchBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "agenttimes",
   "name": "AgentTimes",
   "operator": "The Agent Times",
   "operator_url": "https://theagenttimes.com/about",
   "category": "dataset",
   "user_agent": "AgentTimes",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "awario",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "ai2bot",
   "name": "AI2Bot",
   "operator": "Ai2",
   "operator_url": "https://allenai.org/crawler",
   "category": "training",
   "user_agent": "AI2Bot",
   "user_agent_exact": false,
   "aliases": [
    "Ai2Bot-Dolma"
   ],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": "https://allenai.org/crawler",
   "verification_detail": "Ai2 documents the crawler and an opt-out but publishes no address list. Its output feeds the open OLMo and Dolma datasets, so a block here removes the site from public research corpora as well as from a model.",
   "docs": "https://allenai.org/crawler",
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "facebookbot",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "ai2bot-deepresearcheval",
   "name": "AI2Bot-DeepResearchEval",
   "operator": "Ai2, a non-profit AI research institute",
   "operator_url": null,
   "category": "assistant",
   "user_agent": "AI2Bot-DeepResearchEval",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "aihitbot",
   "name": "aiHitBot",
   "operator": "aiHit",
   "operator_url": "https://www.aihitdata.com/about",
   "category": "dataset",
   "user_agent": "aiHitBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "apifybot",
    "apifywebsitecontentcrawler",
    "awario",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "aiwebindex",
   "name": "AIWebIndex",
   "operator": "Lyrenth (Aleksma Ai, Inc.)",
   "operator_url": "https://lyrenth.com/bot",
   "category": "search",
   "user_agent": "AIWebIndex/2.0 (+https://lyrenth.com/bot; AI-readable web index)",
   "user_agent_exact": true,
   "aliases": [
    "AIWebIndex-Agent"
   ],
   "respects_robots": "yes",
   "respects_robots_source": "https://lyrenth.com/crawler-policy",
   "verification_method": "ip-list+reverse-dns",
   "verification_source": "https://lyrenth.com/bot/ip-ranges.json",
   "verification_detail": "Lyrenth publishes its crawl IPs as JSON (gptbot.json format). Reverse DNS resolves under lyrenth.com and forward-resolves back to the same IP. A Web Bot Auth (RFC 9421) key directory is also published at api.lyrenth.com/.well-known/http-message-signatures-directory. All 22 published addresses forward-confirmed on 2026-08-24.",
   "docs": "https://lyrenth.com/crawler-policy",
   "related": [
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot",
    "claude-searchbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "amazon-kendra",
   "name": "amazon-kendra",
   "operator": "Amazon",
   "operator_url": null,
   "category": "search",
   "user_agent": "amazon-kendra",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "aiwebindex",
    "amazonbot",
    "amazonbuyforme",
    "amzn-searchbot",
    "applebot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "amazon-qbusiness",
   "name": "amazon-QBusiness",
   "operator": "Amazon Web Services",
   "operator_url": "https://aws.amazon.com/q/business/",
   "category": "assistant",
   "user_agent": "amazon-QBusiness",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "amazonbot",
   "name": "Amazonbot",
   "operator": "Amazon",
   "operator_url": null,
   "category": "search",
   "user_agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/119.0.0.0 Safari/537.36 (compatible; Amazonbot/0.1; +https://developer.amazon.com/support/amazonbot)",
   "user_agent_exact": true,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "reverse-dns",
   "verification_source": "https://developer.amazon.com/amazonbot",
   "verification_detail": "Verify with a reverse lookup: the PTR record resolves under the Amazon crawler domain, and a forward lookup of that hostname must return the original IP.",
   "docs": "https://developer.amazon.com/amazonbot",
   "related": [
    "aiwebindex",
    "amazon-kendra",
    "amazonbuyforme",
    "amzn-searchbot",
    "applebot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "amazonbuyforme",
   "name": "AmazonBuyForMe",
   "operator": "Amazon",
   "operator_url": "https://amazon.com",
   "category": "agent",
   "user_agent": "AmazonBuyForMe",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amazon-kendra",
    "amazonbot",
    "amzn-searchbot",
    "chatgpt-agent",
    "claude-web"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "amzn-searchbot",
   "name": "Amzn-SearchBot",
   "operator": "Amazon",
   "operator_url": "https://developer.amazon.com/amazonbot",
   "category": "search",
   "user_agent": "Amzn-SearchBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "reverse-dns",
   "verification_source": "https://developer.amazon.com/amazonbot",
   "verification_detail": "Part of the Amazon crawler family documented alongside Amazonbot. Verify with the same reverse-then-forward DNS check.",
   "docs": null,
   "related": [
    "aiwebindex",
    "amazon-kendra",
    "amazonbot",
    "amazonbuyforme",
    "applebot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "amzn-user",
   "name": "Amzn-User",
   "operator": "Amazon",
   "operator_url": "https://developer.amazon.com/amazonbot",
   "category": "assistant",
   "user_agent": "Amzn-User",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "reverse-dns",
   "verification_source": "https://developer.amazon.com/amazonbot",
   "verification_detail": "User-triggered fetch from an Amazon assistant. Same reverse-then-forward DNS check as Amazonbot.",
   "docs": null,
   "related": [
    "amazon-kendra",
    "amazonbot",
    "amazonbuyforme",
    "chatgpt-user",
    "claude-user"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "andibot",
   "name": "Andibot",
   "operator": "Andi",
   "operator_url": "https://andisearch.com/",
   "category": "search",
   "user_agent": "Andibot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "anomura-crawler",
   "name": "Anomura",
   "operator": "Direqt",
   "operator_url": "https://direqt.ai",
   "category": "search",
   "user_agent": "Anomura",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "anthropic-ai",
   "name": "anthropic-ai",
   "operator": "Anthropic",
   "operator_url": "https://www.anthropic.com",
   "category": "training",
   "user_agent": "anthropic-ai",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "ip-list",
   "verification_source": "https://claude.com/crawling/bots.json",
   "verification_detail": "An older token widely copied into robots.txt files. Anthropic now identifies its crawlers as ClaudeBot, Claude-User and Claude-SearchBot, so keep this rule but do not rely on it alone.",
   "docs": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
   "related": [
    "applebot-extended",
    "claude-code",
    "claude-searchbot",
    "claude-user",
    "claudebot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "apifybot",
   "name": "ApifyBot",
   "operator": null,
   "operator_url": null,
   "category": "dataset",
   "user_agent": "ApifyBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifywebsitecontentcrawler",
    "awario",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "apifywebsitecontentcrawler",
   "name": "ApifyWebsiteContentCrawler",
   "operator": null,
   "operator_url": null,
   "category": "dataset",
   "user_agent": "ApifyWebsiteContentCrawler",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "awario",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "applebot",
   "name": "Applebot",
   "operator": "Apple",
   "operator_url": "https://www.apple.com",
   "category": "search",
   "user_agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_5) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/13.1.1 Safari/605.1.15 (Applebot/0.1; +http://www.apple.com/go/applebot)",
   "user_agent_exact": true,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://support.apple.com/en-us/119829#retrieval",
   "verification_method": "ip-list+reverse-dns",
   "verification_source": "https://search.developer.apple.com/applebot.json",
   "verification_detail": "Reverse DNS resolves under applebot.apple.com and must forward-resolve back to the same IP. Apple also publishes the ranges as JSON.",
   "docs": "https://support.apple.com/en-us/119829",
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot-extended",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "applebot-extended",
   "name": "Applebot-Extended",
   "operator": "Apple",
   "operator_url": "https://support.apple.com/en-us/119829#datausage",
   "category": "training",
   "user_agent": "Applebot-Extended",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "robots-token",
   "verification_source": "https://support.apple.com/en-us/119829",
   "verification_detail": "Applebot-Extended is never sent as a User-Agent. It is a robots.txt token that tells Apple not to use already-crawled pages for Apple Intelligence training, while Applebot keeps crawling for Siri and Spotlight.",
   "docs": "https://support.apple.com/en-us/119829",
   "related": [
    "anthropic-ai",
    "applebot",
    "claudebot",
    "facebookbot",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "aranet-searchbot",
   "name": "Aranet-SearchBot",
   "operator": null,
   "operator_url": null,
   "category": "agent",
   "user_agent": "Aranet-SearchBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amazonbuyforme",
    "chatgpt-agent",
    "claude-web",
    "google-cloudvertexbot",
    "openai-operator"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "atlassian-bot",
   "name": "atlassian-bot",
   "operator": "Atlassian",
   "operator_url": "https://www.atlassian.com",
   "category": "search",
   "user_agent": "atlassian-bot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://support.atlassian.com/organization-administration/docs/connect-custom-website-to-rovo/#Editing-your-robots.txt",
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "awario",
   "name": "Awario",
   "operator": "Awario",
   "operator_url": null,
   "category": "dataset",
   "user_agent": "Awario",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "azureai-searchbot",
   "name": "AzureAI-SearchBot",
   "operator": "Microsoft",
   "operator_url": "https://azure.microsoft.com",
   "category": "search",
   "user_agent": "AzureAI-SearchBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "bedrockbot",
   "name": "bedrockbot",
   "operator": "Amazon",
   "operator_url": "https://amazon.com",
   "category": "dataset",
   "user_agent": "bedrockbot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://docs.aws.amazon.com/bedrock/latest/userguide/webcrawl-data-source-connector.html#configuration-webcrawl-connector",
   "verification_method": "none",
   "verification_source": "https://docs.aws.amazon.com/bedrock/latest/userguide/webcrawl-data-source-connector.html",
   "verification_detail": "Runs from an AWS customer account, so the source IP belongs to whoever configured the Bedrock knowledge base rather than to AWS as a crawler operator.",
   "docs": null,
   "related": [
    "agenttimes",
    "amazon-kendra",
    "amazonbot",
    "amazonbuyforme",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "bigsur-ai",
   "name": "bigsur.ai",
   "operator": null,
   "operator_url": null,
   "category": "assistant",
   "user_agent": "bigsur.ai",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "bingbot",
   "name": "Bingbot",
   "operator": "Microsoft",
   "operator_url": "https://www.bing.com",
   "category": "search",
   "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; bingbot/2.0; +http://www.bing.com/bingbot.htm) Chrome/W.X.Y.Z Safari/537.36",
   "user_agent_exact": true,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
   "verification_method": "ip-list+reverse-dns",
   "verification_source": "https://www.bing.com/toolbox/bingbot.json",
   "verification_detail": "Reverse DNS must resolve to a search.msn.com host and forward-resolve back to the same IP. Microsoft also publishes the ranges as JSON.",
   "docs": "https://www.bing.com/webmasters/help/how-to-verify-bingbot-3905dc26",
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "azureai-searchbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "bravebot",
   "name": "Bravebot",
   "operator": "Brave",
   "operator_url": "https://brave.com",
   "category": "dataset",
   "user_agent": "Bravebot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "brightbot",
   "name": "Brightbot",
   "operator": null,
   "operator_url": null,
   "category": "dataset",
   "user_agent": "Brightbot",
   "user_agent_exact": false,
   "aliases": [
    "Brightbot 1.0"
   ],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "buddybot",
   "name": "BuddyBot",
   "operator": "BuddyBotLearning",
   "operator_url": "https://www.buddybotlearning.com",
   "category": "other",
   "user_agent": "BuddyBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "echoboxbot",
    "facebookexternalhit",
    "google-firebase",
    "openai-bot",
    "panscient"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "bytespider",
   "name": "Bytespider",
   "operator": "ByteDance",
   "operator_url": null,
   "category": "training",
   "user_agent": "Bytespider",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "no",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "ByteDance publishes no IP ranges and no reverse-DNS convention for Bytespider, and it has been widely reported crawling aggressively from addresses that do not resolve back to ByteDance. Rate limiting is more reliable than a robots.txt rule.",
   "docs": null,
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "tiktokspider",
    "trae-agent"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "ccbot",
   "name": "CCBot",
   "operator": "Common Crawl Foundation",
   "operator_url": "https://commoncrawl.org",
   "category": "dataset",
   "user_agent": "CCBot/2.0 (https://commoncrawl.org/faq/)",
   "user_agent_exact": true,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://commoncrawl.org/ccbot",
   "verification_method": "none",
   "verification_source": "https://commoncrawl.org/ccbot",
   "verification_detail": "Common Crawl publishes no IP list and runs from rented cloud capacity, so the address alone proves nothing. Its archive is a training input for many models, which is why blocking it has knock-on effects.",
   "docs": "https://commoncrawl.org/ccbot",
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "channel3bot",
   "name": "Channel3Bot",
   "operator": null,
   "operator_url": null,
   "category": "search",
   "user_agent": "Channel3Bot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "chatglm-spider",
   "name": "ChatGLM-Spider",
   "operator": "Zhipu AI",
   "operator_url": "https://www.zhipuai.cn",
   "category": "dataset",
   "user_agent": "ChatGLM-Spider",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "chatgpt-agent",
   "name": "ChatGPT Agent",
   "operator": "OpenAI",
   "operator_url": "https://openai.com",
   "category": "agent",
   "user_agent": "ChatGPT Agent",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "ip-list",
   "verification_source": "https://platform.openai.com/docs/bots",
   "verification_detail": "OpenAI publishes one range file per bot. Use the file listed against this agent in the bots documentation, not the GPTBot file.",
   "docs": "https://platform.openai.com/docs/bots",
   "related": [
    "chatgpt-user",
    "claude-web",
    "google-cloudvertexbot",
    "gptbot",
    "oai-searchbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "chatgpt-user",
   "name": "ChatGPT-User",
   "operator": "OpenAI",
   "operator_url": "https://openai.com",
   "category": "assistant",
   "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; ChatGPT-User/1.0; +https://openai.com/bot",
   "user_agent_exact": true,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "ip-list",
   "verification_source": "https://openai.com/chatgpt-user.json",
   "verification_detail": "Fetches are triggered by a person asking ChatGPT about a URL, so traffic is spiky rather than a steady crawl. The egress ranges are published as JSON.",
   "docs": "https://platform.openai.com/docs/bots",
   "related": [
    "amzn-user",
    "chatgpt-agent",
    "claude-user",
    "gptbot",
    "oai-searchbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "claude-code",
   "name": "Claude-Code",
   "operator": "Anthropic",
   "operator_url": "https://www.anthropic.com",
   "category": "coding",
   "user_agent": "Claude-Code",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "ip-list",
   "verification_source": "https://claude.com/crawling/bots.json",
   "verification_detail": "Runs on a developer machine or CI runner, so the source IP is usually the developer, not Anthropic. The published prefix file will not match most of this traffic.",
   "docs": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
   "related": [
    "anthropic-ai",
    "claude-searchbot",
    "claude-user",
    "cursor-agent",
    "github-copilot-code"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "claude-searchbot",
   "name": "Claude-SearchBot",
   "operator": "Anthropic",
   "operator_url": "https://www.anthropic.com",
   "category": "search",
   "user_agent": "Mozilla/5.0 (compatible; Claude-SearchBot/1.0; +Claude-SearchBot@anthropic.com)",
   "user_agent_exact": true,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://support.anthropic.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
   "verification_method": "ip-list",
   "verification_source": "https://claude.com/crawling/bots.json",
   "verification_detail": "Indexes pages so Claude can cite them in search answers. Same published prefix file as the other Anthropic agents.",
   "docs": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
   "related": [
    "aiwebindex",
    "amazonbot",
    "anthropic-ai",
    "claude-code",
    "claude-user"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "claude-user",
   "name": "Claude-User",
   "operator": "Anthropic",
   "operator_url": "https://www.anthropic.com",
   "category": "assistant",
   "user_agent": "Mozilla/5.0 (compatible; Claude-User/1.0; +Claude-User@anthropic.com)",
   "user_agent_exact": true,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://support.anthropic.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
   "verification_method": "ip-list",
   "verification_source": "https://claude.com/crawling/bots.json",
   "verification_detail": "Fired when a Claude user asks about a specific page, so it arrives one URL at a time. Covered by the same published prefix file as ClaudeBot.",
   "docs": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
   "related": [
    "amzn-user",
    "anthropic-ai",
    "chatgpt-user",
    "claude-code",
    "claude-searchbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "claude-web",
   "name": "Claude-Web",
   "operator": "Anthropic",
   "operator_url": null,
   "category": "agent",
   "user_agent": "Claude-Web",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "ip-list",
   "verification_source": "https://claude.com/crawling/bots.json",
   "verification_detail": "A legacy token that predates the current ClaudeBot naming. Site owners still see it in old logs and old robots.txt files; check the published prefix file before trusting it.",
   "docs": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
   "related": [
    "anthropic-ai",
    "chatgpt-agent",
    "claude-code",
    "claude-searchbot",
    "google-cloudvertexbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "claudebot",
   "name": "ClaudeBot",
   "operator": "Anthropic",
   "operator_url": "https://www.anthropic.com",
   "category": "training",
   "user_agent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
   "user_agent_exact": true,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://support.anthropic.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
   "verification_method": "ip-list",
   "verification_source": "https://claude.com/crawling/bots.json",
   "verification_detail": "Anthropic publishes a single prefix file covering ClaudeBot, Claude-User and Claude-SearchBot. A request claiming to be ClaudeBot from outside those prefixes is a forgery.",
   "docs": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claude-code",
    "claude-searchbot",
    "facebookbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "cloudflare-autorag",
   "name": "Cloudflare-AutoRAG",
   "operator": "Cloudflare",
   "operator_url": "https://developers.cloudflare.com/autorag",
   "category": "search",
   "user_agent": "Cloudflare-AutoRAG",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "cohere-ai",
   "name": "cohere-ai",
   "operator": "Cohere",
   "operator_url": "https://cohere.com",
   "category": "training",
   "user_agent": "cohere-ai",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "Cohere publishes no IP range list for this token. The separate cohere-training-data-crawler token is the one aimed at model training.",
   "docs": null,
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "cohere-training-data-crawler",
    "facebookbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "cohere-training-data-crawler",
   "name": "cohere-training-data-crawler",
   "operator": "Cohere",
   "operator_url": "https://cohere.com",
   "category": "dataset",
   "user_agent": "cohere-training-data-crawler",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "cohere-ai",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "cotoyogi",
   "name": "Cotoyogi",
   "operator": "ROIS",
   "operator_url": "https://ds.rois.ac.jp/en_center8/en_crawler/",
   "category": "training",
   "user_agent": "Cotoyogi",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "facebookbot",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "cragcrawler",
   "name": "CragCrawler",
   "operator": null,
   "operator_url": null,
   "category": "dataset",
   "user_agent": "CragCrawler",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "crawl4ai",
   "name": "Crawl4AI",
   "operator": "Crawl4AI (open source)",
   "operator_url": "https://github.com/unclecode/crawl4ai",
   "category": "agent",
   "user_agent": "Crawl4AI",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": "https://github.com/unclecode/crawl4ai",
   "verification_detail": "Open source crawler library used to feed pages into LLM pipelines. Anyone can run it under any user agent, so this token is a courtesy rather than an identity.",
   "docs": null,
   "related": [
    "amazonbuyforme",
    "chatgpt-agent",
    "claude-web",
    "google-cloudvertexbot",
    "openai-operator"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "crawlspace",
   "name": "Crawlspace",
   "operator": "Crawlspace",
   "operator_url": "https://crawlspace.dev",
   "category": "dataset",
   "user_agent": "Crawlspace",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://news.ycombinator.com/item?id=42756654",
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "cursor-agent",
   "name": "Cursor",
   "operator": "Cursor",
   "operator_url": "https://cursor.com",
   "category": "coding",
   "user_agent": "Cursor",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "claude-code",
    "devin-ai",
    "github-copilot-code",
    "google-gemini-cli",
    "opencode"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "datenbank-crawler",
   "name": "Datenbank Crawler",
   "operator": "Datenbank",
   "operator_url": null,
   "category": "dataset",
   "user_agent": "Datenbank Crawler",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "deepseekbot",
   "name": "DeepSeekBot",
   "operator": "DeepSeek",
   "operator_url": null,
   "category": "training",
   "user_agent": "DeepSeekBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "no",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "facebookbot",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "devin-ai",
   "name": "Devin",
   "operator": "Devin AI",
   "operator_url": null,
   "category": "coding",
   "user_agent": "Devin",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "claude-code",
    "cursor-agent",
    "github-copilot-code",
    "google-gemini-cli",
    "opencode"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "diffbot",
   "name": "Diffbot",
   "operator": "Diffbot",
   "operator_url": "https://www.diffbot.com/",
   "category": "dataset",
   "user_agent": "Diffbot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": "https://docs.diffbot.com/docs/en/guides-diffbot-crawler",
   "verification_detail": "Whether it obeys robots.txt is set per customer, so two Diffbot crawls of the same site can behave differently. No published IP list.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "duckassistbot",
   "name": "DuckAssistBot",
   "operator": null,
   "operator_url": null,
   "category": "assistant",
   "user_agent": "DuckAssistBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot/",
   "verification_method": "ip-list",
   "verification_source": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot/",
   "verification_detail": "DuckDuckGo documents the agent and the addresses it fetches from on its help pages.",
   "docs": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot/",
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "google-notebooklm",
    "meta-externalfetcher"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "duckduckbot",
   "name": "DuckDuckBot",
   "operator": "DuckDuckGo",
   "operator_url": "https://duckduckgo.com",
   "category": "search",
   "user_agent": "DuckDuckBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://duckduckgo.com/duckduckgo-help-pages/results/duckduckbot/",
   "verification_method": "ip-list",
   "verification_source": "https://duckduckgo.com/duckduckgo-help-pages/results/duckduckbot/",
   "verification_detail": "DuckDuckGo publishes the crawler addresses on its help pages.",
   "docs": "https://duckduckgo.com/duckduckgo-help-pages/results/duckduckbot/",
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "echobot-bot",
   "name": "Echobot Bot",
   "operator": "Echobox",
   "operator_url": null,
   "category": "dataset",
   "user_agent": "Echobot Bot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "echoboxbot",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "echoboxbot",
   "name": "EchoboxBot",
   "operator": "Echobox",
   "operator_url": "https://echobox.com",
   "category": "other",
   "user_agent": "EchoboxBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "buddybot",
    "echobot-bot",
    "facebookexternalhit",
    "google-firebase",
    "openai-bot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "exabot",
   "name": "ExaBot",
   "operator": "Exa",
   "operator_url": "https://exa.ai",
   "category": "dataset",
   "user_agent": "ExaBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "exasearchbot",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "exasearchbot",
   "name": "ExaSearchBot",
   "operator": "Exa",
   "operator_url": "https://exa.ai",
   "category": "search",
   "user_agent": "ExaSearchBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "exabot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "facebookbot",
   "name": "FacebookBot",
   "operator": "Meta/Facebook",
   "operator_url": null,
   "category": "training",
   "user_agent": "FacebookBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://developers.facebook.com/docs/sharing/bot/",
   "verification_method": "asn",
   "verification_source": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
   "verification_detail": "Check the source IP against AS32934. Meta publishes no per-bot range file.",
   "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "facebookexternalhit",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "facebookexternalhit",
   "name": "facebookexternalhit",
   "operator": "Meta/Facebook",
   "operator_url": null,
   "category": "other",
   "user_agent": "facebookexternalhit",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "no",
   "respects_robots_source": "https://github.com/ai-robots-txt/ai.robots.txt/issues/40#issuecomment-2524591313",
   "verification_method": "asn",
   "verification_source": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
   "verification_detail": "The link-preview scraper. It has existed far longer than the AI agents and is the most commonly spoofed Meta token, so check AS32934 before trusting it.",
   "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
   "related": [
    "buddybot",
    "echoboxbot",
    "facebookbot",
    "google-firebase",
    "openai-bot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "factset-spyderbot",
   "name": "Factset_spyderbot",
   "operator": "Factset",
   "operator_url": "https://www.factset.com/ai",
   "category": "training",
   "user_agent": "Factset_spyderbot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "facebookbot",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "firecrawlagent",
   "name": "FirecrawlAgent",
   "operator": "Firecrawl",
   "operator_url": "https://firecrawl.dev",
   "category": "dataset",
   "user_agent": "FirecrawlAgent",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": "https://docs.firecrawl.dev",
   "verification_detail": "Firecrawl runs crawls on behalf of its customers, so the traffic belongs to whichever developer called the API rather than to Firecrawl itself.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "friendlycrawler",
   "name": "FriendlyCrawler",
   "operator": "Unknown",
   "operator_url": null,
   "category": "training",
   "user_agent": "FriendlyCrawler",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://imho.alex-kunz.com/2024/01/25/an-update-on-friendly-crawler",
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "facebookbot",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "geisthaus-pagefetcher",
   "name": "GeistHaus-PageFetcher",
   "operator": null,
   "operator_url": null,
   "category": "assistant",
   "user_agent": "GeistHaus-PageFetcher",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "gemini-deep-research",
   "name": "Gemini-Deep-Research",
   "operator": "Google",
   "operator_url": "https://www.google.com",
   "category": "assistant",
   "user_agent": "Gemini-Deep-Research",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "related": [
    "amzn-user",
    "chatgpt-user",
    "google-agent",
    "google-cloudvertexbot",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "github-copilot-code",
   "name": "Code",
   "operator": "GitHub Copilot",
   "operator_url": "https://github.com/features/copilot",
   "category": "coding",
   "user_agent": "Code",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "claude-code",
    "cursor-agent",
    "devin-ai",
    "google-gemini-cli",
    "opencode"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "google-agent",
   "name": "Google-Agent",
   "operator": "Google",
   "operator_url": "https://www.google.com",
   "category": "agent",
   "user_agent": "Google-Agent",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers#google-agent",
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "related": [
    "chatgpt-agent",
    "claude-web",
    "gemini-deep-research",
    "google-cloudvertexbot",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "google-cloudvertexbot",
   "name": "Google-CloudVertexBot",
   "operator": "Google",
   "operator_url": null,
   "category": "agent",
   "user_agent": "Google-CloudVertexBot",
   "user_agent_exact": false,
   "aliases": [
    "CloudVertexBot"
   ],
   "respects_robots": "yes",
   "respects_robots_source": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "verification_method": "ip-list",
   "verification_source": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "verification_detail": "Crawls on behalf of a Vertex AI customer building their own grounded agent. Blocking it blocks that customer, not Google Search.",
   "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "related": [
    "chatgpt-agent",
    "claude-web",
    "gemini-deep-research",
    "google-agent",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "google-extended",
   "name": "Google-Extended",
   "operator": "Google",
   "operator_url": null,
   "category": "training",
   "user_agent": "Google-Extended",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "verification_method": "robots-token",
   "verification_source": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "verification_detail": "Google-Extended never appears in a User-Agent header. It is a robots.txt token only, used to opt out of Gemini training and grounding while leaving Google Search crawling untouched. There is nothing to IP-verify.",
   "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "gemini-deep-research",
    "google-agent",
    "google-cloudvertexbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "google-firebase",
   "name": "Google-Firebase",
   "operator": "Google",
   "operator_url": null,
   "category": "other",
   "user_agent": "Google-Firebase",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "related": [
    "facebookexternalhit",
    "gemini-deep-research",
    "google-agent",
    "google-cloudvertexbot",
    "openai-bot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "google-gemini-cli",
   "name": "Google-Gemini-CLI",
   "operator": "Google",
   "operator_url": "https://www.google.com",
   "category": "coding",
   "user_agent": "Google-Gemini-CLI",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "related": [
    "claude-code",
    "gemini-deep-research",
    "github-copilot-code",
    "google-agent",
    "google-cloudvertexbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "google-notebooklm",
   "name": "Google-NotebookLM",
   "operator": "Google",
   "operator_url": "https://www.google.com",
   "category": "assistant",
   "user_agent": "Google-NotebookLM",
   "user_agent_exact": false,
   "aliases": [
    "NotebookLM"
   ],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "ip-list",
   "verification_source": "https://developers.google.com/static/search/apis/ipranges/user-triggered-fetchers-google.json",
   "verification_detail": "Fetches a URL a NotebookLM user pasted in. Google groups this with its user-triggered fetchers, which have their own range file.",
   "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "related": [
    "amzn-user",
    "chatgpt-user",
    "gemini-deep-research",
    "google-agent",
    "google-cloudvertexbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "googleagent-mariner",
   "name": "GoogleAgent-Mariner",
   "operator": "Google",
   "operator_url": null,
   "category": "agent",
   "user_agent": "GoogleAgent-Mariner",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "related": [
    "chatgpt-agent",
    "claude-web",
    "gemini-deep-research",
    "google-agent",
    "google-cloudvertexbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "googleagent-urlcontext",
   "name": "GoogleAgent-URLContext",
   "operator": "Google",
   "operator_url": "https://www.google.com",
   "category": "assistant",
   "user_agent": "GoogleAgent-URLContext",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "related": [
    "amzn-user",
    "chatgpt-user",
    "gemini-deep-research",
    "google-agent",
    "google-cloudvertexbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "googlebot",
   "name": "Googlebot",
   "operator": "Google",
   "operator_url": "https://www.google.com",
   "category": "search",
   "user_agent": "Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)",
   "user_agent_exact": true,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "verification_method": "ip-list+reverse-dns",
   "verification_source": "https://developers.google.com/static/search/apis/ipranges/googlebot.json",
   "verification_detail": "Two options: reverse DNS on the source IP must return a googlebot.com or google.com host and a forward lookup of that host must return the same IP, or the IP must appear in googlebot.json.",
   "docs": "https://developers.google.com/search/docs/crawling-indexing/verifying-googlebot",
   "related": [
    "aiwebindex",
    "amazonbot",
    "gemini-deep-research",
    "google-agent",
    "google-cloudvertexbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "googleother",
   "name": "GoogleOther",
   "operator": "Google",
   "operator_url": null,
   "category": "search",
   "user_agent": "GoogleOther",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "verification_method": "ip-list+reverse-dns",
   "verification_source": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "verification_detail": "Listed in Google's special-crawlers range file, and reverse DNS resolves under googlebot.com. Used by internal Google teams for one-off crawls rather than by Search indexing.",
   "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "related": [
    "aiwebindex",
    "amazonbot",
    "gemini-deep-research",
    "google-agent",
    "google-cloudvertexbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "googleother-image",
   "name": "GoogleOther-Image",
   "operator": "Google",
   "operator_url": null,
   "category": "search",
   "user_agent": "GoogleOther-Image",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "verification_method": "ip-list+reverse-dns",
   "verification_source": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "verification_detail": "Image variant of GoogleOther, covered by the same special-crawlers range file.",
   "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "related": [
    "aiwebindex",
    "amazonbot",
    "gemini-deep-research",
    "google-agent",
    "google-cloudvertexbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "googleother-video",
   "name": "GoogleOther-Video",
   "operator": "Google",
   "operator_url": null,
   "category": "search",
   "user_agent": "GoogleOther-Video",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "verification_method": "ip-list+reverse-dns",
   "verification_source": "https://developers.google.com/static/search/apis/ipranges/special-crawlers.json",
   "verification_detail": "Video variant of GoogleOther, covered by the same special-crawlers range file.",
   "docs": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers",
   "related": [
    "aiwebindex",
    "amazonbot",
    "gemini-deep-research",
    "google-agent",
    "google-cloudvertexbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "gptbot",
   "name": "GPTBot",
   "operator": "OpenAI",
   "operator_url": "https://openai.com",
   "category": "training",
   "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.2; +https://openai.com/gptbot",
   "user_agent_exact": true,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "ip-list",
   "verification_source": "https://openai.com/gptbot.json",
   "verification_detail": "OpenAI publishes the GPTBot egress ranges as a JSON file. Match the request IP against that file; there is no reverse-DNS convention.",
   "docs": "https://platform.openai.com/docs/bots",
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "chatgpt-agent",
    "chatgpt-user",
    "oai-searchbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "henkbot",
   "name": "HenkBot",
   "operator": null,
   "operator_url": null,
   "category": "dataset",
   "user_agent": "HenkBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "iaskbot",
   "name": "iAskBot",
   "operator": "iAsk",
   "operator_url": "https://iask.ai",
   "category": "agent",
   "user_agent": "iAskBot",
   "user_agent_exact": false,
   "aliases": [
    "iaskspider",
    "iaskspider/2.0"
   ],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amazonbuyforme",
    "chatgpt-agent",
    "claude-web",
    "google-cloudvertexbot",
    "openai-operator"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "icc-crawler",
   "name": "ICC-Crawler",
   "operator": "NICT",
   "operator_url": "https://nict.go.jp",
   "category": "training",
   "user_agent": "ICC-Crawler",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "facebookbot",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "imagesiftbot",
   "name": "ImagesiftBot",
   "operator": "ImageSift",
   "operator_url": "https://imagesift.com",
   "category": "search",
   "user_agent": "ImagesiftBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://imagesift.com/about",
   "verification_method": "none",
   "verification_source": "https://imagesift.com/about",
   "verification_detail": "Crawls for images rather than text, so a text-only robots.txt review can miss it entirely.",
   "docs": "https://imagesift.com/about",
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "imagespider",
   "name": "imageSpider",
   "operator": null,
   "operator_url": null,
   "category": "dataset",
   "user_agent": "imageSpider",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "img2dataset",
   "name": "img2dataset",
   "operator": "img2dataset",
   "operator_url": "https://github.com/rom1504/img2dataset",
   "category": "training",
   "user_agent": "img2dataset",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "facebookbot",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "isscyberriskcrawler",
   "name": "ISSCyberRiskCrawler",
   "operator": "ISS-Corporate",
   "operator_url": "https://iss-cyber.com",
   "category": "training",
   "user_agent": "ISSCyberRiskCrawler",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "no",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "facebookbot",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "kagi-fetcher",
   "name": "kagi-fetcher",
   "operator": null,
   "operator_url": null,
   "category": "assistant",
   "user_agent": "kagi-fetcher",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "kangaroo-bot",
   "name": "Kangaroo Bot",
   "operator": "Kangaroo LLM",
   "operator_url": "https://kangaroollm.com",
   "category": "dataset",
   "user_agent": "Kangaroo Bot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "kimi-user",
   "name": "Kimi-User",
   "operator": "Moonshot AI",
   "operator_url": "https://kimi.moonshot.cn",
   "category": "assistant",
   "user_agent": "Kimi-User",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "klaviyoaibot",
   "name": "KlaviyoAIBot",
   "operator": "Klaviyo",
   "operator_url": "https://www.klaviyo.com",
   "category": "assistant",
   "user_agent": "KlaviyoAIBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://help.klaviyo.com/hc/en-us/articles/40496146232219",
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "kunatocrawler",
   "name": "KunatoCrawler",
   "operator": null,
   "operator_url": null,
   "category": "agent",
   "user_agent": "KunatoCrawler",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amazonbuyforme",
    "chatgpt-agent",
    "claude-web",
    "google-cloudvertexbot",
    "openai-operator"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "laion-huggingface-processor",
   "name": "laion-huggingface-processor",
   "operator": "LAION",
   "operator_url": "https://laion.ai",
   "category": "dataset",
   "user_agent": "laion-huggingface-processor",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "laiondownloader",
   "name": "LAIONDownloader",
   "operator": "Large-scale Artificial Intelligence Open Network",
   "operator_url": "https://laion.ai/",
   "category": "dataset",
   "user_agent": "LAIONDownloader",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "no",
   "respects_robots_source": "https://laion.ai/faq/",
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "lcc-crawler",
   "name": "LCC",
   "operator": null,
   "operator_url": null,
   "category": "dataset",
   "user_agent": "LCC",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "lightpanda",
   "name": "Lightpanda",
   "operator": null,
   "operator_url": null,
   "category": "dataset",
   "user_agent": "Lightpanda",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "linerbot",
   "name": "LinerBot",
   "operator": null,
   "operator_url": null,
   "category": "assistant",
   "user_agent": "LinerBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "linguee-bot",
   "name": "Linguee Bot",
   "operator": "Linguee",
   "operator_url": "https://www.linguee.com",
   "category": "training",
   "user_agent": "Linguee Bot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "no",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "facebookbot",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "linkupbot",
   "name": "LinkupBot",
   "operator": "Linkup",
   "operator_url": "https://www.linkup.so",
   "category": "search",
   "user_agent": "LinkupBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "manus-user",
   "name": "Manus-User",
   "operator": "Butterfly Effect, a company based in China",
   "operator_url": null,
   "category": "agent",
   "user_agent": "Manus-User",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amazonbuyforme",
    "chatgpt-agent",
    "claude-web",
    "google-cloudvertexbot",
    "openai-operator"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "meta-externalagent",
   "name": "Meta-ExternalAgent",
   "operator": "Meta",
   "operator_url": "https://www.meta.com",
   "category": "training",
   "user_agent": "Meta-ExternalAgent",
   "user_agent_exact": false,
   "aliases": [
    "meta-externalagent"
   ],
   "respects_robots": "yes",
   "respects_robots_source": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
   "verification_method": "asn",
   "verification_source": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
   "verification_detail": "Meta does not publish a per-bot range file. The practical check is whether the source IP is announced by AS32934 (Facebook), which you can query from a routing registry.",
   "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "meta-externalfetcher",
    "meta-webindexer"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "meta-externalfetcher",
   "name": "Meta-ExternalFetcher",
   "operator": "Meta",
   "operator_url": "https://www.meta.com",
   "category": "assistant",
   "user_agent": "Meta-ExternalFetcher",
   "user_agent_exact": false,
   "aliases": [
    "meta-externalfetcher"
   ],
   "respects_robots": "no",
   "respects_robots_source": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
   "verification_method": "asn",
   "verification_source": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
   "verification_detail": "Same AS32934 check as the other Meta agents. This one fetches a single URL a user shared with a Meta AI assistant.",
   "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "meta-externalagent",
    "meta-webindexer"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "meta-webindexer",
   "name": "meta-webindexer",
   "operator": "Meta",
   "operator_url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
   "category": "search",
   "user_agent": "meta-webindexer",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "asn",
   "verification_source": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
   "verification_detail": "Check the source IP against AS32934. Meta publishes no per-bot range file.",
   "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "meta-externalagent",
    "meta-externalfetcher"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "mistralai-user",
   "name": "MistralAI-User",
   "operator": "Mistral",
   "operator_url": null,
   "category": "assistant",
   "user_agent": "MistralAI-User",
   "user_agent_exact": false,
   "aliases": [
    "MistralAI-User/1.0"
   ],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "mozilla-tabstack",
   "name": "Mozilla-Tabstack",
   "operator": "Mozilla",
   "operator_url": "https://docs.tabstack.ai/trust/controlling-access",
   "category": "dataset",
   "user_agent": "Mozilla-Tabstack",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "mycentralaiscraperbot",
   "name": "MyCentralAIScraperBot",
   "operator": null,
   "operator_url": null,
   "category": "dataset",
   "user_agent": "MyCentralAIScraperBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "nagetbot",
   "name": "NagetBot",
   "operator": null,
   "operator_url": null,
   "category": "agent",
   "user_agent": "NagetBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amazonbuyforme",
    "chatgpt-agent",
    "claude-web",
    "google-cloudvertexbot",
    "openai-operator"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "netestate-imprint-crawler",
   "name": "netEstate Imprint Crawler",
   "operator": "netEstate",
   "operator_url": null,
   "category": "dataset",
   "user_agent": "netEstate Imprint Crawler",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "newsai",
   "name": "newsai",
   "operator": null,
   "operator_url": null,
   "category": "agent",
   "user_agent": "newsai",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amazonbuyforme",
    "chatgpt-agent",
    "claude-web",
    "google-cloudvertexbot",
    "openai-operator"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "novaact",
   "name": "NovaAct",
   "operator": "Amazon",
   "operator_url": "https://labs.amazon.science/blog/nova-act",
   "category": "agent",
   "user_agent": "NovaAct",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amazon-kendra",
    "amazonbot",
    "amazonbuyforme",
    "chatgpt-agent",
    "claude-web"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "oai-searchbot",
   "name": "OAI-SearchBot",
   "operator": "OpenAI",
   "operator_url": "https://openai.com",
   "category": "search",
   "user_agent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; OAI-SearchBot/1.0; +https://openai.com/searchbot",
   "user_agent_exact": true,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://platform.openai.com/docs/bots",
   "verification_method": "ip-list",
   "verification_source": "https://openai.com/searchbot.json",
   "verification_detail": "OpenAI publishes a separate range file for the search crawler. Blocking it removes the site from ChatGPT search results, which is a different decision from blocking training.",
   "docs": "https://platform.openai.com/docs/bots",
   "related": [
    "aiwebindex",
    "amazonbot",
    "chatgpt-agent",
    "chatgpt-user",
    "gptbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "omgili",
   "name": "omgili",
   "operator": "Webz.io",
   "operator_url": "https://webz.io/",
   "category": "dataset",
   "user_agent": "omgili",
   "user_agent_exact": false,
   "aliases": [
    "omgilibot"
   ],
   "respects_robots": "yes",
   "respects_robots_source": "https://webz.io/blog/web-data/what-is-the-omgili-bot-and-why-is-it-crawling-your-website/",
   "verification_method": "none",
   "verification_source": "https://webz.io/blog/web-data/what-is-the-omgili-bot-and-why-is-it-crawling-your-website/",
   "verification_detail": "Webz.io offers its crawl output as licensed data feeds, so blocking it limits future collection of your pages. No published address list.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "openai-bot",
   "name": "OpenAI",
   "operator": "OpenAI",
   "operator_url": "https://openai.com",
   "category": "other",
   "user_agent": "OpenAI",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "ip-list",
   "verification_source": "https://platform.openai.com/docs/bots",
   "verification_detail": "A bare \"OpenAI\" token is not one of the three documented crawlers. Treat it as unverified unless the source IP appears in one of the published OpenAI range files.",
   "docs": "https://platform.openai.com/docs/bots",
   "related": [
    "buddybot",
    "chatgpt-agent",
    "chatgpt-user",
    "facebookexternalhit",
    "gptbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "openai-operator",
   "name": "Operator",
   "operator": "OpenAI",
   "operator_url": "https://openai.com",
   "category": "agent",
   "user_agent": "Operator",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "ip-list",
   "verification_source": "https://platform.openai.com/docs/bots",
   "verification_detail": "Browser-driving agent run on behalf of a signed-in user. OpenAI lists the matching egress range file in its bots documentation.",
   "docs": "https://platform.openai.com/docs/bots",
   "related": [
    "chatgpt-agent",
    "chatgpt-user",
    "claude-web",
    "google-cloudvertexbot",
    "gptbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "opencode",
   "name": "opencode",
   "operator": "opencode (open source)",
   "operator_url": "https://opencode.ai",
   "category": "coding",
   "user_agent": "opencode",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "claude-code",
    "cursor-agent",
    "devin-ai",
    "github-copilot-code",
    "google-gemini-cli"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "pangubot",
   "name": "PanguBot",
   "operator": "the Chinese company Huawei",
   "operator_url": null,
   "category": "dataset",
   "user_agent": "PanguBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "panscient",
   "name": "Panscient",
   "operator": "Panscient",
   "operator_url": "https://panscient.com",
   "category": "other",
   "user_agent": "Panscient",
   "user_agent_exact": false,
   "aliases": [
    "panscient.com"
   ],
   "respects_robots": "yes",
   "respects_robots_source": "https://panscient.com/faq.htm",
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "buddybot",
    "echoboxbot",
    "facebookexternalhit",
    "google-firebase",
    "openai-bot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "perplexity-user",
   "name": "Perplexity-User",
   "operator": "Perplexity",
   "operator_url": "https://www.perplexity.ai/",
   "category": "assistant",
   "user_agent": "Perplexity-User",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "no",
   "respects_robots_source": "https://docs.perplexity.ai/guides/bots",
   "verification_method": "ip-list",
   "verification_source": "https://www.perplexity.com/perplexity-user.json",
   "verification_detail": "User-triggered fetch with its own published range file. Perplexity documents that this agent ignores robots.txt because a person asked for the page.",
   "docs": "https://docs.perplexity.ai/guides/bots",
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "perplexitybot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "perplexitybot",
   "name": "PerplexityBot",
   "operator": "Perplexity",
   "operator_url": "https://www.perplexity.ai/",
   "category": "search",
   "user_agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot",
   "user_agent_exact": true,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://docs.perplexity.ai/guides/bots",
   "verification_method": "ip-list",
   "verification_source": "https://www.perplexity.com/perplexitybot.json",
   "verification_detail": "Perplexity publishes the crawler ranges as JSON. Independent researchers have reported traffic that claims to be PerplexityBot from outside those ranges, so the IP check matters here more than most.",
   "docs": "https://docs.perplexity.ai/guides/bots",
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "perplexity-user"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "petalbot",
   "name": "PetalBot",
   "operator": "Huawei",
   "operator_url": "https://huawei.com/",
   "category": "search",
   "user_agent": "PetalBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": "https://webmaster.petalsearch.com/site/petalbot",
   "verification_detail": "Huawei documents the crawler in its webmaster help. Traffic originates from Huawei Cloud ranges but no per-bot list is published.",
   "docs": null,
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "phindbot",
   "name": "PhindBot",
   "operator": "phind",
   "operator_url": "https://www.phind.com/",
   "category": "assistant",
   "user_agent": "PhindBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "poggio-citations",
   "name": "Poggio-Citations",
   "operator": null,
   "operator_url": null,
   "category": "assistant",
   "user_agent": "Poggio-Citations",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "poseidon-research-crawler",
   "name": "Poseidon Research Crawler",
   "operator": "Poseidon Research",
   "operator_url": "https://www.poseidonresearch.com",
   "category": "other",
   "user_agent": "Poseidon Research Crawler",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "buddybot",
    "echoboxbot",
    "facebookexternalhit",
    "google-firebase",
    "openai-bot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "qualifiedbot",
   "name": "QualifiedBot",
   "operator": "Qualified",
   "operator_url": "https://www.qualified.com",
   "category": "assistant",
   "user_agent": "QualifiedBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "queritbot",
   "name": "QueritBot",
   "operator": "Querit",
   "operator_url": null,
   "category": "dataset",
   "user_agent": "QueritBot",
   "user_agent_exact": false,
   "aliases": [
    "Querit-SearchBot"
   ],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "quillbot",
   "name": "QuillBot",
   "operator": "Quillbot",
   "operator_url": "https://quillbot.com",
   "category": "other",
   "user_agent": "QuillBot",
   "user_agent_exact": false,
   "aliases": [
    "quillbot.com"
   ],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "buddybot",
    "echoboxbot",
    "facebookexternalhit",
    "google-firebase",
    "openai-bot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "reflectionbot",
   "name": "Reflectionbot",
   "operator": "Reflection",
   "operator_url": "https://reflection.ai/",
   "category": "agent",
   "user_agent": "Reflectionbot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amazonbuyforme",
    "chatgpt-agent",
    "claude-web",
    "google-cloudvertexbot",
    "openai-operator"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "sbintuitionsbot",
   "name": "SBIntuitionsBot",
   "operator": "SB Intuitions",
   "operator_url": "https://www.sbintuitions.co.jp/en/",
   "category": "other",
   "user_agent": "SBIntuitionsBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://www.sbintuitions.co.jp/en/bot/",
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "buddybot",
    "echoboxbot",
    "facebookexternalhit",
    "google-firebase",
    "openai-bot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "scrapy",
   "name": "Scrapy",
   "operator": "Zyte",
   "operator_url": "https://www.zyte.com",
   "category": "training",
   "user_agent": "Scrapy",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": "https://www.zyte.com",
   "verification_detail": "Scrapy is an open source framework, not one operator. The token means somebody ran the default settings; anybody can change it, so seeing it says more about the operator's carelessness than their identity.",
   "docs": null,
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "facebookbot",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "semrushbot-ocob",
   "name": "SemrushBot-OCOB",
   "operator": "Semrush",
   "operator_url": "https://www.semrush.com/",
   "category": "dataset",
   "user_agent": "SemrushBot-OCOB",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://www.semrush.com/bot/",
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "semrushbot-swa",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "semrushbot-swa",
   "name": "SemrushBot-SWA",
   "operator": "Semrush",
   "operator_url": "https://www.semrush.com/",
   "category": "assistant",
   "user_agent": "SemrushBot-SWA",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://www.semrush.com/bot/",
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "semrushbot-ocob"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "shap-user",
   "name": "Shap-User",
   "operator": "Parallel",
   "operator_url": "https://parallel.ai",
   "category": "assistant",
   "user_agent": "Shap-User",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "shapbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "shapbot",
   "name": "ShapBot",
   "operator": "Parallel",
   "operator_url": "https://parallel.ai",
   "category": "dataset",
   "user_agent": "ShapBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://docs.parallel.ai/features/crawler",
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "shap-user",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "sidetrade-indexer-bot",
   "name": "Sidetrade indexer bot",
   "operator": "Sidetrade",
   "operator_url": "https://www.sidetrade.com",
   "category": "training",
   "user_agent": "Sidetrade indexer bot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "claudebot",
    "facebookbot",
    "google-extended"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "spider-user-agent",
   "name": "Spider",
   "operator": null,
   "operator_url": null,
   "category": "dataset",
   "user_agent": "Spider",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "tavilybot",
   "name": "TavilyBot",
   "operator": "Tavily",
   "operator_url": "https://tavily.com",
   "category": "dataset",
   "user_agent": "TavilyBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "terracotta",
   "name": "TerraCotta",
   "operator": "Ceramic AI",
   "operator_url": "https://ceramic.ai/",
   "category": "dataset",
   "user_agent": "TerraCotta",
   "user_agent_exact": false,
   "aliases": [
    "Terra Cotta"
   ],
   "respects_robots": "yes",
   "respects_robots_source": "https://github.com/CeramicTeam/CeramicTerracotta",
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "thinkbot",
   "name": "Thinkbot",
   "operator": "Thinkbot",
   "operator_url": "https://www.thinkbot.agency",
   "category": "other",
   "user_agent": "Thinkbot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "no",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "buddybot",
    "echoboxbot",
    "facebookexternalhit",
    "google-firebase",
    "openai-bot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "tiktokspider",
   "name": "TikTokSpider",
   "operator": "ByteDance",
   "operator_url": null,
   "category": "training",
   "user_agent": "TikTokSpider",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "anthropic-ai",
    "applebot-extended",
    "bytespider",
    "claudebot",
    "trae-agent"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "timpibot",
   "name": "Timpibot",
   "operator": "Timpi",
   "operator_url": "https://timpi.io",
   "category": "search",
   "user_agent": "Timpibot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "Timpi builds a decentralised index that is partly crawled by volunteer nodes, so requests can come from residential addresses. No published range list exists to check against.",
   "docs": null,
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "tongyibot",
   "name": "TongyiBot",
   "operator": "Alibaba",
   "operator_url": "https://tongyi.aliyun.com",
   "category": "assistant",
   "user_agent": "TongyiBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "trae-agent",
   "name": "Trae",
   "operator": "ByteDance",
   "operator_url": "https://www.trae.ai",
   "category": "coding",
   "user_agent": "Trae",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "bytespider",
    "claude-code",
    "cursor-agent",
    "github-copilot-code",
    "tiktokspider"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "twinagent",
   "name": "TwinAgent",
   "operator": null,
   "operator_url": null,
   "category": "agent",
   "user_agent": "TwinAgent",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amazonbuyforme",
    "chatgpt-agent",
    "claude-web",
    "google-cloudvertexbot",
    "openai-operator"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "useai",
   "name": "UseAI",
   "operator": null,
   "operator_url": null,
   "category": "assistant",
   "user_agent": "UseAI",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "velenpublicwebcrawler",
   "name": "VelenPublicWebCrawler",
   "operator": "Velen Crawler",
   "operator_url": "https://velen.io",
   "category": "dataset",
   "user_agent": "VelenPublicWebCrawler",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://velen.io",
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "wardbot",
   "name": "WARDBot",
   "operator": "WEBSPARK",
   "operator_url": null,
   "category": "dataset",
   "user_agent": "WARDBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "webzio-extended",
   "name": "Webzio-Extended",
   "operator": null,
   "operator_url": null,
   "category": "dataset",
   "user_agent": "Webzio-Extended",
   "user_agent_exact": false,
   "aliases": [
    "webzio-extended"
   ],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "yandexadditional"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "wpbot",
   "name": "wpbot",
   "operator": "QuantumCloud",
   "operator_url": "https://www.quantumcloud.com",
   "category": "other",
   "user_agent": "wpbot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "buddybot",
    "echoboxbot",
    "facebookexternalhit",
    "google-firebase",
    "openai-bot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "wrtnbot",
   "name": "WRTNBot",
   "operator": "Wrtn Technologies",
   "operator_url": "https://wrtn.ai",
   "category": "agent",
   "user_agent": "WRTNBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amazonbuyforme",
    "chatgpt-agent",
    "claude-web",
    "google-cloudvertexbot",
    "openai-operator"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "yak-crawler",
   "name": "YaK",
   "operator": "Meltwater",
   "operator_url": "https://www.meltwater.com/en/suite/consumer-intelligence",
   "category": "other",
   "user_agent": "YaK",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "buddybot",
    "echoboxbot",
    "facebookexternalhit",
    "google-firebase",
    "openai-bot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "yandexadditional",
   "name": "YandexAdditional",
   "operator": "Yandex",
   "operator_url": "https://yandex.ru",
   "category": "dataset",
   "user_agent": "YandexAdditional",
   "user_agent_exact": false,
   "aliases": [
    "YandexAdditionalBot"
   ],
   "respects_robots": "yes",
   "respects_robots_source": "https://yandex.ru/support/webmaster/en/search-appearance/fast.html?lang=en",
   "verification_method": "reverse-dns",
   "verification_source": "https://yandex.com/support/webmaster/robot-workings/check-yandex-robots.html",
   "verification_detail": "Yandex documents a reverse-then-forward DNS check against yandex.ru, yandex.net and yandex.com hostnames. The same check covers the YandexGPT crawlers.",
   "docs": null,
   "related": [
    "agenttimes",
    "aihitbot",
    "apifybot",
    "apifywebsitecontentcrawler",
    "awario"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "yiyanbot",
   "name": "YiyanBot",
   "operator": "Baidu that fetches web content for the yiyan",
   "operator_url": null,
   "category": "assistant",
   "user_agent": "YiyanBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "amzn-user",
    "chatgpt-user",
    "claude-user",
    "duckassistbot",
    "google-notebooklm"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "youbot",
   "name": "YouBot",
   "operator": "You",
   "operator_url": "https://about.you.com/youchat/",
   "category": "search",
   "user_agent": "YouBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "yes",
   "respects_robots_source": "https://about.you.com/youbot/",
   "verification_method": "none",
   "verification_source": "https://about.you.com/youbot/",
   "verification_detail": "You.com documents the crawler and honours robots.txt but publishes no address list.",
   "docs": "https://about.you.com/youbot/",
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  },
  {
   "slug": "zanistabot",
   "name": "ZanistaBot",
   "operator": null,
   "operator_url": null,
   "category": "search",
   "user_agent": "ZanistaBot",
   "user_agent_exact": false,
   "aliases": [],
   "respects_robots": "unclear",
   "respects_robots_source": null,
   "verification_method": "none",
   "verification_source": null,
   "verification_detail": "No IP range file and no reverse-DNS convention have been published for this agent. The user-agent string is the only signal, and any client can send it, so treat a match as a claim rather than proof.",
   "docs": null,
   "related": [
    "aiwebindex",
    "amazonbot",
    "amzn-searchbot",
    "applebot",
    "bingbot"
   ],
   "source_last_updated": "2026-08-24"
  }
 ]
}
