{"name":"WebDecoy crawler verification directory","description":"Every request in WebDecoy's production detection corpus whose User-Agent declares one of these crawlers, checked IP-by-IP against the operator's own published IP range list. A request is counted as verified when its source IP falls inside a published prefix, and forged when it does not. Operators that publish no machine-readable range list cannot be checked this way and are reported without a forgery figure.","revision":"2026-08-19","temporalCoverage":"2025-11-14/2026-08-19","rangesCheckedOn":"2026-08-19","license":{"name":"Creative Commons Attribution 4.0 International","url":"https://creativecommons.org/licenses/by/4.0/","attribution":"WebDecoy (https://webdecoy.com/bots/)"},"corpus":{"totalDetections":81034,"distinctIps":17187,"since":"2025-11-14","through":"2026-08-19"},"methodologyUrl":"https://webdecoy.com/bots/","bots":[{"slug":"chatgpt-user","name":"ChatGPT-User","organization":"OpenAI","category":"User-triggered fetcher","purpose":"Fetches a page in real time because a ChatGPT user asked a question that requires it. Not a training crawler.","userAgent":"ChatGPT-User/1.0; +https://openai.com/bot","robotsName":"ChatGPT-User","docsUrl":"https://platform.openai.com/docs/bots","rangeListUrl":"https://openai.com/chatgpt-user.json","requests":3693,"distinctIps":1511,"decoyHits":16,"firstSeen":"2026-07-26","lastSeen":"2026-08-19","verifiedRequests":1735,"forgedRequests":1958,"forgedPct":53,"verifiedIps":885,"forgedIps":626,"hasPublishedRanges":true},{"slug":"amazonbot","name":"Amazonbot","organization":"Amazon","category":"Search / assistant crawler","purpose":"Crawls pages to support Alexa and Amazon search features.","userAgent":"Amazonbot/0.1; +https://developer.amazon.com/support/amazonbot","robotsName":"Amazonbot","docsUrl":"https://developer.amazon.com/amazonbot","rangeListUrl":null,"requests":1692,"distinctIps":366,"decoyHits":13,"firstSeen":"2026-06-26","lastSeen":"2026-08-19","verifiedRequests":null,"forgedRequests":null,"forgedPct":null,"verifiedIps":null,"forgedIps":null,"hasPublishedRanges":false},{"slug":"meta-externalagent","name":"Meta-ExternalAgent","organization":"Meta","category":"AI training crawler","purpose":"Collects public web data used to train Meta AI models.","userAgent":"meta-externalagent/1.1","robotsName":"meta-externalagent","docsUrl":"https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/","rangeListUrl":null,"requests":1667,"distinctIps":214,"decoyHits":3,"firstSeen":"2026-02-05","lastSeen":"2026-08-19","verifiedRequests":null,"forgedRequests":null,"forgedPct":null,"verifiedIps":null,"forgedIps":null,"hasPublishedRanges":false},{"slug":"claudebot","name":"ClaudeBot","organization":"Anthropic","category":"AI training crawler","purpose":"Collects public web content used to train Claude models.","userAgent":"ClaudeBot/1.0; +claudebot@anthropic.com","robotsName":"ClaudeBot","docsUrl":"https://support.anthropic.com/en/articles/8896518","rangeListUrl":null,"requests":1401,"distinctIps":68,"decoyHits":13,"firstSeen":"2025-12-31","lastSeen":"2026-08-18","verifiedRequests":null,"forgedRequests":null,"forgedPct":null,"verifiedIps":null,"forgedIps":null,"hasPublishedRanges":false},{"slug":"googlebot","name":"Googlebot","organization":"Google","category":"Search crawler","purpose":"Google Search indexing. The single most impersonated crawler on the web.","userAgent":"Googlebot/2.1; +http://www.google.com/bot.html","robotsName":"Googlebot","docsUrl":"https://developers.google.com/search/docs/crawling-indexing/googlebot","rangeListUrl":"https://developers.google.com/search/apis/ipranges/googlebot.json","requests":1194,"distinctIps":136,"decoyHits":91,"firstSeen":"2026-01-11","lastSeen":"2026-08-19","verifiedRequests":639,"forgedRequests":555,"forgedPct":46.5,"verifiedIps":24,"forgedIps":112,"hasPublishedRanges":true},{"slug":"bytespider","name":"Bytespider","organization":"ByteDance","category":"AI training crawler","purpose":"ByteDance crawler collecting training data. Widely reported to ignore robots.txt.","userAgent":"Bytespider","robotsName":"Bytespider","docsUrl":"https://bytedance.com","rangeListUrl":null,"requests":1001,"distinctIps":538,"decoyHits":0,"firstSeen":"2026-07-26","lastSeen":"2026-08-19","verifiedRequests":null,"forgedRequests":null,"forgedPct":null,"verifiedIps":null,"forgedIps":null,"hasPublishedRanges":false},{"slug":"bingbot","name":"bingbot","organization":"Microsoft","category":"Search crawler","purpose":"Bing Search indexing, and the retrieval layer behind Copilot.","userAgent":"bingbot/2.0; +http://www.bing.com/bingbot.htm","robotsName":"bingbot","docsUrl":"https://www.bing.com/webmasters/help/bingbot","rangeListUrl":"https://www.bing.com/toolbox/bingbot.json","requests":878,"distinctIps":447,"decoyHits":2,"firstSeen":"2025-12-16","lastSeen":"2026-08-19","verifiedRequests":783,"forgedRequests":95,"forgedPct":10.8,"verifiedIps":425,"forgedIps":22,"hasPublishedRanges":true},{"slug":"perplexitybot","name":"PerplexityBot","organization":"Perplexity AI","category":"AI search crawler","purpose":"Indexes pages so Perplexity can cite them in answers.","userAgent":"PerplexityBot/1.0; +https://perplexity.ai/perplexitybot","robotsName":"PerplexityBot","docsUrl":"https://docs.perplexity.ai/guides/bots","rangeListUrl":"https://www.perplexity.com/perplexitybot.json","requests":738,"distinctIps":26,"decoyHits":3,"firstSeen":"2026-06-26","lastSeen":"2026-08-19","verifiedRequests":469,"forgedRequests":269,"forgedPct":36.4,"verifiedIps":7,"forgedIps":19,"hasPublishedRanges":true},{"slug":"gptbot","name":"GPTBot","organization":"OpenAI","category":"AI training crawler","purpose":"Collects public web content used to train OpenAI models.","userAgent":"GPTBot/1.2; +https://openai.com/gptbot","robotsName":"GPTBot","docsUrl":"https://platform.openai.com/docs/gptbot","rangeListUrl":"https://openai.com/gptbot.json","requests":619,"distinctIps":153,"decoyHits":238,"firstSeen":"2025-11-18","lastSeen":"2026-08-18","verifiedRequests":279,"forgedRequests":340,"forgedPct":54.9,"verifiedIps":112,"forgedIps":41,"hasPublishedRanges":true},{"slug":"oai-searchbot","name":"OAI-SearchBot","organization":"OpenAI","category":"AI search crawler","purpose":"Builds the index behind ChatGPT search results and citations.","userAgent":"OAI-SearchBot/1.0; +https://openai.com/searchbot","robotsName":"OAI-SearchBot","docsUrl":"https://platform.openai.com/docs/bots","rangeListUrl":"https://openai.com/searchbot.json","requests":495,"distinctIps":90,"decoyHits":7,"firstSeen":"2026-07-26","lastSeen":"2026-08-19","verifiedRequests":227,"forgedRequests":268,"forgedPct":54.1,"verifiedIps":67,"forgedIps":23,"hasPublishedRanges":true}]}