Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
23 changes: 22 additions & 1 deletion packages/webdecoy/src/bots/bots.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,27 @@ describe('matchUserAgent', () => {
expect(claude?.category).toBe('training_crawler');
});

it('keeps AI search and user-triggered fetchers out of training', () => {
// Every Anthropic UA carries an @anthropic.com contact, so a bare
// "anthropic" training pattern once caught all of them.
const user = matchUserAgent(
'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Claude-User/1.0; +Claude-User@anthropic.com)',
);
expect(user?.id).toBe('claude-user');
expect(user?.category).toBe('ai_agent');

const search = matchUserAgent(
'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)',
);
expect(search?.category).toBe('ai_search_crawler');

const mistral = matchUserAgent(
'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; MistralAI-User/1.0; +https://docs.mistral.ai/robots)',
);
expect(mistral?.id).toBe('mistralai-user');
expect(mistral?.category).toBe('ai_assistant');
});

it('is case-insensitive, because User-Agent casing is not a contract', () => {
expect(matchUserAgent('GPTBOT/1.1')?.id).toBe('gptbot');
expect(matchUserAgent('gptbot/1.1')?.id).toBe('gptbot');
Expand Down Expand Up @@ -60,7 +81,7 @@ describe('matchUserAgent', () => {

describe('registry integrity', () => {
it('carries the full generated table', () => {
expect(BOT_REGISTRY.length).toBe(182);
expect(BOT_REGISTRY.length).toBe(186);
expect(BOT_CATEGORIES).toContain('training_crawler');
expect(BOT_CATEGORIES).toContain('search_crawler');
});
Expand Down
60 changes: 48 additions & 12 deletions packages/webdecoy/src/bots/parity-vectors.generated.json
Original file line number Diff line number Diff line change
Expand Up @@ -84,13 +84,13 @@
"matched": true
},
{
"userAgent": "anthropic",
"userAgent": "anthropic-ai",
"id": "anthropic",
"category": "training_crawler",
"matched": true
},
{
"userAgent": "Mozilla/5.0 (compatible; anthropic/1.0; +http://example.com/bot)",
"userAgent": "Mozilla/5.0 (compatible; anthropic-ai/1.0; +http://example.com/bot)",
"id": "anthropic",
"category": "training_crawler",
"matched": true
Expand Down Expand Up @@ -372,15 +372,15 @@
"matched": true
},
{
"userAgent": "chatgpt",
"id": "chatgpt-user",
"category": "ai_assistant",
"userAgent": "claude-searchbot",
"id": "claude-searchbot",
"category": "ai_search_crawler",
"matched": true
},
{
"userAgent": "Mozilla/5.0 (compatible; chatgpt/1.0; +http://example.com/bot)",
"id": "chatgpt-user",
"category": "ai_assistant",
"userAgent": "Mozilla/5.0 (compatible; claude-searchbot/1.0; +http://example.com/bot)",
"id": "claude-searchbot",
"category": "ai_search_crawler",
"matched": true
},
{
Expand Down Expand Up @@ -1223,6 +1223,18 @@
"category": "training_crawler",
"matched": true
},
{
"userAgent": "meta-externalfetcher",
"id": "meta-externalfetcher",
"category": "ai_assistant",
"matched": true
},
{
"userAgent": "Mozilla/5.0 (compatible; meta-externalfetcher/1.0; +http://example.com/bot)",
"id": "meta-externalfetcher",
"category": "ai_assistant",
"matched": true
},
{
"userAgent": "metaphor",
"id": "metaphor",
Expand Down Expand Up @@ -1260,13 +1272,25 @@
"matched": true
},
{
"userAgent": "mistral",
"userAgent": "mistralai-user",
"id": "mistralai-user",
"category": "ai_assistant",
"matched": true
},
{
"userAgent": "Mozilla/5.0 (compatible; mistralai-user/1.0; +http://example.com/bot)",
"id": "mistralai-user",
"category": "ai_assistant",
"matched": true
},
{
"userAgent": "mistralbot",
"id": "mistral",
"category": "training_crawler",
"matched": true
},
{
"userAgent": "Mozilla/5.0 (compatible; mistral/1.0; +http://example.com/bot)",
"userAgent": "Mozilla/5.0 (compatible; mistralbot/1.0; +http://example.com/bot)",
"id": "mistral",
"category": "training_crawler",
"matched": true
Expand Down Expand Up @@ -1523,16 +1547,28 @@
"category": "generic_scraper",
"matched": true
},
{
"userAgent": "perplexity-user",
"id": "perplexity-user",
"category": "ai_assistant",
"matched": true
},
{
"userAgent": "Mozilla/5.0 (compatible; perplexity-user/1.0; +http://example.com/bot)",
"id": "perplexity-user",
"category": "ai_assistant",
"matched": true
},
{
"userAgent": "perplexitybot",
"id": "perplexitybot",
"category": "training_crawler",
"category": "ai_search_crawler",
"matched": true
},
{
"userAgent": "Mozilla/5.0 (compatible; perplexitybot/1.0; +http://example.com/bot)",
"id": "perplexitybot",
"category": "training_crawler",
"category": "ai_search_crawler",
"matched": true
},
{
Expand Down
14 changes: 9 additions & 5 deletions packages/webdecoy/src/bots/registry.generated.ts
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,7 @@
* agents share User-Agent substrings, so sorting this array changes how real
* traffic is classified.
*
* 182 agents across 15 categories.
* 186 agents across 15 categories.
*/

/**
Expand Down Expand Up @@ -78,7 +78,7 @@ export const BOT_REGISTRY: readonly BotAgent[] = [
{ id: "reflectionbot", name: "Reflectionbot", category: "training_crawler", organization: "Reflection AI", baseScore: 70, respectsRobots: true, uaPatterns: ["reflectionbot"] },
{ id: "gptbot", name: "GPTBot", category: "training_crawler", organization: "OpenAI", baseScore: 85, respectsRobots: true, uaPatterns: ["gptbot"] },
{ id: "claudebot", name: "ClaudeBot", category: "training_crawler", organization: "Anthropic", baseScore: 85, respectsRobots: true, uaPatterns: ["claudebot"] },
{ id: "anthropic", name: "Anthropic", category: "training_crawler", organization: "Anthropic", baseScore: 85, respectsRobots: true, uaPatterns: ["anthropic"] },
{ id: "anthropic", name: "Anthropic", category: "training_crawler", organization: "Anthropic", baseScore: 85, respectsRobots: true, uaPatterns: ["anthropic-ai"] },
{ id: "claude-web", name: "Claude-Web", category: "training_crawler", organization: "Anthropic", baseScore: 85, respectsRobots: true, uaPatterns: ["claude-web"] },
{ id: "ccbot", name: "CCBot", category: "training_crawler", organization: "Common Crawl", baseScore: 80, respectsRobots: true, uaPatterns: ["ccbot"] },
{ id: "google-extended", name: "Google-Extended", category: "training_crawler", organization: "Google", baseScore: 80, respectsRobots: true, uaPatterns: ["google-extended"] },
Expand All @@ -87,10 +87,9 @@ export const BOT_REGISTRY: readonly BotAgent[] = [
{ id: "facebookbot", name: "FacebookBot", category: "training_crawler", organization: "Meta", baseScore: 70, respectsRobots: true, uaPatterns: ["facebookbot"] },
{ id: "meta-externalagent", name: "Meta-ExternalAgent", category: "training_crawler", organization: "Meta", baseScore: 75, respectsRobots: true, uaPatterns: ["meta-externalagent"] },
{ id: "cohere-ai", name: "Cohere", category: "training_crawler", organization: "Cohere", baseScore: 80, respectsRobots: true, uaPatterns: ["cohere-ai", "cohere"] },
{ id: "perplexitybot", name: "PerplexityBot", category: "training_crawler", organization: "Perplexity AI", baseScore: 80, respectsRobots: true, uaPatterns: ["perplexitybot"] },
{ id: "applebot-extended", name: "Applebot-Extended", category: "training_crawler", organization: "Apple", baseScore: 75, respectsRobots: true, uaPatterns: ["applebot-extended"] },
{ id: "youbot", name: "YouBot", category: "training_crawler", organization: "You.com", baseScore: 75, respectsRobots: true, uaPatterns: ["youbot"] },
{ id: "mistral", name: "MistralBot", category: "training_crawler", organization: "Mistral AI", baseScore: 80, respectsRobots: true, uaPatterns: ["mistral"] },
{ id: "mistral", name: "MistralBot", category: "training_crawler", organization: "Mistral AI", baseScore: 80, respectsRobots: true, uaPatterns: ["mistralbot"] },
{ id: "gemini", name: "Gemini", category: "training_crawler", organization: "Google", baseScore: 80, respectsRobots: true, uaPatterns: ["gemini"] },
{ id: "ai2bot", name: "AI2Bot", category: "training_crawler", organization: "Allen Institute for AI", baseScore: 75, respectsRobots: true, uaPatterns: ["ai2bot"] },
{ id: "deepseek", name: "DeepSeek", category: "training_crawler", organization: "DeepSeek", baseScore: 80, respectsRobots: true, uaPatterns: ["deepseek"] },
Expand All @@ -107,6 +106,8 @@ export const BOT_REGISTRY: readonly BotAgent[] = [
{ id: "sofyabot", name: "SofyaBot", category: "ai_search_crawler", organization: "Sofya", baseScore: 65, respectsRobots: true, uaPatterns: ["sofyabot"] },
{ id: "xai-searchbot", name: "xAI-SearchBot", category: "ai_search_crawler", organization: "xAI", baseScore: 70, respectsRobots: true, uaPatterns: ["xai-searchbot"] },
{ id: "linkupbot", name: "LinkupBot", category: "ai_search_crawler", organization: "Linkup", baseScore: 65, respectsRobots: true, uaPatterns: ["linkupbot"] },
{ id: "perplexitybot", name: "PerplexityBot", category: "ai_search_crawler", organization: "Perplexity AI", baseScore: 80, respectsRobots: true, uaPatterns: ["perplexitybot"] },
{ id: "claude-searchbot", name: "Claude-SearchBot", category: "ai_search_crawler", organization: "Anthropic", baseScore: 75, respectsRobots: true, uaPatterns: ["claude-searchbot"] },
{ id: "oai-searchbot", name: "OAI-SearchBot", category: "ai_search_crawler", organization: "OpenAI", baseScore: 80, respectsRobots: true, uaPatterns: ["oai-searchbot"] },
{ id: "searchgpt", name: "SearchGPT", category: "ai_search_crawler", organization: "OpenAI", baseScore: 75, respectsRobots: true, uaPatterns: ["searchgpt"] },
{ id: "phind", name: "Phind", category: "ai_search_crawler", organization: "Phind", baseScore: 70, respectsRobots: true, uaPatterns: ["phind"] },
Expand Down Expand Up @@ -252,7 +253,10 @@ export const BOT_REGISTRY: readonly BotAgent[] = [
{ id: "apify", name: "Apify", category: "commercial_scraper", organization: "Apify", baseScore: 65, respectsRobots: true, uaPatterns: ["apify"] },
{ id: "crawlbase", name: "Crawlbase", category: "commercial_scraper", organization: "Crawlbase", baseScore: 65, respectsRobots: true, uaPatterns: ["crawlbase"] },
{ id: "webscrapingapi", name: "WebScrapingAPI", category: "commercial_scraper", organization: "WebScrapingAPI", baseScore: 65, respectsRobots: true, uaPatterns: ["webscrapingapi"] },
{ id: "chatgpt-user", name: "ChatGPT-User", category: "ai_assistant", organization: "OpenAI", baseScore: 85, respectsRobots: true, uaPatterns: ["chatgpt-user", "chatgpt"] },
{ id: "chatgpt-user", name: "ChatGPT-User", category: "ai_assistant", organization: "OpenAI", baseScore: 85, respectsRobots: true, uaPatterns: ["chatgpt-user"] },
{ id: "perplexity-user", name: "Perplexity-User", category: "ai_assistant", organization: "Perplexity AI", baseScore: 75, respectsRobots: false, uaPatterns: ["perplexity-user"] },
{ id: "mistralai-user", name: "MistralAI-User", category: "ai_assistant", organization: "Mistral AI", baseScore: 75, respectsRobots: true, uaPatterns: ["mistralai-user"] },
{ id: "meta-externalfetcher", name: "Meta-ExternalFetcher", category: "ai_assistant", organization: "Meta", baseScore: 70, respectsRobots: false, uaPatterns: ["meta-externalfetcher"] },
{ id: "gemini-deep-research", name: "Gemini-Deep-Research", category: "ai_assistant", organization: "Google", baseScore: 65, respectsRobots: true, uaPatterns: ["gemini-deep-research"] },
{ id: "copilot", name: "Microsoft Copilot", category: "ai_assistant", organization: "Microsoft", baseScore: 65, respectsRobots: true, uaPatterns: ["copilot"] },
{ id: "cortana", name: "Cortana", category: "ai_assistant", organization: "Microsoft", baseScore: 60, respectsRobots: true, uaPatterns: ["cortana"] },
Expand Down
Loading