From 53617adbf943a25366ba56b4d9774e6caae78d85 Mon Sep 17 00:00:00 2001 From: Chris Portscheller Date: Tue, 22 Sep 2026 21:57:09 -0500 Subject: [PATCH 1/2] chore(bots): regenerate registry with seven plugin-only agents (182) Google-InspectionTool, MSNBot, Claude-Web, Facebot, GTmetrix, Lighthouse and Feedfetcher-Google, added to pkg/agents so the WordPress plugin's own list can retire in favour of the generated one (WebDecoy/app#1275). --- packages/webdecoy/src/bots/bots.test.ts | 2 +- .../src/bots/parity-vectors.generated.json | 96 +++++++++++++++++++ .../webdecoy/src/bots/registry.generated.ts | 9 +- 3 files changed, 105 insertions(+), 2 deletions(-) diff --git a/packages/webdecoy/src/bots/bots.test.ts b/packages/webdecoy/src/bots/bots.test.ts index 7994120..210a149 100644 --- a/packages/webdecoy/src/bots/bots.test.ts +++ b/packages/webdecoy/src/bots/bots.test.ts @@ -60,7 +60,7 @@ describe('matchUserAgent', () => { describe('registry integrity', () => { it('carries the full generated table', () => { - expect(BOT_REGISTRY.length).toBe(175); + expect(BOT_REGISTRY.length).toBe(182); expect(BOT_CATEGORIES).toContain('training_crawler'); expect(BOT_CATEGORIES).toContain('search_crawler'); }); diff --git a/packages/webdecoy/src/bots/parity-vectors.generated.json b/packages/webdecoy/src/bots/parity-vectors.generated.json index 77ad224..af72f27 100644 --- a/packages/webdecoy/src/bots/parity-vectors.generated.json +++ b/packages/webdecoy/src/bots/parity-vectors.generated.json @@ -395,6 +395,18 @@ "category": "ai_agent", "matched": true }, + { + "userAgent": "claude-web", + "id": "claude-web", + "category": "training_crawler", + "matched": true + }, + { + "userAgent": "Mozilla/5.0 (compatible; claude-web/1.0; +http://example.com/bot)", + "id": "claude-web", + "category": "training_crawler", + "matched": true + }, { "userAgent": "claudebot", "id": "claudebot", @@ -671,6 +683,18 @@ "category": "training_crawler", "matched": true }, + { + "userAgent": "facebot", + "id": "facebot", + "category": "fetcher", + "matched": true + }, + { + "userAgent": "Mozilla/5.0 (compatible; facebot/1.0; +http://example.com/bot)", + "id": "facebot", + "category": "fetcher", + "matched": true + }, { "userAgent": "faraday", "id": "faraday", @@ -695,6 +719,18 @@ "category": "feed_reader", "matched": true }, + { + "userAgent": "feedfetcher", + "id": "feedfetcher-google", + "category": "feed_reader", + "matched": true + }, + { + "userAgent": "Mozilla/5.0 (compatible; feedfetcher/1.0; +http://example.com/bot)", + "id": "feedfetcher-google", + "category": "feed_reader", + "matched": true + }, { "userAgent": "feedly", "id": "feedly", @@ -755,6 +791,18 @@ "category": "training_crawler", "matched": true }, + { + "userAgent": "gtmetrix", + "id": "gtmetrix", + "category": "monitoring", + "matched": true + }, + { + "userAgent": "Mozilla/5.0 (compatible; gtmetrix/1.0; +http://example.com/bot)", + "id": "gtmetrix", + "category": "monitoring", + "matched": true + }, { "userAgent": "gemini", "id": "gemini", @@ -815,6 +863,18 @@ "category": "training_crawler", "matched": true }, + { + "userAgent": "google-inspectiontool", + "id": "google-inspectiontool", + "category": "search_crawler", + "matched": true + }, + { + "userAgent": "Mozilla/5.0 (compatible; google-inspectiontool/1.0; +http://example.com/bot)", + "id": "google-inspectiontool", + "category": "search_crawler", + "matched": true + }, { "userAgent": "googleagent-mariner", "id": "googleagent-mariner", @@ -1067,6 +1127,30 @@ "category": "fetcher", "matched": true }, + { + "userAgent": "chrome-lighthouse", + "id": "chrome-lighthouse", + "category": "monitoring", + "matched": true + }, + { + "userAgent": "Mozilla/5.0 (compatible; chrome-lighthouse/1.0; +http://example.com/bot)", + "id": "chrome-lighthouse", + "category": "monitoring", + "matched": true + }, + { + "userAgent": "google page speed", + "id": "chrome-lighthouse", + "category": "monitoring", + "matched": true + }, + { + "userAgent": "Mozilla/5.0 (compatible; google page speed/1.0; +http://example.com/bot)", + "id": "chrome-lighthouse", + "category": "monitoring", + "matched": true + }, { "userAgent": "linkedinbot", "id": "linkedinbot", @@ -1103,6 +1187,18 @@ "category": "seo_crawler", "matched": true }, + { + "userAgent": "msnbot", + "id": "msnbot", + "category": "search_crawler", + "matched": true + }, + { + "userAgent": "Mozilla/5.0 (compatible; msnbot/1.0; +http://example.com/bot)", + "id": "msnbot", + "category": "search_crawler", + "matched": true + }, { "userAgent": "mechanize", "id": "mechanize", diff --git a/packages/webdecoy/src/bots/registry.generated.ts b/packages/webdecoy/src/bots/registry.generated.ts index 1134bdb..d9c8e64 100644 --- a/packages/webdecoy/src/bots/registry.generated.ts +++ b/packages/webdecoy/src/bots/registry.generated.ts @@ -11,7 +11,7 @@ * agents share User-Agent substrings, so sorting this array changes how real * traffic is classified. * - * 175 agents across 15 categories. + * 182 agents across 15 categories. */ /** @@ -81,6 +81,7 @@ export const BOT_REGISTRY: readonly BotAgent[] = [ { id: "oai-searchbot", name: "OAI-SearchBot", category: "training_crawler", organization: "OpenAI", baseScore: 80, respectsRobots: true, uaPatterns: ["oai-searchbot"] }, { id: "claudebot", name: "ClaudeBot", category: "training_crawler", organization: "Anthropic", baseScore: 85, respectsRobots: true, uaPatterns: ["claudebot"] }, { id: "anthropic", name: "Anthropic", category: "training_crawler", organization: "Anthropic", baseScore: 85, respectsRobots: true, uaPatterns: ["anthropic"] }, + { id: "claude-web", name: "Claude-Web", category: "training_crawler", organization: "Anthropic", baseScore: 85, respectsRobots: true, uaPatterns: ["claude-web"] }, { id: "ccbot", name: "CCBot", category: "training_crawler", organization: "Common Crawl", baseScore: 80, respectsRobots: true, uaPatterns: ["ccbot"] }, { id: "google-extended", name: "Google-Extended", category: "training_crawler", organization: "Google", baseScore: 80, respectsRobots: true, uaPatterns: ["google-extended"] }, { id: "bytespider", name: "ByteSpider", category: "training_crawler", organization: "ByteDance", baseScore: 75, respectsRobots: false, uaPatterns: ["bytespider"] }, @@ -126,7 +127,9 @@ export const BOT_REGISTRY: readonly BotAgent[] = [ { id: "multion", name: "MultiOn", category: "ai_agent", organization: "MultiOn", baseScore: 70, respectsRobots: false, uaPatterns: ["multion"] }, { id: "iboubot", name: "IbouBot", category: "search_crawler", organization: "Ibou", baseScore: 30, respectsRobots: true, uaPatterns: ["iboubot"] }, { id: "googlebot", name: "Googlebot", category: "search_crawler", organization: "Google", baseScore: 30, respectsRobots: true, uaPatterns: ["googlebot"] }, + { id: "google-inspectiontool", name: "Google-InspectionTool", category: "search_crawler", organization: "Google", baseScore: 30, respectsRobots: true, uaPatterns: ["google-inspectiontool"] }, { id: "bingbot", name: "Bingbot", category: "search_crawler", organization: "Microsoft", baseScore: 30, respectsRobots: true, uaPatterns: ["bingbot"] }, + { id: "msnbot", name: "MSNBot", category: "search_crawler", organization: "Microsoft", baseScore: 30, respectsRobots: true, uaPatterns: ["msnbot"] }, { id: "yandexbot", name: "YandexBot", category: "search_crawler", organization: "Yandex", baseScore: 35, respectsRobots: true, uaPatterns: ["yandexbot"] }, { id: "baiduspider", name: "Baiduspider", category: "search_crawler", organization: "Baidu", baseScore: 40, respectsRobots: true, uaPatterns: ["baiduspider"] }, { id: "duckduckbot", name: "DuckDuckBot", category: "search_crawler", organization: "DuckDuckGo", baseScore: 30, respectsRobots: true, uaPatterns: ["duckduckbot"] }, @@ -201,6 +204,7 @@ export const BOT_REGISTRY: readonly BotAgent[] = [ { id: "heritrix", name: "Heritrix", category: "archiver", organization: "Internet Archive", baseScore: 30, respectsRobots: true, uaPatterns: ["heritrix"] }, { id: "brozzler", name: "Brozzler", category: "archiver", organization: "Internet Archive", baseScore: 30, respectsRobots: true, uaPatterns: ["brozzler"] }, { id: "facebookexternalhit", name: "Facebook", category: "fetcher", organization: "Meta", baseScore: 25, respectsRobots: true, uaPatterns: ["facebookexternalhit"] }, + { id: "facebot", name: "Facebot", category: "fetcher", organization: "Meta", baseScore: 25, respectsRobots: true, uaPatterns: ["facebot"] }, { id: "twitterbot", name: "Twitterbot", category: "fetcher", organization: "X Corp", baseScore: 25, respectsRobots: true, uaPatterns: ["twitterbot"] }, { id: "linkedinbot", name: "LinkedInBot", category: "fetcher", organization: "LinkedIn", baseScore: 25, respectsRobots: true, uaPatterns: ["linkedinbot"] }, { id: "slackbot", name: "Slackbot", category: "fetcher", organization: "Slack", baseScore: 25, respectsRobots: true, uaPatterns: ["slackbot"] }, @@ -228,7 +232,10 @@ export const BOT_REGISTRY: readonly BotAgent[] = [ { id: "freshping", name: "Freshping", category: "monitoring", organization: "Freshworks", baseScore: 20, respectsRobots: true, uaPatterns: ["freshping"] }, { id: "hetrixtools", name: "HetrixTools", category: "monitoring", organization: "HetrixTools", baseScore: 20, respectsRobots: true, uaPatterns: ["hetrixtools"] }, { id: "nodeping", name: "NodePing", category: "monitoring", organization: "NodePing", baseScore: 20, respectsRobots: true, uaPatterns: ["nodeping"] }, + { id: "gtmetrix", name: "GTmetrix", category: "monitoring", organization: "GTmetrix", baseScore: 20, respectsRobots: true, uaPatterns: ["gtmetrix"] }, + { id: "chrome-lighthouse", name: "Lighthouse", category: "monitoring", organization: "Google", baseScore: 20, respectsRobots: true, uaPatterns: ["chrome-lighthouse", "google page speed"] }, { id: "feedly", name: "Feedly", category: "feed_reader", organization: "Feedly", baseScore: 25, respectsRobots: true, uaPatterns: ["feedly"] }, + { id: "feedfetcher-google", name: "Feedfetcher-Google", category: "feed_reader", organization: "Google", baseScore: 25, respectsRobots: true, uaPatterns: ["feedfetcher"] }, { id: "newsblur", name: "NewsBlur", category: "feed_reader", organization: "NewsBlur", baseScore: 25, respectsRobots: true, uaPatterns: ["newsblur"] }, { id: "inoreader", name: "Inoreader", category: "feed_reader", organization: "Inoreader", baseScore: 25, respectsRobots: true, uaPatterns: ["inoreader"] }, { id: "theoldreader", name: "The Old Reader", category: "feed_reader", organization: "The Old Reader", baseScore: 25, respectsRobots: true, uaPatterns: ["theoldreader"] }, From 233ebaab5dfdd16034843305ac66c7afaaa40040 Mon Sep 17 00:00:00 2001 From: Chris Portscheller Date: Tue, 22 Sep 2026 22:06:55 -0500 Subject: [PATCH 2/2] chore(bots): ChatGPT-User is ai_assistant, OAI-SearchBot ai_search_crawler --- .../webdecoy/src/bots/parity-vectors.generated.json | 12 ++++++------ packages/webdecoy/src/bots/registry.generated.ts | 4 ++-- 2 files changed, 8 insertions(+), 8 deletions(-) diff --git a/packages/webdecoy/src/bots/parity-vectors.generated.json b/packages/webdecoy/src/bots/parity-vectors.generated.json index af72f27..8332f95 100644 --- a/packages/webdecoy/src/bots/parity-vectors.generated.json +++ b/packages/webdecoy/src/bots/parity-vectors.generated.json @@ -362,25 +362,25 @@ { "userAgent": "chatgpt-user", "id": "chatgpt-user", - "category": "training_crawler", + "category": "ai_assistant", "matched": true }, { "userAgent": "Mozilla/5.0 (compatible; chatgpt-user/1.0; +http://example.com/bot)", "id": "chatgpt-user", - "category": "training_crawler", + "category": "ai_assistant", "matched": true }, { "userAgent": "chatgpt", "id": "chatgpt-user", - "category": "training_crawler", + "category": "ai_assistant", "matched": true }, { "userAgent": "Mozilla/5.0 (compatible; chatgpt/1.0; +http://example.com/bot)", "id": "chatgpt-user", - "category": "training_crawler", + "category": "ai_assistant", "matched": true }, { @@ -1430,13 +1430,13 @@ { "userAgent": "oai-searchbot", "id": "oai-searchbot", - "category": "training_crawler", + "category": "ai_search_crawler", "matched": true }, { "userAgent": "Mozilla/5.0 (compatible; oai-searchbot/1.0; +http://example.com/bot)", "id": "oai-searchbot", - "category": "training_crawler", + "category": "ai_search_crawler", "matched": true }, { diff --git a/packages/webdecoy/src/bots/registry.generated.ts b/packages/webdecoy/src/bots/registry.generated.ts index d9c8e64..19e8417 100644 --- a/packages/webdecoy/src/bots/registry.generated.ts +++ b/packages/webdecoy/src/bots/registry.generated.ts @@ -77,8 +77,6 @@ export const BOT_CATEGORIES: readonly BotCategory[] = [ export const BOT_REGISTRY: readonly BotAgent[] = [ { id: "reflectionbot", name: "Reflectionbot", category: "training_crawler", organization: "Reflection AI", baseScore: 70, respectsRobots: true, uaPatterns: ["reflectionbot"] }, { id: "gptbot", name: "GPTBot", category: "training_crawler", organization: "OpenAI", baseScore: 85, respectsRobots: true, uaPatterns: ["gptbot"] }, - { id: "chatgpt-user", name: "ChatGPT-User", category: "training_crawler", organization: "OpenAI", baseScore: 85, respectsRobots: true, uaPatterns: ["chatgpt-user", "chatgpt"] }, - { id: "oai-searchbot", name: "OAI-SearchBot", category: "training_crawler", organization: "OpenAI", baseScore: 80, respectsRobots: true, uaPatterns: ["oai-searchbot"] }, { id: "claudebot", name: "ClaudeBot", category: "training_crawler", organization: "Anthropic", baseScore: 85, respectsRobots: true, uaPatterns: ["claudebot"] }, { id: "anthropic", name: "Anthropic", category: "training_crawler", organization: "Anthropic", baseScore: 85, respectsRobots: true, uaPatterns: ["anthropic"] }, { id: "claude-web", name: "Claude-Web", category: "training_crawler", organization: "Anthropic", baseScore: 85, respectsRobots: true, uaPatterns: ["claude-web"] }, @@ -109,6 +107,7 @@ export const BOT_REGISTRY: readonly BotAgent[] = [ { id: "sofyabot", name: "SofyaBot", category: "ai_search_crawler", organization: "Sofya", baseScore: 65, respectsRobots: true, uaPatterns: ["sofyabot"] }, { id: "xai-searchbot", name: "xAI-SearchBot", category: "ai_search_crawler", organization: "xAI", baseScore: 70, respectsRobots: true, uaPatterns: ["xai-searchbot"] }, { id: "linkupbot", name: "LinkupBot", category: "ai_search_crawler", organization: "Linkup", baseScore: 65, respectsRobots: true, uaPatterns: ["linkupbot"] }, + { id: "oai-searchbot", name: "OAI-SearchBot", category: "ai_search_crawler", organization: "OpenAI", baseScore: 80, respectsRobots: true, uaPatterns: ["oai-searchbot"] }, { id: "searchgpt", name: "SearchGPT", category: "ai_search_crawler", organization: "OpenAI", baseScore: 75, respectsRobots: true, uaPatterns: ["searchgpt"] }, { id: "phind", name: "Phind", category: "ai_search_crawler", organization: "Phind", baseScore: 70, respectsRobots: true, uaPatterns: ["phind"] }, { id: "kagi", name: "Kagi", category: "ai_search_crawler", organization: "Kagi", baseScore: 65, respectsRobots: true, uaPatterns: ["kagi"] }, @@ -253,6 +252,7 @@ export const BOT_REGISTRY: readonly BotAgent[] = [ { id: "apify", name: "Apify", category: "commercial_scraper", organization: "Apify", baseScore: 65, respectsRobots: true, uaPatterns: ["apify"] }, { id: "crawlbase", name: "Crawlbase", category: "commercial_scraper", organization: "Crawlbase", baseScore: 65, respectsRobots: true, uaPatterns: ["crawlbase"] }, { id: "webscrapingapi", name: "WebScrapingAPI", category: "commercial_scraper", organization: "WebScrapingAPI", baseScore: 65, respectsRobots: true, uaPatterns: ["webscrapingapi"] }, + { id: "chatgpt-user", name: "ChatGPT-User", category: "ai_assistant", organization: "OpenAI", baseScore: 85, respectsRobots: true, uaPatterns: ["chatgpt-user", "chatgpt"] }, { id: "gemini-deep-research", name: "Gemini-Deep-Research", category: "ai_assistant", organization: "Google", baseScore: 65, respectsRobots: true, uaPatterns: ["gemini-deep-research"] }, { id: "copilot", name: "Microsoft Copilot", category: "ai_assistant", organization: "Microsoft", baseScore: 65, respectsRobots: true, uaPatterns: ["copilot"] }, { id: "cortana", name: "Cortana", category: "ai_assistant", organization: "Microsoft", baseScore: 60, respectsRobots: true, uaPatterns: ["cortana"] },