From 6ce86768d4b2287c23754536e33fcf262578a0a5 Mon Sep 17 00:00:00 2001 From: hrbrmstr Date: Mon, 7 Sep 2026 10:09:24 -0400 Subject: chore: initial commit --- ua-regex.json | 67 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 67 insertions(+) create mode 100644 ua-regex.json (limited to 'ua-regex.json') diff --git a/ua-regex.json b/ua-regex.json new file mode 100644 index 0000000..fc959ba --- /dev/null +++ b/ua-regex.json @@ -0,0 +1,67 @@ +{ + "generated_at": "2026-09-07T08:42:05+00:00", + "note": "Case-insensitive substring alternations, already escaped. A user-agent match is a claim, not a proof: pair it with /ip-ranges/ or reverse DNS before you trust it.", + "all": "(AdsBot\\-Google|AdsBot\\-Google\\-Mobile|AdsBot\\-Google\\-Mobile\\-Apps|AhrefsBot|AhrefsSiteAudit|AI2Bot|Ai2Bot\\-Dolma|aiHitBot|AIWebIndex|Amazonbot|Andibot|Anomura|anthropic\\-ai|APIs\\-Google|Applebot|archive\\.org_bot|atlassian\\-bot|AwarioRssBot|AwarioSmartBot|Baiduspider|barkrowler|bedrockbot|bingbot|Bytespider|CCBot|ChatGPT\\ Agent|ChatGPT\\-User|Claude\\-SearchBot|Claude\\-User|Claude\\-Web|ClaudeBot|Cloudflare\\-AutoRAG|cohere\\-ai|cohere\\-training\\-data\\-crawler|Cotoyogi|Crawl4AI|Crawlspace|DataForSeoBot|Diffbot|dotbot|DuckAssistBot|DuckDuckBot|EchoboxBot|ExaSearchBot|FacebookBot|facebookexternalhit|Factset_spyderbot|FeedFetcher\\-Google|FirecrawlAgent|Google\\-Agent|Google\\-CloudVertexBot|Google\\-CWS|Google\\-GeminiNotebook|Google\\-InspectionTool|Google\\-Pinpoint|Google\\-Read\\-Aloud|Google\\-Safety|Google\\-Site\\-Verification|Googlebot|Googlebot\\-Image|Googlebot\\-News|Googlebot\\-Video|GoogleMessages|GoogleOther|GoogleOther\\-Image|GoogleOther\\-Video|GoogleProducer|GPTBot|ia_archiver|ICC\\-Crawler|ImagesiftBot|img2dataset|ISSCyberRiskCrawler|Kagibot|KlaviyoAIBot|LAIONDownloader|Lightpanda|Linguee\\ Bot|Mediapartners\\-Google|meta\\-externalagent|meta\\-externalfetcher|Meta\\-WebIndexer|MistralAI\\-User|MJ12bot|MojeekBot|OAI\\-SearchBot|omgili|omgilibot|panscient\\.com|Perplexity\\-User|PerplexityBot|PetalBot|PhindBot|Pinterestbot|Poseidon\\ Research\\ Crawler|QualifiedBot|QuillBot|Qwantbot|Qwantbot\\-news|Reflectionbot|rogerbot|SBIntuitionsBot|Scrapy|Screaming\\ Frog\\ SEO\\ Spider|SemrushBot|SemrushBot\\-BA|SemrushBot\\-ESI|SemrushBot\\-FT|SemrushBot\\-OCOB|SemrushBot\\-SI|SemrushBot\\-SWA|SEOkicks|serpstatbot|SeznamBot|ShapBot|Sidetrade\\ indexer\\ bot|SiteAuditBot|Slackbot|Slackbot\\-LinkExpanding|SplitSignalBot|Storebot\\-Google|TerraCotta|Thinkbot|TikTokSpider|Timpibot|VelenPublicWebCrawler|Webzio\\-Extended|wpbot|YaK|YandexAdditional|YandexAdditionalBot|YandexBlogs|YandexBot|YandexCalendar|YandexComBot|YandexDirect|YandexFavicons|YandexImages|YandexMarket|YandexMedia|YandexMetrika|YandexMobileBot|YandexRenderResourcesBot|YandexScreenshotBot|YandexVideo|YandexWebmaster|Yeti|YouBot)", + "ai_only": "(AI2Bot|Ai2Bot\\-Dolma|aiHitBot|AIWebIndex|Amazonbot|Andibot|Anomura|anthropic\\-ai|atlassian\\-bot|AwarioRssBot|AwarioSmartBot|bedrockbot|Bytespider|CCBot|ChatGPT\\ Agent|ChatGPT\\-User|Claude\\-SearchBot|Claude\\-User|Claude\\-Web|ClaudeBot|Cloudflare\\-AutoRAG|cohere\\-ai|cohere\\-training\\-data\\-crawler|Cotoyogi|Diffbot|DuckAssistBot|EchoboxBot|ExaSearchBot|FacebookBot|Factset_spyderbot|Google\\-Agent|Google\\-CloudVertexBot|Google\\-GeminiNotebook|Google\\-Pinpoint|Google\\-Read\\-Aloud|GoogleOther|GoogleOther\\-Image|GoogleOther\\-Video|GPTBot|ICC\\-Crawler|ImagesiftBot|img2dataset|ISSCyberRiskCrawler|KlaviyoAIBot|LAIONDownloader|Linguee\\ Bot|meta\\-externalagent|meta\\-externalfetcher|Meta\\-WebIndexer|MistralAI\\-User|OAI\\-SearchBot|omgili|omgilibot|panscient\\.com|Perplexity\\-User|PerplexityBot|PhindBot|Poseidon\\ Research\\ Crawler|QualifiedBot|QuillBot|Reflectionbot|SBIntuitionsBot|SemrushBot\\-OCOB|ShapBot|Sidetrade\\ indexer\\ bot|TerraCotta|Thinkbot|TikTokSpider|VelenPublicWebCrawler|Webzio\\-Extended|YaK|YandexAdditional|YandexAdditionalBot|YandexCalendar|YouBot)", + "by_category": { + "ai-training": "(anthropic\\-ai|Bytespider|ClaudeBot|cohere\\-training\\-data\\-crawler|Cotoyogi|FacebookBot|Factset_spyderbot|GoogleOther|GoogleOther\\-Image|GoogleOther\\-Video|GPTBot|ICC\\-Crawler|ISSCyberRiskCrawler|Linguee\\ Bot|meta\\-externalagent|Poseidon\\ Research\\ Crawler|QuillBot|Reflectionbot|SBIntuitionsBot|SemrushBot\\-OCOB|Sidetrade\\ indexer\\ bot|TikTokSpider|Webzio\\-Extended|YandexAdditional|YandexAdditionalBot)", + "ai-search": "(AIWebIndex|Amazonbot|Andibot|Anomura|atlassian\\-bot|bedrockbot|Claude\\-SearchBot|Claude\\-Web|Cloudflare\\-AutoRAG|DuckAssistBot|ExaSearchBot|Google\\-CloudVertexBot|KlaviyoAIBot|Meta\\-WebIndexer|OAI\\-SearchBot|PerplexityBot|PhindBot|QualifiedBot|ShapBot|TerraCotta|YouBot)", + "user-fetch": "(ChatGPT\\ Agent|ChatGPT\\-User|Claude\\-User|cohere\\-ai|Google\\-Agent|Google\\-GeminiNotebook|Google\\-Pinpoint|Google\\-Read\\-Aloud|meta\\-externalfetcher|MistralAI\\-User|Perplexity\\-User|YandexCalendar)", + "search": "(Applebot|Baiduspider|bingbot|DuckDuckBot|Googlebot|Googlebot\\-Image|Googlebot\\-News|Googlebot\\-Video|Kagibot|MojeekBot|PetalBot|Pinterestbot|Qwantbot|Qwantbot\\-news|SeznamBot|Storebot\\-Google|Timpibot|YandexBlogs|YandexBot|YandexComBot|YandexFavicons|YandexImages|YandexMarket|YandexMedia|YandexMobileBot|YandexRenderResourcesBot|YandexVideo|Yeti)", + "tool": "(AdsBot\\-Google|AdsBot\\-Google\\-Mobile|AdsBot\\-Google\\-Mobile\\-Apps|APIs\\-Google|Crawl4AI|Crawlspace|FeedFetcher\\-Google|FirecrawlAgent|Google\\-CWS|Google\\-InspectionTool|Google\\-Safety|Google\\-Site\\-Verification|GoogleProducer|Lightpanda|Mediapartners\\-Google|Scrapy|Screaming\\ Frog\\ SEO\\ Spider|wpbot|YandexDirect|YandexMetrika|YandexScreenshotBot|YandexWebmaster)", + "dataset": "(AI2Bot|Ai2Bot\\-Dolma|aiHitBot|AwarioRssBot|AwarioSmartBot|CCBot|Diffbot|EchoboxBot|ImagesiftBot|img2dataset|LAIONDownloader|omgili|omgilibot|panscient\\.com|Thinkbot|VelenPublicWebCrawler|YaK)", + "preview": "(facebookexternalhit|GoogleMessages|Slackbot|Slackbot\\-LinkExpanding)", + "seo": "(AhrefsBot|AhrefsSiteAudit|barkrowler|DataForSeoBot|dotbot|MJ12bot|rogerbot|SemrushBot|SemrushBot\\-BA|SemrushBot\\-ESI|SemrushBot\\-FT|SemrushBot\\-SI|SemrushBot\\-SWA|SEOkicks|serpstatbot|SiteAuditBot|SplitSignalBot)", + "archive": "(archive\\.org_bot|ia_archiver)" + }, + "links": [ + { + "rel": "self", + "href": "https://www.pathwren.workers.dev/data/ua-regex.json", + "type": "application/json" + }, + { + "rel": "changes", + "href": "https://www.pathwren.workers.dev/changes.json?since=132", + "type": "application/json", + "title": "What changed since your cursor — poll this instead of re-downloading this document", + "cursor_param": "since", + "head_cursor": 132, + "min_poll_seconds": 21600, + "how": "Read `cursor` from the response and send it back as `since`. It advances only when something really changed, so an unchanged answer is proof rather than luck — about 2.5 KB, or a 304 with no body if you send back the ETag." + }, + { + "rel": "related", + "href": "https://www.pathwren.workers.dev/data/agents.json", + "type": "application/json", + "title": "Every crawler record in one file" + }, + { + "rel": "related", + "href": "https://www.pathwren.workers.dev/ip-ranges/all.json", + "type": "application/json", + "title": "Every operator-published prefix, unioned and grouped by source" + }, + { + "rel": "alternate", + "href": "https://www.pathwren.workers.dev/sitemap.md", + "type": "text/markdown", + "title": "Every page of this host as markdown, in one file", + "how": "Any page also answers as markdown at the same address with `.md` — and at `.mdx`, `.html.md` and `.html.mdx`, which are the same bytes. `Accept: text/markdown` on the page itself returns the same document. The HTML page stays canonical and every mirror says so in a Link header." + }, + { + "rel": "service-desc", + "href": "https://www.pathwren.workers.dev/openapi.json", + "type": "application/json", + "title": "Every read endpoint, described formally" + } + ], + "observed_at_edge": "(402explorer|a2a\\-adoption\\-audit|a2a\\-directory\\-discovery|a2a\\-directory\\-liveness|a2a\\-hub\\-pilot|A2A\\-Indexer|A2A\\-Registry|A2A\\-Registry\\-Healthbot|A2A\\-Registry\\-HealthCheck|A2A\\-Registry\\-Scanner|A2A\\-Registry\\-TaskProbe|a2a\\-security\\-research\\-crawler|ACC\\-Scout|Achilles|ActableSiteBot|ActionExtension|advisorsai\\-check|AetherLink\\-Public\\-Agent\\-Card\\-Policy\\-Check|AetherLink\\-Public\\-Discovery\\-Evidence|AetherLinkDiscoveryEvidence|AffsignalCrawler|AgenstryBot|agent\\-card\\-crawler|agent\\-card\\-crawler\\-chat|agent\\-guild\\-scout|agent\\-ready\\-scanner|agent\\-tools\\.cloud\\-crawler|agent\\-world\\-probe|AgentAlmanac\\-Snapshot|AgentCatalogBot|AgentDisco|AgentGaugeBot|AgentGrade|AgentIndexBot|AgentPointsDirectoryEnricher|agentprobe|AgentReputationBot|Agentry\\-Registry|AgentSure\\-MCPScan|AgentTrust\\-Monitor|Agoragentic\\-HealthMonitor|Agoragentic\\-SafeFetch|Agoragentic\\-Sandbox\\-Runner|AgoragenticAdminHealth|AhrefsBot|ai\\-crawler\\-logs|ai\\-crawler\\-robots|ai\\-crawler\\-verify|aisec\\-registry|AIVE\\-MCP\\-Discover|AIVE\\-MCP\\-EndpointProbe|Akkoma|AllMCPs\\-Verification|Amazonbot|api\\-forge\\-mcp\\-index|APIEvangelist|apievangelist\\-domain\\-probe|apievangelist\\-security\\-probe|apis\\.io\\-submit|APIs\\.json\\-Showcase\\-Verifier|AppleCoreMedia|archive\\.org_bot|Arctic|ardcrawl|AzureAI\\-SearchBot|bingbot|Blogtrottr|Boost|bot|Bun|Bytespider|CAPI2\\-A2A\\-Discovery|CAPI2\\-Neural\\-Revenue\\-Flow|Catodon|CBI|CBI\\-Penny\\-Utility\\-Demand\\-Probe|CensusBot|certfeed\\-prober|ChainWitness\\-PR36|choosemcp\\-harvester|Chrome|claude\\-user|ClaudeBot|Client|ColonistOne|com\\.apple\\.WebKit\\.Networking|CommaFeed|Conway\\-Replicatio\\-r91\\-strict\\-a2a\\-probe|crawler|Dart|DataForSeoBot|davefeedread|Discordbot|Doximity\\-Pipeline|DuckDuckBot|DuckDuckGo|EarnCompany\\-LanternScout|EndpointAudit|Enerlio|EPAIChat\\-A2A\\-AgentCard|EPAIChat\\-A2A\\-Compatibility\\-Test|exaforce\\-mcprep|ExaSearchBot|facebookexternalhit|FediAct|Fediverse|Feedly|Feedrabbit|ForgeScanner|FreePublicAPIs|Friendica|frndOS|GF\\-Agent\\-Toll\\-Outbound|GF\\-Agent\\-Toll\\-Remediation|ghxst\\-feed\\-importer|git|git\\.jimmy\\-b\\.se|GoCommand|GolemreachTrustBot|GoModuleMirror|Google|Google\\-Read\\-Aloud|Google\\-Structured\\-Data\\-Testing\\-Tool|Googlebot|Googlebot\\-Image|GoogleOther|GPTBot|GraphAdvocate\\-Outreach|gtm\\-engine|guild\\-reachability\\-probe|hhvm\\-internal|hlido\\-a50\\-probe|http\\.fetch|http\\.rb|hultra\\-link|InfrawatchCrawler|Inoreader|inoreader\\.com|invinoveritas\\-handshake|io\\.verifymcp|itinai\\-importer|jscrawler|ktor\\-client|lastseen\\-schema\\-probe|Lemmy|LemmyKotlinApi|Lightpanda|Linkwarden|llm4agents\\-cimd\\-audit|llms\\-txt\\-generator\\.de|LLMSE|loop\\-mcp\\-catalog\\-fetch|MachineCensus|Mastodon|MCP\\-Catalog|mcp\\-checker|MCP\\-Cloud\\-AboutBot|mcp\\-drift\\-monitor|mcp\\-observatory|mcp\\-registry|mcp\\-rugpull\\-research|mcp\\-schema\\-archive|mcp\\-scraper|mcp\\-server\\.io\\-healthcheck|MCP\\-Stats\\-Prober|mcp\\-ui\\-parity\\-audit|mcp\\-uptime|mcp\\-watch|mcp\\-widget\\-size\\-census|mcp2\\-research|mcpbeat|MCPCatalogSync|mcpcensusbot|mcpcheck|mcpgrade\\-probe|mcpi|mcplookup\\.com\\-probe|MCPMeter|mcpmon|mcporbit\\-oauth\\-probe|mcpqueen\\-grader|mcprush\\-verify|mcpscan|mcpserver\\-lol\\-probe|MCPWatch|MCPWitness|measure\\-mcp\\-schema|Mercury\\-Velvt|merlonix\\-attestation\\-verifier|meta\\-externalagent|Misskey|MisskeyMediaProxy|movanas\\-registry\\-snapshot|Mozilla|mwmbl|NeonCampLinksBot|NetworkingExtension|Neuronto|NewsBlur|nickw\\-tripwire|node\\-fetch|NotHumanSearch|oauth4webapi|openai\\-mcp|OpenGraph\\.io|oranges\\.live|Orbit\\-MCP\\-Registry\\-IconResolver|OttermindAgent|PackageHound|packages\\.ecosyste\\.ms|PerplexityBot|personal\\-agent\\-platform\\-readonly\\-audit|PieFed|ping\\.blo\\.gs|PingZen|pip|Pleroma|pod\\-directory\\-probe|PostmanRuntime|PrivacyBrowser|ProofBench|PyLova|Python|python\\-opengraph\\-jaywink|python\\-requests|QtCreator|quilr\\-catalog\\-probe|ReadYou|redditbot|reliability\\-bureau\\-spike|repology\\-linkchecker|req|Rokha\\-Probe|rokmcp\\-collector|RoninForgeObservatory|rootz\\-mcp\\-registry\\-prober|RSS\\-Parrot\\-Bot|Ruby|RubyGems\\.org|SaSame\\-Census\\-Era\\-Probe|SaSame\\-MCP\\-Audit|SaSameAgentAudit|Schema\\-Markup\\-Validator|scraper\\.io|SecurityScanner|SemrushBot|SentinelOracle|SEO\\-Audit\\-MCP|ShapBot|Sharkey|SiteGuardian|slack|Slackbot|SmitheryBot|snyk\\-advisor|SofyaBot|SolvedEarthPriceBot|spanly\\-enrich|Spider_Bot|strand\\-mcp|SummalyBot|Superfeedr|swagger\\-validator|Sync|Tapestry|TAR\\-Directory\\-Indexer|TAR\\-Discovery|TAR\\-Health|Telegram|TelegramBot|teppi\\-probe|the402\\-mcp|theoldreader\\.com|TikTokSpider|TrimtabVerifier|truespar\\-mcp\\-registry|trustoven\\-manifest\\-observer|Universal\\-Service\\-Registry\\-MCP\\-Tool\\-Probe|UptimeSignal|Upvote|url2md|utopian\\-foundry\\-probe|Validator\\.nu|VerifyMCP\\-OwnersBot|Waggle|Watchpup|WebAlert|WebSub\\.rocks|WellknownBot|WhatsApp|x1r\\-replication\\-probe|x402\\-observatory|yacybot|YandexBot|zevruna\\-monitor)", + "observed_at_edge_note": "1128 named clients observed requesting this host between 2026-08-31T20:58:11+00:00 and 2026-09-07T08:37:00+00:00, one page each under https://www.pathwren.workers.dev/bot/ and the whole set at https://www.pathwren.workers.dev/data/observed-clients.json. Observation, not a directory: it says these names arrived here in that window and nothing about what they are for.", + "changed_at": "2026-09-07T08:42:05+00:00", + "recent_changes": [ + "2026-09-07T08:42Z - changed: observed_at_edge_note; 12776->13059 bytes", + "2026-09-07T02:48Z - changed: links", + "2026-09-07T02:39Z - changed: observed_at_edge_note; 12721->12776 bytes" + ] +} \ No newline at end of file -- cgit v1.2.3