diff --git a/.github/workflows/main.yml b/.github/workflows/main.yml
index 36e7a1c..7bbb11a 100644
--- a/.github/workflows/main.yml
+++ b/.github/workflows/main.yml
@@ -21,7 +21,10 @@ jobs:
- uses: actions/checkout@v4
with:
fetch-depth: 2
- - run: |
+ - env:
+ INPUT_MESSAGE: ${{ inputs.message }}
+ HEAD_COMMIT_MESSAGE: ${{ github.event.head_commit.message }}
+ run: |
pip install beautifulsoup4
git config --global user.name "ai.robots.txt"
git config --global user.email "ai.robots.txt@users.noreply.github.com"
@@ -39,10 +42,10 @@ jobs:
echo "No staged changes to commit. Skipping commit and push."
exit 0
fi
- if [ -n "${{ inputs.message }}" ]; then
- git commit -m "${{ inputs.message }}"
+ if [ -n "$INPUT_MESSAGE" ]; then
+ git commit -m "$INPUT_MESSAGE"
else
- git commit -m "${{ github.event.head_commit.message }}"
+ git commit -m "$HEAD_COMMIT_MESSAGE"
fi
git push
shell: bash
diff --git a/.htaccess b/.htaccess
index 65a420a..2e15f5b 100644
--- a/.htaccess
+++ b/.htaccess
@@ -1,3 +1,3 @@
RewriteEngine On
-RewriteCond %{HTTP_USER_AGENT} (^(AddSearchBot|AgentTimes|AI2Bot|AI2Bot\-DeepResearchEval|Ai2Bot\-Dolma|aiHitBot|AIWebIndex|amazon\-kendra|amazon\-QBusiness|Amazonbot|AmazonBuyForMe|Amzn\-SearchBot|Amzn\-User|Andibot|Anomura|anthropic\-ai|ApifyBot|ApifyWebsiteContentCrawler|Applebot|Applebot\-Extended|Aranet\-SearchBot|atlassian\-bot|Awario|AzureAI\-SearchBot|bedrockbot|bigsur\.ai|Bravebot|Brightbot|Brightbot\ 1\.0|BuddyBot|Bytespider|CCBot|Channel3Bot|ChatGLM\-Spider|ChatGPT\ Agent|ChatGPT\-User|Claude\-Code|Claude\-SearchBot|Claude\-User|Claude\-Web|ClaudeBot|Cloudflare\-AutoRAG|CloudVertexBot|Code|cohere\-ai|cohere\-training\-data\-crawler|Cotoyogi|CragCrawler|Crawl4AI|Crawlspace|Cursor|Datenbank\ Crawler|DeepSeekBot|Devin|Diffbot|DuckAssistBot|Echobot\ Bot|EchoboxBot|ExaBot|FacebookBot|facebookexternalhit|Factset_spyderbot|FirecrawlAgent|FriendlyCrawler|GeistHaus\-PageFetcher|Gemini\-Deep\-Research|Google\-Agent|Google\-CloudVertexBot|Google\-Extended|Google\-Firebase|Google\-Gemini\-CLI|Google\-NotebookLM|GoogleAgent\-Mariner|GoogleAgent\-URLContext|GoogleOther|GoogleOther\-Image|GoogleOther\-Video|GPTBot|HenkBot|iAskBot|iaskspider|iaskspider/2\.0|IbouBot|ICC\-Crawler|ImagesiftBot|imageSpider|img2dataset|ISSCyberRiskCrawler|kagi\-fetcher|Kangaroo\ Bot|Kimi\-User|KlaviyoAIBot|KunatoCrawler|laion\-huggingface\-processor|LAIONDownloader|LCC|LinerBot|Linguee\ Bot|LinkupBot|Manus\-User|meta\-externalagent|Meta\-ExternalAgent|meta\-externalfetcher|Meta\-ExternalFetcher|meta\-webindexer|MistralAI\-User|MistralAI\-User/1\.0|Mozilla\-Tabstack|MyCentralAIScraperBot|NagetBot|netEstate\ Imprint\ Crawler|newsai|NotebookLM|NovaAct|OAI\-SearchBot|omgili|omgilibot|OpenAI|opencode|Operator|PanguBot|Panscient|panscient\.com|Perplexity\-User|PerplexityBot|PetalBot|PhindBot|Poggio\-Citations|Poseidon\ Research\ Crawler|QualifiedBot|Querit\-SearchBot|QueritBot|QuillBot|quillbot\.com|SBIntuitionsBot|Scrapy|SemrushBot\-OCOB|SemrushBot\-SWA|Shap\-User|ShapBot|Sidetrade\ indexer\ bot|Spider|TavilyBot|Terra\ Cotta|TerraCotta|Thinkbot|TikTokSpider|Timpibot|TongyiBot|Trae|TwinAgent|UseAI|VelenPublicWebCrawler|WARDBot|Webzio\-Extended|webzio\-extended|wpbot|WRTNBot|YaK|YandexAdditional|YandexAdditionalBot|YiyanBot|YouBot|ZanistaBot)$|Code/[0-9.]+) [NC]
+RewriteCond %{HTTP_USER_AGENT} (\b(AddSearchBot|AgentTimes|AI2Bot|AI2Bot-DeepResearchEval|Ai2Bot-Dolma|aiHitBot|AIWebIndex|amazon-kendra|amazon-QBusiness|Amazonbot|AmazonBuyForMe|Amzn-SearchBot|Amzn-User|Andibot|Anomura|anthropic-ai|ApifyBot|ApifyWebsiteContentCrawler|Applebot|Applebot-Extended|Aranet-SearchBot|atlassian-bot|Awario|AzureAI-SearchBot|bedrockbot|bigsur\.ai|Bravebot|Brightbot|Brightbot\ 1\.0|BuddyBot|Bytespider|CCBot|Channel3Bot|ChatGLM-Spider|ChatGPT\ Agent|ChatGPT-User|Claude-Code|Claude-SearchBot|Claude-User|Claude-Web|ClaudeBot|Cloudflare-AutoRAG|CloudVertexBot|Code|cohere-ai|cohere-training-data-crawler|Cotoyogi|CragCrawler|Crawl4AI|Crawlspace|Cursor|Datenbank\ Crawler|DeepSeekBot|Devin|Diffbot|Diffbot-User|DuckAssistBot|Echobot\ Bot|EchoboxBot|ExaBot|ExaSearchBot|FacebookBot|facebookexternalhit|Factset_spyderbot|FirecrawlAgent|FriendlyCrawler|GeistHaus-PageFetcher|Gemini-Deep-Research|Google-Agent|Google-CloudVertexBot|Google-Extended|Google-Firebase|Google-Gemini-CLI|Google-NotebookLM|GoogleAgent-Mariner|GoogleAgent-URLContext|GoogleOther|GoogleOther-Image|GoogleOther-Video|GPTBot|HenkBot|iAskBot|iaskspider|iaskspider/2\.0|ICC-Crawler|ImagesiftBot|imageSpider|img2dataset|ISSCyberRiskCrawler|kagi-fetcher|Kangaroo\ Bot|Kimi-SearchBot|Kimi-User|KimiBot|KlaviyoAIBot|KunatoCrawler|laion-huggingface-processor|LAIONDownloader|LCC|Lightpanda|LinerBot|Linguee\ Bot|LinkupBot|Manus-User|meta-externalagent|Meta-ExternalAgent|meta-externalfetcher|Meta-ExternalFetcher|meta-webindexer|MistralAI-Training|MistralAI-User|MistralAI-User/1\.0|Mozilla-Tabstack|MyCentralAIScraperBot|NagetBot|netEstate\ Imprint\ Crawler|newsai|NotebookLM|NovaAct|OAI-AdsBot|OAI-SearchBot|omgili|omgilibot|OpenAI|opencode|Operator|PanguBot|Panscient|panscient\.com|Perplexity-User|PerplexityBot|PetalBot|PhindBot|Poggio-Citations|Poseidon\ Research\ Crawler|QualifiedBot|Querit-SearchBot|QueritBot|QuillBot|quillbot\.com|Reflectionbot|SBIntuitionsBot|Scrapy|SemrushBot-OCOB|SemrushBot-SWA|Shap-User|ShapBot|Sidetrade\ indexer\ bot|Spider|TavilyBot|Terra\ Cotta|TerraCotta|Thinkbot|TikTokSpider|Timpibot|TongyiBot|Trae|TwinAgent|UseAI|VelenPublicWebCrawler|WARDBot|Webzio-Extended|webzio-extended|wpbot|WRTNBot|YaK|YandexAdditional|YandexAdditionalBot|YiyanBot|YouBot|ZanistaBot)\b) [NC]
RewriteRule !^/?robots\.txt$ - [F]
diff --git a/Caddyfile b/Caddyfile
index d73b9ad..c14cf22 100644
--- a/Caddyfile
+++ b/Caddyfile
@@ -1,3 +1,3 @@
@aibots {
- header_regexp User-Agent "(^(AddSearchBot|AgentTimes|AI2Bot|AI2Bot\-DeepResearchEval|Ai2Bot\-Dolma|aiHitBot|AIWebIndex|amazon\-kendra|amazon\-QBusiness|Amazonbot|AmazonBuyForMe|Amzn\-SearchBot|Amzn\-User|Andibot|Anomura|anthropic\-ai|ApifyBot|ApifyWebsiteContentCrawler|Applebot|Applebot\-Extended|Aranet\-SearchBot|atlassian\-bot|Awario|AzureAI\-SearchBot|bedrockbot|bigsur\.ai|Bravebot|Brightbot|Brightbot\ 1\.0|BuddyBot|Bytespider|CCBot|Channel3Bot|ChatGLM\-Spider|ChatGPT\ Agent|ChatGPT\-User|Claude\-Code|Claude\-SearchBot|Claude\-User|Claude\-Web|ClaudeBot|Cloudflare\-AutoRAG|CloudVertexBot|Code|cohere\-ai|cohere\-training\-data\-crawler|Cotoyogi|CragCrawler|Crawl4AI|Crawlspace|Cursor|Datenbank\ Crawler|DeepSeekBot|Devin|Diffbot|DuckAssistBot|Echobot\ Bot|EchoboxBot|ExaBot|FacebookBot|facebookexternalhit|Factset_spyderbot|FirecrawlAgent|FriendlyCrawler|GeistHaus\-PageFetcher|Gemini\-Deep\-Research|Google\-Agent|Google\-CloudVertexBot|Google\-Extended|Google\-Firebase|Google\-Gemini\-CLI|Google\-NotebookLM|GoogleAgent\-Mariner|GoogleAgent\-URLContext|GoogleOther|GoogleOther\-Image|GoogleOther\-Video|GPTBot|HenkBot|iAskBot|iaskspider|iaskspider/2\.0|IbouBot|ICC\-Crawler|ImagesiftBot|imageSpider|img2dataset|ISSCyberRiskCrawler|kagi\-fetcher|Kangaroo\ Bot|Kimi\-User|KlaviyoAIBot|KunatoCrawler|laion\-huggingface\-processor|LAIONDownloader|LCC|LinerBot|Linguee\ Bot|LinkupBot|Manus\-User|meta\-externalagent|Meta\-ExternalAgent|meta\-externalfetcher|Meta\-ExternalFetcher|meta\-webindexer|MistralAI\-User|MistralAI\-User/1\.0|Mozilla\-Tabstack|MyCentralAIScraperBot|NagetBot|netEstate\ Imprint\ Crawler|newsai|NotebookLM|NovaAct|OAI\-SearchBot|omgili|omgilibot|OpenAI|opencode|Operator|PanguBot|Panscient|panscient\.com|Perplexity\-User|PerplexityBot|PetalBot|PhindBot|Poggio\-Citations|Poseidon\ Research\ Crawler|QualifiedBot|Querit\-SearchBot|QueritBot|QuillBot|quillbot\.com|SBIntuitionsBot|Scrapy|SemrushBot\-OCOB|SemrushBot\-SWA|Shap\-User|ShapBot|Sidetrade\ indexer\ bot|Spider|TavilyBot|Terra\ Cotta|TerraCotta|Thinkbot|TikTokSpider|Timpibot|TongyiBot|Trae|TwinAgent|UseAI|VelenPublicWebCrawler|WARDBot|Webzio\-Extended|webzio\-extended|wpbot|WRTNBot|YaK|YandexAdditional|YandexAdditionalBot|YiyanBot|YouBot|ZanistaBot)$|Code/[0-9.]+)"
+ header_regexp User-Agent "\b(AddSearchBot|AgentTimes|AI2Bot|AI2Bot-DeepResearchEval|Ai2Bot-Dolma|aiHitBot|AIWebIndex|amazon-kendra|amazon-QBusiness|Amazonbot|AmazonBuyForMe|Amzn-SearchBot|Amzn-User|Andibot|Anomura|anthropic-ai|ApifyBot|ApifyWebsiteContentCrawler|Applebot|Applebot-Extended|Aranet-SearchBot|atlassian-bot|Awario|AzureAI-SearchBot|bedrockbot|bigsur\.ai|Bravebot|Brightbot|Brightbot\ 1\.0|BuddyBot|Bytespider|CCBot|Channel3Bot|ChatGLM-Spider|ChatGPT\ Agent|ChatGPT-User|Claude-Code|Claude-SearchBot|Claude-User|Claude-Web|ClaudeBot|Cloudflare-AutoRAG|CloudVertexBot|Code|cohere-ai|cohere-training-data-crawler|Cotoyogi|CragCrawler|Crawl4AI|Crawlspace|Cursor|Datenbank\ Crawler|DeepSeekBot|Devin|Diffbot|Diffbot-User|DuckAssistBot|Echobot\ Bot|EchoboxBot|ExaBot|ExaSearchBot|FacebookBot|facebookexternalhit|Factset_spyderbot|FirecrawlAgent|FriendlyCrawler|GeistHaus-PageFetcher|Gemini-Deep-Research|Google-Agent|Google-CloudVertexBot|Google-Extended|Google-Firebase|Google-Gemini-CLI|Google-NotebookLM|GoogleAgent-Mariner|GoogleAgent-URLContext|GoogleOther|GoogleOther-Image|GoogleOther-Video|GPTBot|HenkBot|iAskBot|iaskspider|iaskspider/2\.0|ICC-Crawler|ImagesiftBot|imageSpider|img2dataset|ISSCyberRiskCrawler|kagi-fetcher|Kangaroo\ Bot|Kimi-SearchBot|Kimi-User|KimiBot|KlaviyoAIBot|KunatoCrawler|laion-huggingface-processor|LAIONDownloader|LCC|Lightpanda|LinerBot|Linguee\ Bot|LinkupBot|Manus-User|meta-externalagent|Meta-ExternalAgent|meta-externalfetcher|Meta-ExternalFetcher|meta-webindexer|MistralAI-Training|MistralAI-User|MistralAI-User/1\.0|Mozilla-Tabstack|MyCentralAIScraperBot|NagetBot|netEstate\ Imprint\ Crawler|newsai|NotebookLM|NovaAct|OAI-AdsBot|OAI-SearchBot|omgili|omgilibot|OpenAI|opencode|Operator|PanguBot|Panscient|panscient\.com|Perplexity-User|PerplexityBot|PetalBot|PhindBot|Poggio-Citations|Poseidon\ Research\ Crawler|QualifiedBot|Querit-SearchBot|QueritBot|QuillBot|quillbot\.com|Reflectionbot|SBIntuitionsBot|Scrapy|SemrushBot-OCOB|SemrushBot-SWA|Shap-User|ShapBot|Sidetrade\ indexer\ bot|Spider|TavilyBot|Terra\ Cotta|TerraCotta|Thinkbot|TikTokSpider|Timpibot|TongyiBot|Trae|TwinAgent|UseAI|VelenPublicWebCrawler|WARDBot|Webzio-Extended|webzio-extended|wpbot|WRTNBot|YaK|YandexAdditional|YandexAdditionalBot|YiyanBot|YouBot|ZanistaBot)\b"
}
\ No newline at end of file
diff --git a/FAQ.md b/FAQ.md
index 7264819..ee81b3b 100644
--- a/FAQ.md
+++ b/FAQ.md
@@ -32,6 +32,16 @@ Yes, provided the crawlers identify themselves and your application/hosting supp
Some crawlers — [such as Perplexity](https://rknight.me/blog/perplexity-ai-is-lying-about-its-user-agent/) — do not identify themselves via their user agent strings and, as such, are difficult to block.
+## Can I use `robots.json` directly in my own tooling?
+
+You're welcome to, with a caveat. `robots.json` isn't intended as a primary
+deliverable of this project — the generated configuration files are. If you consume
+it yourself, note that the agent names should be matched as whole words rather than
+substrings.
+
+The generated configs wrap items in `\b(...)\b` for this reason. Without
+word boundaries, agent names match inside unrelated strings (see [issue 208](https://github.com/ai-robots-txt/ai.robots.txt/issues/208) for an example).
+
## What can we do if a bot doesn't respect `robots.txt`?
That depends on your stack.
diff --git a/README.md b/README.md
index 9af0d86..6e979aa 100644
--- a/README.md
+++ b/README.md
@@ -1,16 +1,17 @@
# ai.robots.txt
-
+
This list contains AI-related crawlers of all types, regardless of purpose. We encourage you to contribute to and implement this list on your own site. See [information about the listed crawlers](./table-of-bot-metrics.md) and the [FAQ](https://github.com/ai-robots-txt/ai.robots.txt/blob/main/FAQ.md).
A number of these crawlers have been sourced from [Known Agents](https://knownagents.com) and we appreciate the ongoing effort they put in to track these crawlers.
-If you'd like to add information about a crawler to the list, please make a pull request with the bot name added to `robots.txt`, `ai.txt`, and any relevant details in `table-of-bot-metrics.md` to help people understand what's crawling.
+If you'd like to add an AI-related crawler to the list, please see "Contributing" below.
## Usage
This repository provides the following files:
+
- `robots.txt`
- `.htaccess`
- `nginx-block-ai-bots.conf`
@@ -27,13 +28,16 @@ Note that, as stated in the [httpd documentation](https://httpd.apache.org/docs/
`Caddyfile` includes a Header Regex matcher group you can copy or import into your Caddyfile, the rejection can then be handled as followed `abort @aibots`
-`haproxy-block-ai-bots.txt` may be used to configure HAProxy to block AI bots. To implement it;
+`haproxy-block-ai-bots.txt` may be used to configure HAProxy to block AI bots. To implement it:
+
1. Add the file to the config directory of HAProxy
-2. Add the following lines in the `frontend` section;
+2. Add the following lines in the `frontend` section:
+
```
acl ai_robot hdr_sub(user-agent) -i -f /etc/haproxy/haproxy-block-ai-bots.txt
http-request deny if ai_robot
```
+
(Note that the path of the `haproxy-block-ai-bots.txt` may be different in your environment.)
`lighttpd-block-ai-bots.conf` can be included with `include "fragments/lighttpd-block-ai-bots.conf"` in your lighttpd configuration either globally or in any conditional section.
@@ -47,15 +51,23 @@ middleware plugin for [Traefik](https://traefik.io/traefik/) to automatically ad
file on-the-fly.
- Alternatively you can [manually configure Traefik](./docs/traefik-manual-setup.md) to centrally serve a static `robots.txt`.
+
+- [Bot Ledger](https://farrelldan.github.io/ai-bot-directory/): free, static directory of verified AI crawlers with a one-click `robots.txt` and `llms.txt` generator. No signup required.
+
+- [KI-Zugangsindex](https://peppe1337.github.io/ki-zugangsindex/): open dataset on how widely this kind of blocking is actually deployed in the German (`.de`) web, measured on a fixed panel of 600 domains so the same sites can be re-checked over time.
+
## Contributing
A note about contributing: updates should be added/made to `robots.json`. A GitHub action will then generate the updated `robots.txt`, `table-of-bot-metrics.md`, `.htaccess` and `nginx-block-ai-bots.conf`.
You can run the tests by [installing](https://www.python.org/about/gettingstarted/) Python 3, installing the dependencies:
+
```console
pip install -r requirements.txt
```
+
and then issuing:
+
```console
code/tests.py
```
@@ -66,21 +78,18 @@ The `.editorconfig` file provides standard editor options for this project. See
Admins may ship a new release `v1.n` (where `n` increments the minor version of the current release) as follows:
-* Navigate to the [new release page](https://github.com/ai-robots-txt/ai.robots.txt/releases/new) on GitHub.
-* Click `Select tag`, choose `Create new tag`, enter `v1.n` in the pop-up, and click `Create`.
-* Enter a suitable release title (e.g. `v1.n: adds user-agent1, user-agent2`).
-* Click `Generate release notes`.
-* Click `Publish release`.
+- Navigate to the [new release page](https://github.com/ai-robots-txt/ai.robots.txt/releases/new) on GitHub.
+- Click `Select tag`, choose `Create new tag`, enter `v1.n` in the pop-up, and click `Create`.
+- Enter a suitable release title (e.g. `v1.n: adds user-agent1, user-agent2`).
+- Click `Generate release notes`.
+- Click `Publish release`.
A GitHub action will then add the asset `robots.txt` to the release. That's it.
## Subscribe to updates
You can subscribe to list updates via RSS/Atom with the releases feed:
-
-```
-https://github.com/ai-robots-txt/ai.robots.txt/releases.atom
-```
+`https://github.com/ai-robots-txt/ai.robots.txt/releases.atom`.
You can subscribe with [Feedly](https://feedly.com/i/subscription/feed/https://github.com/ai-robots-txt/ai.robots.txt/releases.atom), [Inoreader](https://www.inoreader.com/?add_feed=https://github.com/ai-robots-txt/ai.robots.txt/releases.atom), [The Old Reader](https://theoldreader.com/feeds/subscribe?url=https://github.com/ai-robots-txt/ai.robots.txt/releases.atom), [Feedbin](https://feedbin.me/?subscribe=https://github.com/ai-robots-txt/ai.robots.txt/releases.atom), or any other reader app.
@@ -97,6 +106,7 @@ implements RSL as well as payment processing for WordPress sites.
If you use [Cloudflare's hard block](https://blog.cloudflare.com/declaring-your-aindependence-block-ai-bots-scrapers-and-crawlers-with-a-single-click) alongside this list, you can report abusive crawlers that don't respect `robots.txt` [here](https://docs.google.com/forms/d/e/1FAIpQLScbUZ2vlNSdcsb8LyTeSF7uLzQI96s0BKGoJ6wQ6ocUFNOKEg/viewform).
But even if you don't use Cloudflare's hard block, their list of [verified bots](https://radar.cloudflare.com/traffic/verified-bots) may come in handy.
+
## Additional resources
- [Blocking Bots with Nginx](https://rknight.me/blog/blocking-bots-with-nginx/) by Robb Knight
@@ -105,3 +115,4 @@ But even if you don't use Cloudflare's hard block, their list of [verified bots]
- [Blockin' bots on Netlify](https://www.jeremiak.com/blog/block-bots-netlify-edge-functions/) by Jeremia Kimelman
- [Blocking AI web crawlers](https://underlap.org/blocking-ai-web-crawlers) by Glyn Normington
- [Block AI Bots from Crawling Websites Using Robots.txt](https://originality.ai/ai-bot-blocking) by Jonathan Gillham, Originality.AI
+- [AI Access Checker: see which AI crawlers a site's robots.txt allows or blocks](https://www.greadme.com/ai-access-checker) by Saar Twito, Greadme
diff --git a/code/robots.py b/code/robots.py
index 48584f2..536175a 100755
--- a/code/robots.py
+++ b/code/robots.py
@@ -175,47 +175,58 @@ def json_to_table(robots_json):
def list_to_pcre(robots_json):
# Python re is not 100% identical to PCRE which is used by Apache, but it
# should probably be close enough in the real world for re.escape to work.
- exact_agents = "|".join(map(re.escape, robots_json))
- patterns = [f"^({exact_agents})$"]
- patterns.extend(
- f"{re.escape(agent)}/[0-9.]+"
- for agent, config in robots_json.items()
- if config.get("has_name_and_version", False)
- )
- return f"({'|'.join(patterns)})"
+ # We additionally un-escape '-' since it only requires escaping within
+ # character classes (which are also escaped and prevented here) and '/'
+ # since this is not used as the regexp delimeter in any server software.
+ def escape(pattern):
+ pattern = re.escape(pattern)
+ for c in "-/":
+ pattern = pattern.replace(fr"\{c}", c)
+ return pattern
+
+ exact_agents = "|".join(map(escape, robots_json))
+ return f"\\b({exact_agents})\\b"
def json_to_htaccess(robot_json):
# Creates a .htaccess filter file. It uses a regular expression to filter out
# User agents that contain any of the blocked values.
+ # The regular expression is wrapped in parenthesis, so a leading [-!=<>] does
+ # not accidentally change which comparison type is used.
htaccess = "RewriteEngine On\n"
- htaccess += f"RewriteCond %{{HTTP_USER_AGENT}} {list_to_pcre(robot_json)} [NC]\n"
+ htaccess += f"RewriteCond %{{HTTP_USER_AGENT}} ({list_to_pcre(robot_json)}) [NC]\n"
htaccess += "RewriteRule !^/?robots\\.txt$ - [F]\n"
return htaccess
def json_to_nginx(robot_json):
# Creates an Nginx config file. This config snippet can be included in
# nginx server{} blocks to block AI bots.
- config = f"set $block 0;\n\nif ($http_user_agent ~* \"{list_to_pcre(robot_json)}\") {{\n set $block 1;\n}}\n\nif ($request_uri = \"/robots.txt\") {{\n set $block 0;\n}}\n\nif ($block) {{\n return 403;\n}}"
+ config = f"set $block 0;\n\nif ($http_user_agent ~ {list_to_pcre(robot_json)!r}) {{\n set $block 1;\n}}\n\nif ($request_uri = '/robots.txt') {{\n set $block 0;\n}}\n\nif ($block) {{\n return 403;\n}}"
return config
def json_to_lighttpd(robot_json):
# Creates an Lighttpd config file. This config snippet can be included in
# Lighttpd configuration global or in $HTTP conditionals to block AI bots.
- config = f"$HTTP[\"url\"] != \"/robots.txt\" {{ $HTTP[\"user-agent\"] =~ \"{list_to_pcre(robot_json)}\" {{ url.access-deny = ( \"\" ) }} }}"
+ # single quotes (as returned by repr) are not valid string delimeters, so we
+ # must manually quote it end ensure no unescaped quotes are inside.
+ escaped_quotes = list_to_pcre(robot_json).replace('"', '\\"')
+ config = f'$HTTP["url"] != "/robots.txt" {{ $HTTP["user-agent"] =~ "{escaped_quotes}" {{ url.access-deny = ( "" ) }} }}'
return config
def json_to_caddy(robot_json):
+ # single quotes (as returned by repr) are not valid string delimeters, so we
+ # must manually quote it end ensure no unescaped quotes are inside.
+ escaped_quotes = list_to_pcre(robot_json).replace('"', '\\"')
caddyfile = "@aibots {\n "
- caddyfile += f' header_regexp User-Agent "{list_to_pcre(robot_json)}"'
+ caddyfile += f' header_regexp User-Agent "{escaped_quotes}"'
caddyfile += "\n}"
return caddyfile
def json_to_haproxy(robots_json):
# Creates a source file for HAProxy. Follow instructions in the README to implement it.
- txt = "\n".join(f"{k}" for k in robots_json.keys())
+ txt = "\n".join(robots_json.keys())
return txt
@@ -266,8 +277,7 @@ def conversions():
if __name__ == "__main__":
import argparse
-
- parser = argparse.ArgumentParser()
+
parser = argparse.ArgumentParser(
prog="ai-robots",
description="Collects and updates information about web scrapers of AI companies.",
diff --git a/code/test_files/.htaccess b/code/test_files/.htaccess
index f80814c..b7b69c7 100644
--- a/code/test_files/.htaccess
+++ b/code/test_files/.htaccess
@@ -1,3 +1,3 @@
RewriteEngine On
-RewriteCond %{HTTP_USER_AGENT} (^(AI2Bot|Ai2Bot\-Dolma|Amazonbot|anthropic\-ai|Applebot|Applebot\-Extended|Bytespider|CCBot|ChatGPT\-User|Claude\-Web|ClaudeBot|cohere\-ai|Diffbot|FacebookBot|facebookexternalhit|FriendlyCrawler|Google\-Extended|GoogleOther|GoogleOther\-Image|GoogleOther\-Video|GPTBot|iaskspider/2\.0|ICC\-Crawler|ImagesiftBot|img2dataset|ISSCyberRiskCrawler|Kangaroo\ Bot|Meta\-ExternalAgent|Meta\-ExternalFetcher|OAI\-SearchBot|omgili|omgilibot|Perplexity\-User|PerplexityBot|PetalBot|Scrapy|Sidetrade\ indexer\ bot|Timpibot|VelenPublicWebCrawler|Webzio\-Extended|YouBot|crawler\.with\.dots|star\*\*\*crawler|Is\ this\ a\ crawler\?|a\[mazing\]\{42\}\(robot\)|2\^32\$|curl\|sudo\ bash)$) [NC]
+RewriteCond %{HTTP_USER_AGENT} (\b(AI2Bot|Ai2Bot-Dolma|Amazonbot|anthropic-ai|Applebot|Applebot-Extended|Bytespider|CCBot|ChatGPT-User|Claude-Web|ClaudeBot|cohere-ai|Diffbot|FacebookBot|facebookexternalhit|FriendlyCrawler|Google-Extended|GoogleOther|GoogleOther-Image|GoogleOther-Video|GPTBot|iaskspider/2\.0|ICC-Crawler|ImagesiftBot|img2dataset|ISSCyberRiskCrawler|Kangaroo\ Bot|Meta-ExternalAgent|Meta-ExternalFetcher|OAI-SearchBot|omgili|omgilibot|Perplexity-User|PerplexityBot|PetalBot|Scrapy|Sidetrade\ indexer\ bot|Timpibot|VelenPublicWebCrawler|Webzio-Extended|YouBot|crawler\.with\.dots|star\*\*\*crawler|Is\ this\ a\ crawler\?|a\[mazing\]\{42\}\(robot\)|2\^32\$|curl\|sudo\ bash)\b) [NC]
RewriteRule !^/?robots\.txt$ - [F]
diff --git a/code/test_files/Caddyfile b/code/test_files/Caddyfile
index fe69d83..04ba618 100644
--- a/code/test_files/Caddyfile
+++ b/code/test_files/Caddyfile
@@ -1,3 +1,3 @@
@aibots {
- header_regexp User-Agent "(^(AI2Bot|Ai2Bot\-Dolma|Amazonbot|anthropic\-ai|Applebot|Applebot\-Extended|Bytespider|CCBot|ChatGPT\-User|Claude\-Web|ClaudeBot|cohere\-ai|Diffbot|FacebookBot|facebookexternalhit|FriendlyCrawler|Google\-Extended|GoogleOther|GoogleOther\-Image|GoogleOther\-Video|GPTBot|iaskspider/2\.0|ICC\-Crawler|ImagesiftBot|img2dataset|ISSCyberRiskCrawler|Kangaroo\ Bot|Meta\-ExternalAgent|Meta\-ExternalFetcher|OAI\-SearchBot|omgili|omgilibot|Perplexity\-User|PerplexityBot|PetalBot|Scrapy|Sidetrade\ indexer\ bot|Timpibot|VelenPublicWebCrawler|Webzio\-Extended|YouBot|crawler\.with\.dots|star\*\*\*crawler|Is\ this\ a\ crawler\?|a\[mazing\]\{42\}\(robot\)|2\^32\$|curl\|sudo\ bash)$)"
+ header_regexp User-Agent "\b(AI2Bot|Ai2Bot-Dolma|Amazonbot|anthropic-ai|Applebot|Applebot-Extended|Bytespider|CCBot|ChatGPT-User|Claude-Web|ClaudeBot|cohere-ai|Diffbot|FacebookBot|facebookexternalhit|FriendlyCrawler|Google-Extended|GoogleOther|GoogleOther-Image|GoogleOther-Video|GPTBot|iaskspider/2\.0|ICC-Crawler|ImagesiftBot|img2dataset|ISSCyberRiskCrawler|Kangaroo\ Bot|Meta-ExternalAgent|Meta-ExternalFetcher|OAI-SearchBot|omgili|omgilibot|Perplexity-User|PerplexityBot|PetalBot|Scrapy|Sidetrade\ indexer\ bot|Timpibot|VelenPublicWebCrawler|Webzio-Extended|YouBot|crawler\.with\.dots|star\*\*\*crawler|Is\ this\ a\ crawler\?|a\[mazing\]\{42\}\(robot\)|2\^32\$|curl\|sudo\ bash)\b"
}
\ No newline at end of file
diff --git a/code/test_files/lighttpd-block-ai-bots.conf b/code/test_files/lighttpd-block-ai-bots.conf
index 54dbf08..6139ffa 100644
--- a/code/test_files/lighttpd-block-ai-bots.conf
+++ b/code/test_files/lighttpd-block-ai-bots.conf
@@ -1 +1 @@
-$HTTP["url"] != "/robots.txt" { $HTTP["user-agent"] =~ "(^(AI2Bot|Ai2Bot\-Dolma|Amazonbot|anthropic\-ai|Applebot|Applebot\-Extended|Bytespider|CCBot|ChatGPT\-User|Claude\-Web|ClaudeBot|cohere\-ai|Diffbot|FacebookBot|facebookexternalhit|FriendlyCrawler|Google\-Extended|GoogleOther|GoogleOther\-Image|GoogleOther\-Video|GPTBot|iaskspider/2\.0|ICC\-Crawler|ImagesiftBot|img2dataset|ISSCyberRiskCrawler|Kangaroo\ Bot|Meta\-ExternalAgent|Meta\-ExternalFetcher|OAI\-SearchBot|omgili|omgilibot|Perplexity\-User|PerplexityBot|PetalBot|Scrapy|Sidetrade\ indexer\ bot|Timpibot|VelenPublicWebCrawler|Webzio\-Extended|YouBot|crawler\.with\.dots|star\*\*\*crawler|Is\ this\ a\ crawler\?|a\[mazing\]\{42\}\(robot\)|2\^32\$|curl\|sudo\ bash)$)" { url.access-deny = ( "" ) } }
\ No newline at end of file
+$HTTP["url"] != "/robots.txt" { $HTTP["user-agent"] =~ "\b(AI2Bot|Ai2Bot-Dolma|Amazonbot|anthropic-ai|Applebot|Applebot-Extended|Bytespider|CCBot|ChatGPT-User|Claude-Web|ClaudeBot|cohere-ai|Diffbot|FacebookBot|facebookexternalhit|FriendlyCrawler|Google-Extended|GoogleOther|GoogleOther-Image|GoogleOther-Video|GPTBot|iaskspider/2\.0|ICC-Crawler|ImagesiftBot|img2dataset|ISSCyberRiskCrawler|Kangaroo\ Bot|Meta-ExternalAgent|Meta-ExternalFetcher|OAI-SearchBot|omgili|omgilibot|Perplexity-User|PerplexityBot|PetalBot|Scrapy|Sidetrade\ indexer\ bot|Timpibot|VelenPublicWebCrawler|Webzio-Extended|YouBot|crawler\.with\.dots|star\*\*\*crawler|Is\ this\ a\ crawler\?|a\[mazing\]\{42\}\(robot\)|2\^32\$|curl\|sudo\ bash)\b" { url.access-deny = ( "" ) } }
diff --git a/code/test_files/nginx-block-ai-bots.conf b/code/test_files/nginx-block-ai-bots.conf
index d0d47a1..e2e3415 100644
--- a/code/test_files/nginx-block-ai-bots.conf
+++ b/code/test_files/nginx-block-ai-bots.conf
@@ -1,10 +1,10 @@
set $block 0;
-if ($http_user_agent ~* "(^(AI2Bot|Ai2Bot\-Dolma|Amazonbot|anthropic\-ai|Applebot|Applebot\-Extended|Bytespider|CCBot|ChatGPT\-User|Claude\-Web|ClaudeBot|cohere\-ai|Diffbot|FacebookBot|facebookexternalhit|FriendlyCrawler|Google\-Extended|GoogleOther|GoogleOther\-Image|GoogleOther\-Video|GPTBot|iaskspider/2\.0|ICC\-Crawler|ImagesiftBot|img2dataset|ISSCyberRiskCrawler|Kangaroo\ Bot|Meta\-ExternalAgent|Meta\-ExternalFetcher|OAI\-SearchBot|omgili|omgilibot|Perplexity\-User|PerplexityBot|PetalBot|Scrapy|Sidetrade\ indexer\ bot|Timpibot|VelenPublicWebCrawler|Webzio\-Extended|YouBot|crawler\.with\.dots|star\*\*\*crawler|Is\ this\ a\ crawler\?|a\[mazing\]\{42\}\(robot\)|2\^32\$|curl\|sudo\ bash)$)") {
+if ($http_user_agent ~ '\\b(AI2Bot|Ai2Bot-Dolma|Amazonbot|anthropic-ai|Applebot|Applebot-Extended|Bytespider|CCBot|ChatGPT-User|Claude-Web|ClaudeBot|cohere-ai|Diffbot|FacebookBot|facebookexternalhit|FriendlyCrawler|Google-Extended|GoogleOther|GoogleOther-Image|GoogleOther-Video|GPTBot|iaskspider/2\\.0|ICC-Crawler|ImagesiftBot|img2dataset|ISSCyberRiskCrawler|Kangaroo\\ Bot|Meta-ExternalAgent|Meta-ExternalFetcher|OAI-SearchBot|omgili|omgilibot|Perplexity-User|PerplexityBot|PetalBot|Scrapy|Sidetrade\\ indexer\\ bot|Timpibot|VelenPublicWebCrawler|Webzio-Extended|YouBot|crawler\\.with\\.dots|star\\*\\*\\*crawler|Is\\ this\\ a\\ crawler\\?|a\\[mazing\\]\\{42\\}\\(robot\\)|2\\^32\\$|curl\\|sudo\\ bash)\\b') {
set $block 1;
}
-if ($request_uri = "/robots.txt") {
+if ($request_uri = '/robots.txt') {
set $block 0;
}
diff --git a/code/tests.py b/code/tests.py
index 53079cf..0006f5d 100755
--- a/code/tests.py
+++ b/code/tests.py
@@ -28,7 +28,7 @@ class RobotsUnittestExtensions:
with open(f, "rt") as f:
f_contents = f.read()
- return self.assertMultiLineEqual(f_contents, s)
+ return self.assertMultiLineEqual(f_contents.rstrip("\r\n"), s.rstrip("\r\n"))
class TestRobotsTXTGeneration(unittest.TestCase, RobotsUnittestExtensions):
@@ -65,25 +65,80 @@ class TestHtaccessGeneration(unittest.TestCase, RobotsUnittestExtensions):
class TestUserAgentPatternGeneration(unittest.TestCase):
- def test_agents_match_only_the_complete_user_agent_by_default(self):
+ def test_agents_match_user_agents_by_prefix_or_substring(self):
pattern = re.compile(
list_to_pcre({"Spider": {}, "ExampleBot": {}}), re.IGNORECASE
)
self.assertIsNotNone(pattern.search("Spider"))
self.assertIsNotNone(pattern.search("spider"))
- self.assertIsNone(pattern.search("Baiduspider"))
- self.assertIsNone(pattern.search("OurCompanyName Test Spider"))
- self.assertIsNone(pattern.search("Mozilla/5.0 ExampleBot/1.0"))
+ self.assertIsNotNone(pattern.search("Mozilla/5.0 ExampleBot/1.0"))
- def test_name_and_version_agents_match_versioned_tokens(self):
- pattern = re.compile(
- list_to_pcre({"Code": {"has_name_and_version": True}}), re.IGNORECASE
- )
+ def test_generated_regex_against_real_user_agents(self):
+ from pathlib import Path
+ robots_json_path = Path(__file__).parent.parent / "robots.json"
+ if robots_json_path.exists():
+ with open(robots_json_path, "rt", encoding="utf-8") as f:
+ robots_dict = json.load(f)
+ else:
+ robots_dict = self.loadJson("test_files/robots.json")
+
+ pattern = re.compile(list_to_pcre(robots_dict), re.IGNORECASE)
+
+ user_agents = [
+ "CCBot/2.0 (https://commoncrawl.org/faq/)",
+ "Claude-User (claude-code/2.1.220; +https://support.anthropic.com/)",
+ "facebookexternalhit/1.1 (+http://www.facebook.com/externalhit_uatext.php)",
+ "meta-externalagent/1.1 (+https://developers.facebook.com/docs/sharing/webmasters/crawler)",
+ "Scrapy/2.16.0 (+https://scrapy.org)",
+ ]
+
+ for ua in user_agents:
+ with self.subTest(user_agent=ua):
+ self.assertIsNotNone(pattern.search(ua))
+
+ def test_generated_regex_does_not_match_non_ai_user_agents(self):
+ from pathlib import Path
+ robots_json_path = Path(__file__).parent.parent / "robots.json"
+ if robots_json_path.exists():
+ with open(robots_json_path, "rt", encoding="utf-8") as f:
+ robots_dict = json.load(f)
+ else:
+ robots_dict = self.loadJson("test_files/robots.json")
+
+ pattern = re.compile(list_to_pcre(robots_dict), re.IGNORECASE)
+
+ non_ai_user_agents = [
+ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
+ "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.2.1 Safari/605.1.15",
+ "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:121.0) Gecko/20100101 Firefox/121.0",
+ "Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)",
+ "Mozilla/5.0 (compatible; Bingbot/2.0; +http://www.bing.com/bingbot.htm)",
+ "curl/7.68.0",
+ "Wget/1.20.3 (linux-gnu)",
+ "NotCursor/1.0",
+ "CursorNot/1.0",
+ "NotScrapy/2.0",
+ "ScrapyNot/2.0",
+ "NotClaude/1.0",
+ "ClaudeNot/1.0",
+ "NotPerplexity/1.0",
+ "NotAmazonbot/1.0",
+ "NotApplebot/1.0",
+ "NotBytespider/1.0",
+ # Real-world user agents that embed a listed bot name mid-string.
+ # These older Internet Explorer / Trident agents contain "SLCC1" or
+ # "SLCC2" (a Windows licensing component), which spans the listed
+ # agent "LCC". Reported in #208 and fixed by the word boundaries
+ # added in #260; pinned here so the specific report cannot regress.
+ "Mozilla/4.0 (compatible; MSIE 7.0; Windows NT 6.0; SLCC1; .NET CLR 2.0.50727; Media Center PC 5.0; .NET CLR 3.0.30729)",
+ "Mozilla/4.0 (compatible; MSIE 8.0; Windows NT 6.1; Trident/4.0; SLCC2; .NET CLR 2.0.50727; Media Center PC 6.0)",
+ ]
+
+ for ua in non_ai_user_agents:
+ with self.subTest(user_agent=ua):
+ self.assertIsNone(pattern.search(ua))
- self.assertIsNotNone(pattern.search("Code"))
- self.assertIsNotNone(pattern.search("Mozilla/5.0 Code/1.2.3"))
- self.assertIsNone(pattern.search("https://codeberg.org/example"))
class TestNginxConfigGeneration(unittest.TestCase, RobotsUnittestExtensions):
maxDiff = 8192
diff --git a/haproxy-block-ai-bots.txt b/haproxy-block-ai-bots.txt
index 331b031..12e9582 100644
--- a/haproxy-block-ai-bots.txt
+++ b/haproxy-block-ai-bots.txt
@@ -53,10 +53,12 @@ Datenbank Crawler
DeepSeekBot
Devin
Diffbot
+Diffbot-User
DuckAssistBot
Echobot Bot
EchoboxBot
ExaBot
+ExaSearchBot
FacebookBot
facebookexternalhit
Factset_spyderbot
@@ -80,7 +82,6 @@ HenkBot
iAskBot
iaskspider
iaskspider/2.0
-IbouBot
ICC-Crawler
ImagesiftBot
imageSpider
@@ -88,12 +89,15 @@ img2dataset
ISSCyberRiskCrawler
kagi-fetcher
Kangaroo Bot
+Kimi-SearchBot
Kimi-User
+KimiBot
KlaviyoAIBot
KunatoCrawler
laion-huggingface-processor
LAIONDownloader
LCC
+Lightpanda
LinerBot
Linguee Bot
LinkupBot
@@ -103,6 +107,7 @@ Meta-ExternalAgent
meta-externalfetcher
Meta-ExternalFetcher
meta-webindexer
+MistralAI-Training
MistralAI-User
MistralAI-User/1.0
Mozilla-Tabstack
@@ -112,6 +117,7 @@ netEstate Imprint Crawler
newsai
NotebookLM
NovaAct
+OAI-AdsBot
OAI-SearchBot
omgili
omgilibot
@@ -132,6 +138,7 @@ Querit-SearchBot
QueritBot
QuillBot
quillbot.com
+Reflectionbot
SBIntuitionsBot
Scrapy
SemrushBot-OCOB
diff --git a/lighttpd-block-ai-bots.conf b/lighttpd-block-ai-bots.conf
index 470b3da..f7cf8b7 100644
--- a/lighttpd-block-ai-bots.conf
+++ b/lighttpd-block-ai-bots.conf
@@ -1 +1 @@
-$HTTP["url"] != "/robots.txt" { $HTTP["user-agent"] =~ "(^(AddSearchBot|AgentTimes|AI2Bot|AI2Bot\-DeepResearchEval|Ai2Bot\-Dolma|aiHitBot|AIWebIndex|amazon\-kendra|amazon\-QBusiness|Amazonbot|AmazonBuyForMe|Amzn\-SearchBot|Amzn\-User|Andibot|Anomura|anthropic\-ai|ApifyBot|ApifyWebsiteContentCrawler|Applebot|Applebot\-Extended|Aranet\-SearchBot|atlassian\-bot|Awario|AzureAI\-SearchBot|bedrockbot|bigsur\.ai|Bravebot|Brightbot|Brightbot\ 1\.0|BuddyBot|Bytespider|CCBot|Channel3Bot|ChatGLM\-Spider|ChatGPT\ Agent|ChatGPT\-User|Claude\-Code|Claude\-SearchBot|Claude\-User|Claude\-Web|ClaudeBot|Cloudflare\-AutoRAG|CloudVertexBot|Code|cohere\-ai|cohere\-training\-data\-crawler|Cotoyogi|CragCrawler|Crawl4AI|Crawlspace|Cursor|Datenbank\ Crawler|DeepSeekBot|Devin|Diffbot|DuckAssistBot|Echobot\ Bot|EchoboxBot|ExaBot|FacebookBot|facebookexternalhit|Factset_spyderbot|FirecrawlAgent|FriendlyCrawler|GeistHaus\-PageFetcher|Gemini\-Deep\-Research|Google\-Agent|Google\-CloudVertexBot|Google\-Extended|Google\-Firebase|Google\-Gemini\-CLI|Google\-NotebookLM|GoogleAgent\-Mariner|GoogleAgent\-URLContext|GoogleOther|GoogleOther\-Image|GoogleOther\-Video|GPTBot|HenkBot|iAskBot|iaskspider|iaskspider/2\.0|IbouBot|ICC\-Crawler|ImagesiftBot|imageSpider|img2dataset|ISSCyberRiskCrawler|kagi\-fetcher|Kangaroo\ Bot|Kimi\-User|KlaviyoAIBot|KunatoCrawler|laion\-huggingface\-processor|LAIONDownloader|LCC|LinerBot|Linguee\ Bot|LinkupBot|Manus\-User|meta\-externalagent|Meta\-ExternalAgent|meta\-externalfetcher|Meta\-ExternalFetcher|meta\-webindexer|MistralAI\-User|MistralAI\-User/1\.0|Mozilla\-Tabstack|MyCentralAIScraperBot|NagetBot|netEstate\ Imprint\ Crawler|newsai|NotebookLM|NovaAct|OAI\-SearchBot|omgili|omgilibot|OpenAI|opencode|Operator|PanguBot|Panscient|panscient\.com|Perplexity\-User|PerplexityBot|PetalBot|PhindBot|Poggio\-Citations|Poseidon\ Research\ Crawler|QualifiedBot|Querit\-SearchBot|QueritBot|QuillBot|quillbot\.com|SBIntuitionsBot|Scrapy|SemrushBot\-OCOB|SemrushBot\-SWA|Shap\-User|ShapBot|Sidetrade\ indexer\ bot|Spider|TavilyBot|Terra\ Cotta|TerraCotta|Thinkbot|TikTokSpider|Timpibot|TongyiBot|Trae|TwinAgent|UseAI|VelenPublicWebCrawler|WARDBot|Webzio\-Extended|webzio\-extended|wpbot|WRTNBot|YaK|YandexAdditional|YandexAdditionalBot|YiyanBot|YouBot|ZanistaBot)$|Code/[0-9.]+)" { url.access-deny = ( "" ) } }
\ No newline at end of file
+$HTTP["url"] != "/robots.txt" { $HTTP["user-agent"] =~ "\b(AddSearchBot|AgentTimes|AI2Bot|AI2Bot-DeepResearchEval|Ai2Bot-Dolma|aiHitBot|AIWebIndex|amazon-kendra|amazon-QBusiness|Amazonbot|AmazonBuyForMe|Amzn-SearchBot|Amzn-User|Andibot|Anomura|anthropic-ai|ApifyBot|ApifyWebsiteContentCrawler|Applebot|Applebot-Extended|Aranet-SearchBot|atlassian-bot|Awario|AzureAI-SearchBot|bedrockbot|bigsur\.ai|Bravebot|Brightbot|Brightbot\ 1\.0|BuddyBot|Bytespider|CCBot|Channel3Bot|ChatGLM-Spider|ChatGPT\ Agent|ChatGPT-User|Claude-Code|Claude-SearchBot|Claude-User|Claude-Web|ClaudeBot|Cloudflare-AutoRAG|CloudVertexBot|Code|cohere-ai|cohere-training-data-crawler|Cotoyogi|CragCrawler|Crawl4AI|Crawlspace|Cursor|Datenbank\ Crawler|DeepSeekBot|Devin|Diffbot|Diffbot-User|DuckAssistBot|Echobot\ Bot|EchoboxBot|ExaBot|ExaSearchBot|FacebookBot|facebookexternalhit|Factset_spyderbot|FirecrawlAgent|FriendlyCrawler|GeistHaus-PageFetcher|Gemini-Deep-Research|Google-Agent|Google-CloudVertexBot|Google-Extended|Google-Firebase|Google-Gemini-CLI|Google-NotebookLM|GoogleAgent-Mariner|GoogleAgent-URLContext|GoogleOther|GoogleOther-Image|GoogleOther-Video|GPTBot|HenkBot|iAskBot|iaskspider|iaskspider/2\.0|ICC-Crawler|ImagesiftBot|imageSpider|img2dataset|ISSCyberRiskCrawler|kagi-fetcher|Kangaroo\ Bot|Kimi-SearchBot|Kimi-User|KimiBot|KlaviyoAIBot|KunatoCrawler|laion-huggingface-processor|LAIONDownloader|LCC|Lightpanda|LinerBot|Linguee\ Bot|LinkupBot|Manus-User|meta-externalagent|Meta-ExternalAgent|meta-externalfetcher|Meta-ExternalFetcher|meta-webindexer|MistralAI-Training|MistralAI-User|MistralAI-User/1\.0|Mozilla-Tabstack|MyCentralAIScraperBot|NagetBot|netEstate\ Imprint\ Crawler|newsai|NotebookLM|NovaAct|OAI-AdsBot|OAI-SearchBot|omgili|omgilibot|OpenAI|opencode|Operator|PanguBot|Panscient|panscient\.com|Perplexity-User|PerplexityBot|PetalBot|PhindBot|Poggio-Citations|Poseidon\ Research\ Crawler|QualifiedBot|Querit-SearchBot|QueritBot|QuillBot|quillbot\.com|Reflectionbot|SBIntuitionsBot|Scrapy|SemrushBot-OCOB|SemrushBot-SWA|Shap-User|ShapBot|Sidetrade\ indexer\ bot|Spider|TavilyBot|Terra\ Cotta|TerraCotta|Thinkbot|TikTokSpider|Timpibot|TongyiBot|Trae|TwinAgent|UseAI|VelenPublicWebCrawler|WARDBot|Webzio-Extended|webzio-extended|wpbot|WRTNBot|YaK|YandexAdditional|YandexAdditionalBot|YiyanBot|YouBot|ZanistaBot)\b" { url.access-deny = ( "" ) } }
\ No newline at end of file
diff --git a/nginx-block-ai-bots.conf b/nginx-block-ai-bots.conf
index e44c747..be0bd75 100644
--- a/nginx-block-ai-bots.conf
+++ b/nginx-block-ai-bots.conf
@@ -1,10 +1,10 @@
set $block 0;
-if ($http_user_agent ~* "(^(AddSearchBot|AgentTimes|AI2Bot|AI2Bot\-DeepResearchEval|Ai2Bot\-Dolma|aiHitBot|AIWebIndex|amazon\-kendra|amazon\-QBusiness|Amazonbot|AmazonBuyForMe|Amzn\-SearchBot|Amzn\-User|Andibot|Anomura|anthropic\-ai|ApifyBot|ApifyWebsiteContentCrawler|Applebot|Applebot\-Extended|Aranet\-SearchBot|atlassian\-bot|Awario|AzureAI\-SearchBot|bedrockbot|bigsur\.ai|Bravebot|Brightbot|Brightbot\ 1\.0|BuddyBot|Bytespider|CCBot|Channel3Bot|ChatGLM\-Spider|ChatGPT\ Agent|ChatGPT\-User|Claude\-Code|Claude\-SearchBot|Claude\-User|Claude\-Web|ClaudeBot|Cloudflare\-AutoRAG|CloudVertexBot|Code|cohere\-ai|cohere\-training\-data\-crawler|Cotoyogi|CragCrawler|Crawl4AI|Crawlspace|Cursor|Datenbank\ Crawler|DeepSeekBot|Devin|Diffbot|DuckAssistBot|Echobot\ Bot|EchoboxBot|ExaBot|FacebookBot|facebookexternalhit|Factset_spyderbot|FirecrawlAgent|FriendlyCrawler|GeistHaus\-PageFetcher|Gemini\-Deep\-Research|Google\-Agent|Google\-CloudVertexBot|Google\-Extended|Google\-Firebase|Google\-Gemini\-CLI|Google\-NotebookLM|GoogleAgent\-Mariner|GoogleAgent\-URLContext|GoogleOther|GoogleOther\-Image|GoogleOther\-Video|GPTBot|HenkBot|iAskBot|iaskspider|iaskspider/2\.0|IbouBot|ICC\-Crawler|ImagesiftBot|imageSpider|img2dataset|ISSCyberRiskCrawler|kagi\-fetcher|Kangaroo\ Bot|Kimi\-User|KlaviyoAIBot|KunatoCrawler|laion\-huggingface\-processor|LAIONDownloader|LCC|LinerBot|Linguee\ Bot|LinkupBot|Manus\-User|meta\-externalagent|Meta\-ExternalAgent|meta\-externalfetcher|Meta\-ExternalFetcher|meta\-webindexer|MistralAI\-User|MistralAI\-User/1\.0|Mozilla\-Tabstack|MyCentralAIScraperBot|NagetBot|netEstate\ Imprint\ Crawler|newsai|NotebookLM|NovaAct|OAI\-SearchBot|omgili|omgilibot|OpenAI|opencode|Operator|PanguBot|Panscient|panscient\.com|Perplexity\-User|PerplexityBot|PetalBot|PhindBot|Poggio\-Citations|Poseidon\ Research\ Crawler|QualifiedBot|Querit\-SearchBot|QueritBot|QuillBot|quillbot\.com|SBIntuitionsBot|Scrapy|SemrushBot\-OCOB|SemrushBot\-SWA|Shap\-User|ShapBot|Sidetrade\ indexer\ bot|Spider|TavilyBot|Terra\ Cotta|TerraCotta|Thinkbot|TikTokSpider|Timpibot|TongyiBot|Trae|TwinAgent|UseAI|VelenPublicWebCrawler|WARDBot|Webzio\-Extended|webzio\-extended|wpbot|WRTNBot|YaK|YandexAdditional|YandexAdditionalBot|YiyanBot|YouBot|ZanistaBot)$|Code/[0-9.]+)") {
+if ($http_user_agent ~ '\\b(AddSearchBot|AgentTimes|AI2Bot|AI2Bot-DeepResearchEval|Ai2Bot-Dolma|aiHitBot|AIWebIndex|amazon-kendra|amazon-QBusiness|Amazonbot|AmazonBuyForMe|Amzn-SearchBot|Amzn-User|Andibot|Anomura|anthropic-ai|ApifyBot|ApifyWebsiteContentCrawler|Applebot|Applebot-Extended|Aranet-SearchBot|atlassian-bot|Awario|AzureAI-SearchBot|bedrockbot|bigsur\\.ai|Bravebot|Brightbot|Brightbot\\ 1\\.0|BuddyBot|Bytespider|CCBot|Channel3Bot|ChatGLM-Spider|ChatGPT\\ Agent|ChatGPT-User|Claude-Code|Claude-SearchBot|Claude-User|Claude-Web|ClaudeBot|Cloudflare-AutoRAG|CloudVertexBot|Code|cohere-ai|cohere-training-data-crawler|Cotoyogi|CragCrawler|Crawl4AI|Crawlspace|Cursor|Datenbank\\ Crawler|DeepSeekBot|Devin|Diffbot|Diffbot-User|DuckAssistBot|Echobot\\ Bot|EchoboxBot|ExaBot|ExaSearchBot|FacebookBot|facebookexternalhit|Factset_spyderbot|FirecrawlAgent|FriendlyCrawler|GeistHaus-PageFetcher|Gemini-Deep-Research|Google-Agent|Google-CloudVertexBot|Google-Extended|Google-Firebase|Google-Gemini-CLI|Google-NotebookLM|GoogleAgent-Mariner|GoogleAgent-URLContext|GoogleOther|GoogleOther-Image|GoogleOther-Video|GPTBot|HenkBot|iAskBot|iaskspider|iaskspider/2\\.0|ICC-Crawler|ImagesiftBot|imageSpider|img2dataset|ISSCyberRiskCrawler|kagi-fetcher|Kangaroo\\ Bot|Kimi-SearchBot|Kimi-User|KimiBot|KlaviyoAIBot|KunatoCrawler|laion-huggingface-processor|LAIONDownloader|LCC|Lightpanda|LinerBot|Linguee\\ Bot|LinkupBot|Manus-User|meta-externalagent|Meta-ExternalAgent|meta-externalfetcher|Meta-ExternalFetcher|meta-webindexer|MistralAI-Training|MistralAI-User|MistralAI-User/1\\.0|Mozilla-Tabstack|MyCentralAIScraperBot|NagetBot|netEstate\\ Imprint\\ Crawler|newsai|NotebookLM|NovaAct|OAI-AdsBot|OAI-SearchBot|omgili|omgilibot|OpenAI|opencode|Operator|PanguBot|Panscient|panscient\\.com|Perplexity-User|PerplexityBot|PetalBot|PhindBot|Poggio-Citations|Poseidon\\ Research\\ Crawler|QualifiedBot|Querit-SearchBot|QueritBot|QuillBot|quillbot\\.com|Reflectionbot|SBIntuitionsBot|Scrapy|SemrushBot-OCOB|SemrushBot-SWA|Shap-User|ShapBot|Sidetrade\\ indexer\\ bot|Spider|TavilyBot|Terra\\ Cotta|TerraCotta|Thinkbot|TikTokSpider|Timpibot|TongyiBot|Trae|TwinAgent|UseAI|VelenPublicWebCrawler|WARDBot|Webzio-Extended|webzio-extended|wpbot|WRTNBot|YaK|YandexAdditional|YandexAdditionalBot|YiyanBot|YouBot|ZanistaBot)\\b') {
set $block 1;
}
-if ($request_uri = "/robots.txt") {
+if ($request_uri = '/robots.txt') {
set $block 0;
}
diff --git a/robots.json b/robots.json
index 7e220d3..bc9732f 100644
--- a/robots.json
+++ b/robots.json
@@ -42,11 +42,11 @@
"description": "Scrapes data for AI systems."
},
"AIWebIndex": {
- "operator": "Lyrenth that builds an AI-readable index of web content for AI systems",
- "respect": "Unclear at this time.",
- "function": "AI Data Providers",
- "frequency": "Unclear at this time.",
- "description": "AIWebIndex is a web crawler operated by Lyrenth that builds an AI-readable index of web content for AI systems. More info can be found at https://knownagents.com/agents/aiwebindex"
+ "operator": "[Lyrenth](https://lyrenth.com)",
+ "respect": "[Yes](https://lyrenth.com/crawler-policy)",
+ "function": "AI Search Crawlers",
+ "frequency": "At most one request per domain every 2 seconds, and slower where robots.txt sets a longer Crawl-delay.",
+ "description": "Builds an index of public pages and serves them to AI agents as extracted, readable text with attribution and a link back to the source. Does not train foundation models on crawled content. Identity can be checked three ways: published IP ranges at https://lyrenth.com/bot/ip-ranges.json, forward-confirmed reverse DNS under lyrenth.com, and Web Bot Auth signatures (RFC 9421). Full policy at https://lyrenth.com/crawler-policy"
},
"amazon-kendra": {
"operator": "Amazon",
@@ -127,7 +127,7 @@
},
"Applebot": {
"operator": "Unclear at this time.",
- "respect": "Unclear at this time.",
+ "respect": "[Yes](https://support.apple.com/en-us/119829#retrieval)",
"function": "AI Search Crawlers",
"frequency": "Unclear at this time.",
"description": "Applebot is a web crawler used by Apple to index search results that allow the Siri AI Assistant to answer user questions. Siri's answers normally contain references to the website. More info can be found at https://knownagents.com/agents/applebot"
@@ -385,9 +385,16 @@
"frequency": "Unclear at this time.",
"description": "Diffbot is a web crawler that extracts and structures website content using AI-powered visual understanding, providing knowledge graph data for applications like market i\u2026 More info can be found at https://knownagents.com/agents/diffbot"
},
+ "Diffbot-User": {
+ "operator": "[Diffbot](https://www.diffbot.com/)",
+ "respect": "Yes",
+ "function": "AI Assistants",
+ "frequency": "Only when prompted by a user.",
+ "description": "Diffbot-User is used by requests originating on behalf of a human user browsing a URL using Diffbot software, in response to their input. Documented by Diffbot at https://docs.diffbot.com/docs/does-crawl-respect-robotstxt"
+ },
"DuckAssistBot": {
"operator": "Unclear at this time.",
- "respect": "Unclear at this time.",
+ "respect": "[Yes](https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot/)",
"function": "AI Assistants",
"frequency": "Unclear at this time.",
"description": "DuckAssistBot is a web crawler that scans websites to collect content for DuckDuckGo's AI-assisted answers feature, which generates brief responses to search queries usin\u2026 More info can be found at https://knownagents.com/agents/duckassistbot"
@@ -413,6 +420,13 @@
"frequency": "Unclear at this time.",
"description": "ExaBot is a web crawler that indexes web content to power Exa's AI search engine and semantic search APIs for AI applications. More info can be found at https://knownagents.com/agents/exabot"
},
+ "ExaSearchBot": {
+ "operator": "[Exa](https://exa.ai)",
+ "respect": "Unclear at this time.",
+ "function": "AI Search Crawlers",
+ "frequency": "Unclear at this time.",
+ "description": "ExaSearchBot is a web crawler operated by Exa that discovers and indexes public web pages so their content can be found, retrieved, and cited through Exa. More info can be found at https://knownagents.com/agents/exasearchbot"
+ },
"FacebookBot": {
"operator": "Meta/Facebook",
"respect": "[Yes](https://developers.facebook.com/docs/sharing/bot/)",
@@ -464,7 +478,7 @@
},
"Google-Agent": {
"operator": "Unclear at this time.",
- "respect": "Unclear at this time.",
+ "respect": "[Yes](https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers#google-agent)",
"function": "AI Agents",
"frequency": "Unclear at this time.",
"description": "Google-Agent is used by agents hosted on Google infrastructure to navigate the web and perform actions upon user request. More info can be found at https://knownagents.com/agents/google-agent"
@@ -574,13 +588,6 @@
"operator": "iAsk",
"respect": "No"
},
- "IbouBot": {
- "operator": "Ibou",
- "respect": "Yes",
- "function": "Search result generation.",
- "frequency": "Unclear at this time.",
- "description": "Ibou.io operates a crawler service named IbouBot which fuels and updates their graph representation of the World Wide Web. This database and all the metrics are used to provide a search engine."
- },
"ICC-Crawler": {
"operator": "[NICT](https://nict.go.jp)",
"respect": "Yes",
@@ -630,6 +637,13 @@
"frequency": "Unclear at this time.",
"description": "Kangaroo Bot is used by the company Kangaroo LLM to download data to train AI models tailored to Australian language and culture. More info can be found at https://knownagents.com/agents/kangaroo-bot"
},
+ "Kimi-SearchBot": {
+ "operator": "[Moonshot AI](https://www.moonshot.ai)",
+ "respect": "[Yes](https://www.kimi.ai/policies/kimi-crawlers)",
+ "function": "AI Search Crawlers",
+ "frequency": "No information provided.",
+ "description": "Kimi-SearchBot powers Kimi's search features: it analyzes pages for relevance and builds the search index. Documented by Moonshot AI at https://www.kimi.ai/policies/kimi-crawlers"
+ },
"Kimi-User": {
"operator": "Moonshot AI that fetches web content on behalf of users interacting with Kimi",
"respect": "Unclear at this time.",
@@ -637,6 +651,13 @@
"frequency": "Unclear at this time.",
"description": "Kimi-User is a web crawler operated by Moonshot AI that fetches web content on behalf of users interacting with Kimi. When a user asks Kimi to summarize an article or ans\u2026 More info can be found at https://knownagents.com/agents/kimi-user"
},
+ "KimiBot": {
+ "operator": "[Moonshot AI](https://www.moonshot.ai)",
+ "respect": "[Yes](https://www.kimi.ai/policies/kimi-crawlers)",
+ "function": "AI Data Scrapers",
+ "frequency": "No information provided.",
+ "description": "KimiBot crawls content potentially used to train Kimi's foundation models. Documented by Moonshot AI at https://www.kimi.ai/policies/kimi-crawlers"
+ },
"KlaviyoAIBot": {
"operator": "[Klaviyo](https://www.klaviyo.com)",
"respect": "[Yes](https://help.klaviyo.com/hc/en-us/articles/40496146232219)",
@@ -672,6 +693,13 @@
"frequency": "Unclear at this time.",
"description": "Description unavailable from knownagents.com More info can be found at https://knownagents.com/agents/lcc"
},
+ "Lightpanda": {
+ "operator": "Anyone who downloads the Lightpanda client. Possibly being used by a [Grok-adjacent](https://github.com/lightpanda-io/browser/issues/3156#issuecomment-5217843616) organization's botnet.",
+ "respect": "At the [discretion](https://github.com/lightpanda-io/browser/blob/b04c99a9111564ebe06317f644680eda5e3ee83e/src/help.zon#L385) of Lightpanda users.",
+ "function": "AI Data Scrapers",
+ "frequency": "Defined per-user.",
+ "description": "Lightpanda is a custom-built headless browser designed for AI and automation."
+ },
"LinerBot": {
"operator": "Unclear at this time.",
"respect": "Unclear at this time.",
@@ -709,21 +737,21 @@
},
"Meta-ExternalAgent": {
"operator": "Unclear at this time.",
- "respect": "Unclear at this time.",
+ "respect": "[Yes](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/)",
"function": "AI Data Scrapers",
"frequency": "Unclear at this time.",
"description": "Meta-ExternalAgent is a web crawler used by Meta to download training data for its AI models and improve its products by indexing content directly. More info can be found at https://knownagents.com/agents/meta-externalagent"
},
"meta-externalfetcher": {
"operator": "Unclear at this time.",
- "respect": "Unclear at this time.",
+ "respect": "[No](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/)",
"function": "AI Assistants",
"frequency": "Unclear at this time.",
"description": "meta-externalfetcher is used by Meta to perform user-initiated fetches of individual links from AI assistant product functions. More info can be found at https://knownagents.com/agents/meta-externalfetcher"
},
"Meta-ExternalFetcher": {
"operator": "Unclear at this time.",
- "respect": "Unclear at this time.",
+ "respect": "[No](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/)",
"function": "AI Assistants",
"frequency": "Unclear at this time.",
"description": "Meta-ExternalFetcher is dispatched by Meta AI products in response to user prompts, when they need to fetch an individual links. More info can be found at https://knownagents.com/agents/meta-externalfetcher"
@@ -735,6 +763,13 @@
"frequency": "Unhinged, more than 1 per second.",
"description": "As per their documentation, \"The Meta-WebIndexer crawler navigates the web to improve Meta AI search result quality for users. In doing so, Meta analyzes online content to enhance the relevance and accuracy of Meta AI. Allowing Meta-WebIndexer in your robots.txt file helps us cite and link to your content in Meta AI's responses.\""
},
+ "MistralAI-Training": {
+ "operator": "[Mistral AI](https://mistral.ai)",
+ "respect": "[Yes](https://docs.mistral.ai/robots/)",
+ "function": "AI Data Scrapers",
+ "frequency": "No information provided.",
+ "description": "MistralAI-Training crawls web content to build training datasets. Documented by Mistral at https://docs.mistral.ai/robots/"
+ },
"MistralAI-User": {
"operator": "Mistral",
"respect": "Unclear at this time.",
@@ -750,11 +785,11 @@
"respect": "Yes"
},
"Mozilla-Tabstack": {
- "operator": "[Mozilla](https://docs.tabstack.ai/trust/controlling-access)",
+ "operator": "Mozilla that performs programmatic, AI-driven interactions with web content through Tabstack",
"respect": "Yes",
"function": "AI Data Providers",
"frequency": "On demand via API.",
- "description": "Tabstack is a web intelligence API for AI agents. It extracts structured data from web pages and makes it available to AI agents."
+ "description": "Mozilla-Tabstack is an AI agent operated by Mozilla that performs programmatic, AI-driven interactions with web content through Tabstack. More info can be found at https://knownagents.com/agents/mozilla-tabstack"
},
"MyCentralAIScraperBot": {
"operator": "Unclear at this time.",
@@ -798,6 +833,13 @@
"frequency": "Unclear at this time.",
"description": "Nova Act is an AI agent created by Amazon that can use a web browser. It can intelligently navigate and interact with websites to complete multi-step tasks on behalf of a\u2026 More info can be found at https://knownagents.com/agents/novaact"
},
+ "OAI-AdsBot": {
+ "operator": "[OpenAI](https://openai.com)",
+ "respect": "Unclear at this time.",
+ "function": "Validates and targets ads on ChatGPT.",
+ "frequency": "Only when a page is submitted as an ad.",
+ "description": "OAI-AdsBot visits the landing page of a page submitted as an ad on ChatGPT, to check it against OpenAI's policies and, per OpenAI, to \"determine when it's most relevant to show the ad to users\". OpenAI states it only visits pages submitted as ads and that the data is not used to train foundation models. Documented at https://developers.openai.com/api/docs/bots"
+ },
"OAI-SearchBot": {
"operator": "[OpenAI](https://openai.com)",
"respect": "[Yes](https://platform.openai.com/docs/bots)",
@@ -938,6 +980,13 @@
"operator": "[Quillbot](https://quillbot.com)",
"respect": "Unclear at this time."
},
+ "Reflectionbot": {
+ "operator": "[Reflection](https://reflection.ai/)",
+ "respect": "Unclear at this time.",
+ "function": "Undocumented AI Agents",
+ "frequency": "Unclear at this time.",
+ "description": "An undocumented crawler whose user agent links to Reflection, a company that builds AI models."
+ },
"SBIntuitionsBot": {
"operator": "[SB Intuitions](https://www.sbintuitions.co.jp/en/)",
"respect": "[Yes](https://www.sbintuitions.co.jp/en/bot/)",
diff --git a/robots.txt b/robots.txt
index 7933b60..a7ed28d 100644
--- a/robots.txt
+++ b/robots.txt
@@ -53,10 +53,12 @@ User-agent: Datenbank Crawler
User-agent: DeepSeekBot
User-agent: Devin
User-agent: Diffbot
+User-agent: Diffbot-User
User-agent: DuckAssistBot
User-agent: Echobot Bot
User-agent: EchoboxBot
User-agent: ExaBot
+User-agent: ExaSearchBot
User-agent: FacebookBot
User-agent: facebookexternalhit
User-agent: Factset_spyderbot
@@ -80,7 +82,6 @@ User-agent: HenkBot
User-agent: iAskBot
User-agent: iaskspider
User-agent: iaskspider/2.0
-User-agent: IbouBot
User-agent: ICC-Crawler
User-agent: ImagesiftBot
User-agent: imageSpider
@@ -88,12 +89,15 @@ User-agent: img2dataset
User-agent: ISSCyberRiskCrawler
User-agent: kagi-fetcher
User-agent: Kangaroo Bot
+User-agent: Kimi-SearchBot
User-agent: Kimi-User
+User-agent: KimiBot
User-agent: KlaviyoAIBot
User-agent: KunatoCrawler
User-agent: laion-huggingface-processor
User-agent: LAIONDownloader
User-agent: LCC
+User-agent: Lightpanda
User-agent: LinerBot
User-agent: Linguee Bot
User-agent: LinkupBot
@@ -103,6 +107,7 @@ User-agent: Meta-ExternalAgent
User-agent: meta-externalfetcher
User-agent: Meta-ExternalFetcher
User-agent: meta-webindexer
+User-agent: MistralAI-Training
User-agent: MistralAI-User
User-agent: MistralAI-User/1.0
User-agent: Mozilla-Tabstack
@@ -112,6 +117,7 @@ User-agent: netEstate Imprint Crawler
User-agent: newsai
User-agent: NotebookLM
User-agent: NovaAct
+User-agent: OAI-AdsBot
User-agent: OAI-SearchBot
User-agent: omgili
User-agent: omgilibot
@@ -132,6 +138,7 @@ User-agent: Querit-SearchBot
User-agent: QueritBot
User-agent: QuillBot
User-agent: quillbot.com
+User-agent: Reflectionbot
User-agent: SBIntuitionsBot
User-agent: Scrapy
User-agent: SemrushBot-OCOB
diff --git a/table-of-bot-metrics.md b/table-of-bot-metrics.md
index f7bba9e..8775cad 100644
--- a/table-of-bot-metrics.md
+++ b/table-of-bot-metrics.md
@@ -6,7 +6,7 @@
| AI2Bot\-DeepResearchEval | Ai2, a non-profit AI research institute | Unclear at this time. | AI Assistants | Unclear at this time. | Ai2Bot-DeepResearchEval is operated by Ai2, a non-profit AI research institute. It's used to collect and scan resources used in deep research queries performed by Ai2's o… More info can be found at https://knownagents.com/agents/ai2bot-deepresearcheval |
| Ai2Bot\-Dolma | [Ai2](https://allenai.org/crawler) | Yes | Content is used to train open language models. | No information provided. | Explores 'certain domains' to find web content. |
| aiHitBot | [aiHit](https://www.aihitdata.com/about) | Yes | A massive, artificial intelligence/machine learning, automated system. | No information provided. | Scrapes data for AI systems. |
-| AIWebIndex | Lyrenth that builds an AI-readable index of web content for AI systems | Unclear at this time. | AI Data Providers | Unclear at this time. | AIWebIndex is a web crawler operated by Lyrenth that builds an AI-readable index of web content for AI systems. More info can be found at https://knownagents.com/agents/aiwebindex |
+| AIWebIndex | [Lyrenth](https://lyrenth.com) | [Yes](https://lyrenth.com/crawler-policy) | AI Search Crawlers | At most one request per domain every 2 seconds, and slower where robots.txt sets a longer Crawl-delay. | Builds an index of public pages and serves them to AI agents as extracted, readable text with attribution and a link back to the source. Does not train foundation models on crawled content. Identity can be checked three ways: published IP ranges at https://lyrenth.com/bot/ip-ranges.json, forward-confirmed reverse DNS under lyrenth.com, and Web Bot Auth signatures (RFC 9421). Full policy at https://lyrenth.com/crawler-policy |
| amazon\-kendra | Amazon | Yes | Collects data for AI natural language search | No information provided. | Amazon Kendra is a highly accurate intelligent search service that enables your users to search unstructured data using natural language. It returns specific answers to questions, giving users an experience that's close to interacting with a human expert. It is highly scalable and capable of meeting performance demands, tightly integrated with other AWS services such as Amazon S3 and Amazon Lex, and offers enterprise-grade security. |
| amazon\-QBusiness | Unclear at this time. | Unclear at this time. | AI Assistants | Unclear at this time. | amazon-QBusiness is an Amazon Q Business web crawler that fetches and indexes web content for Amazon Q Business applications. More info can be found at https://knownagents.com/agents/amazon-qbusiness |
| Amazonbot | Amazon | Yes | Service improvement and enabling answers for Alexa users. | No information provided. | Includes references to crawled website when surfacing answers via Alexa; does not clearly outline other uses. |
@@ -18,7 +18,7 @@
| anthropic\-ai | [Anthropic](https://www.anthropic.com) | Unclear at this time. | Scrapes data to train Anthropic's AI products. | No information provided. | Scrapes data to train LLMs and AI products offered by Anthropic. |
| ApifyBot | Unclear at this time. | Unclear at this time. | AI Data Providers | Unclear at this time. | ApifyBot is a web scraping and data extraction crawler by Apify that collects website content for use in AI, LLMs, RAG, and automation workflows. More info can be found at https://knownagents.com/agents/apifybot |
| ApifyWebsiteContentCrawler | Unclear at this time. | Unclear at this time. | AI Data Providers | Unclear at this time. | ApifyWebsiteContentCrawler is a web crawler by Apify that extracts and downloads full website content for use in AI, data analysis, and automation workflows. More info can be found at https://knownagents.com/agents/apifywebsitecontentcrawler |
-| Applebot | Unclear at this time. | Unclear at this time. | AI Search Crawlers | Unclear at this time. | Applebot is a web crawler used by Apple to index search results that allow the Siri AI Assistant to answer user questions. Siri's answers normally contain references to the website. More info can be found at https://knownagents.com/agents/applebot |
+| Applebot | Unclear at this time. | [Yes](https://support.apple.com/en-us/119829#retrieval) | AI Search Crawlers | Unclear at this time. | Applebot is a web crawler used by Apple to index search results that allow the Siri AI Assistant to answer user questions. Siri's answers normally contain references to the website. More info can be found at https://knownagents.com/agents/applebot |
| Applebot\-Extended | [Apple](https://support.apple.com/en-us/119829#datausage) | Yes | Powers features in Siri, Spotlight, Safari, Apple Intelligence, and others. | Unclear at this time. | Apple has a secondary user agent, Applebot-Extended ... [that is] used to train Apple's foundation models powering generative AI features across Apple products, including Apple Intelligence, Services, and Developer Tools. |
| Aranet\-SearchBot | Unclear at this time. | Unclear at this time. | Undocumented AI Agents | Unclear at this time. | Description unavailable from knownagents.com More info can be found at https://knownagents.com/agents/aranet-searchbot |
| atlassian\-bot | [Atlassian](https://www.atlassian.com) | [Yes](https://support.atlassian.com/organization-administration/docs/connect-custom-website-to-rovo/#Editing-your-robots.txt) | AI search, assistants and agents | No information provided. | atlassian-bot is a web crawler used to index website content for its AI search, assistants and agents available in its Rovo GenAI product. |
@@ -55,10 +55,12 @@
| DeepSeekBot | DeepSeek | No | Training language models and improving AI products | Unclear at this time. | DeepSeekBot is a web crawler used by DeepSeek to train its language models and improve its AI products. |
| Devin | Devin AI | Yes | AI Coding Agents | Unclear at this time. | Devin is a software engineering AI assistant that can browse websites and perform web-based tasks, functioning as a collaborative AI teammate for engineering teams. More info can be found at https://knownagents.com/agents/devin |
| Diffbot | [Diffbot](https://www.diffbot.com/) | At the discretion of Diffbot users. | AI Data Providers | Unclear at this time. | Diffbot is a web crawler that extracts and structures website content using AI-powered visual understanding, providing knowledge graph data for applications like market i… More info can be found at https://knownagents.com/agents/diffbot |
-| DuckAssistBot | Unclear at this time. | Unclear at this time. | AI Assistants | Unclear at this time. | DuckAssistBot is a web crawler that scans websites to collect content for DuckDuckGo's AI-assisted answers feature, which generates brief responses to search queries usin… More info can be found at https://knownagents.com/agents/duckassistbot |
+| Diffbot\-User | [Diffbot](https://www.diffbot.com/) | Yes | AI Assistants | Only when prompted by a user. | Diffbot-User is used by requests originating on behalf of a human user browsing a URL using Diffbot software, in response to their input. Documented by Diffbot at https://docs.diffbot.com/docs/does-crawl-respect-robotstxt |
+| DuckAssistBot | Unclear at this time. | [Yes](https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot/) | AI Assistants | Unclear at this time. | DuckAssistBot is a web crawler that scans websites to collect content for DuckDuckGo's AI-assisted answers feature, which generates brief responses to search queries usin… More info can be found at https://knownagents.com/agents/duckassistbot |
| Echobot Bot | Echobox | Unclear at this time. | AI Data Scrapers | Unclear at this time. | Echobot Bot is an AI data scraper operated by Echobox. It's not currently known to be artificially intelligent or AI-related. If you think that's incorrect or can provide more detail about its purpose, please contact us. More info can be found at https://knownagents.com/agents/echobot-bot |
| EchoboxBot | [Echobox](https://echobox.com) | Unclear at this time. | Data collection to support AI-powered products. | Unclear at this time. | Supports company's AI-powered social and email management products. |
| ExaBot | Unclear at this time. | Unclear at this time. | AI Data Providers | Unclear at this time. | ExaBot is a web crawler that indexes web content to power Exa's AI search engine and semantic search APIs for AI applications. More info can be found at https://knownagents.com/agents/exabot |
+| ExaSearchBot | [Exa](https://exa.ai) | Unclear at this time. | AI Search Crawlers | Unclear at this time. | ExaSearchBot is a web crawler operated by Exa that discovers and indexes public web pages so their content can be found, retrieved, and cited through Exa. More info can be found at https://knownagents.com/agents/exasearchbot |
| FacebookBot | Meta/Facebook | [Yes](https://developers.facebook.com/docs/sharing/bot/) | Training language models | Up to 1 page per second | Officially used for training Meta "speech recognition technology," unknown if used to train Meta AI specifically. |
| facebookexternalhit | Meta/Facebook | [No](https://github.com/ai-robots-txt/ai.robots.txt/issues/40#issuecomment-2524591313) | Ostensibly only for sharing, but likely used as an AI crawler as well | Unclear at this time. | Note that excluding FacebookExternalHit will block incorporating OpenGraph data when sharing in social media, including rich links in Apple's Messages app. [According to Meta](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/), its purpose is "to crawl the content of an app or website that was shared on one of Meta’s family of apps…". However, see discussions [here](https://github.com/ai-robots-txt/ai.robots.txt/pull/21) and [here](https://github.com/ai-robots-txt/ai.robots.txt/issues/40#issuecomment-2524591313) for evidence to the contrary. |
| Factset\_spyderbot | [Factset](https://www.factset.com/ai) | Unclear at this time. | AI model training. | No information provided. | Scrapes data for AI training. |
@@ -66,7 +68,7 @@
| FriendlyCrawler | Unknown | [Yes](https://imho.alex-kunz.com/2024/01/25/an-update-on-friendly-crawler) | We are using the data from the crawler to build datasets for machine learning experiments. | Unclear at this time. | Unclear who the operator is; but data is used for training/machine learning. |
| GeistHaus\-PageFetcher | GeistHaus, a company developing AI systems for therapy and psychological assessment | Unclear at this time. | AI Assistants | Unclear at this time. | GeistHaus-PageFetcher is a web crawler operated by GeistHaus, a company developing AI systems for therapy and psychological assessment. This bot fetches web pages as part… More info can be found at https://knownagents.com/agents/geisthaus-pagefetcher |
| Gemini\-Deep\-Research | Unclear at this time. | Unclear at this time. | AI Assistants | Unclear at this time. | Gemini-Deep-Research is the agent responsible for collecting and scanning resources used in Google Gemini's Deep Research feature, which acts as a personal research assis… More info can be found at https://knownagents.com/agents/gemini-deep-research |
-| Google\-Agent | Unclear at this time. | Unclear at this time. | AI Agents | Unclear at this time. | Google-Agent is used by agents hosted on Google infrastructure to navigate the web and perform actions upon user request. More info can be found at https://knownagents.com/agents/google-agent |
+| Google\-Agent | Unclear at this time. | [Yes](https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers#google-agent) | AI Agents | Unclear at this time. | Google-Agent is used by agents hosted on Google infrastructure to navigate the web and perform actions upon user request. More info can be found at https://knownagents.com/agents/google-agent |
| Google\-CloudVertexBot | Google | [Yes](https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers) | Build and manage AI models for businesses employing Vertex AI | No information. | Google-CloudVertexBot crawls sites on the site owners' request when building Vertex AI Agents. |
| Google\-Extended | Google | [Yes](https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers) | LLM training. | No information. | Used to train Gemini and Vertex AI generative APIs. Does not impact a site's inclusion or ranking in Google Search. |
| Google\-Firebase | Google | Unclear at this time. | Used as part of AI apps developed by users of Google's Firebase AI products. | Unclear at this time. | Supports Google's Firebase AI products. |
@@ -82,7 +84,6 @@
| iAskBot | Unclear at this time. | Unclear at this time. | Undocumented AI Agents | Unclear at this time. | Description unavailable from knownagents.com More info can be found at https://knownagents.com/agents/iaskbot |
| iaskspider | Unclear at this time. | Unclear at this time. | Undocumented AI Agents | Unclear at this time. | Description unavailable from knownagents.com More info can be found at https://knownagents.com/agents/iaskspider |
| iaskspider/2\.0 | iAsk | No | Crawls sites to provide answers to user queries. | Unclear at this time. | Used to provide answers to user queries. |
-| IbouBot | Ibou | Yes | Search result generation. | Unclear at this time. | Ibou.io operates a crawler service named IbouBot which fuels and updates their graph representation of the World Wide Web. This database and all the metrics are used to provide a search engine. |
| ICC\-Crawler | [NICT](https://nict.go.jp) | Yes | Scrapes data to train and support AI technologies. | No information. | Use the collected data for artificial intelligence technologies; provide data to third parties, including commercial companies; those companies can use the data for their own business. |
| ImagesiftBot | [ImageSift](https://imagesift.com) | [Yes](https://imagesift.com/about) | ImageSiftBot is a web crawler that scrapes the internet for publicly available images to support their suite of web intelligence products | No information. | Once images and text are downloaded from a webpage, ImageSift analyzes this data from the page and stores the information in an index. Their web intelligence products use this index to enable search and retrieval of similar images. |
| imageSpider | Unclear at this time. | Unclear at this time. | AI Data Scrapers | Unclear at this time. | Description unavailable from knownagents.com More info can be found at https://knownagents.com/agents/imagespider |
@@ -90,30 +91,35 @@
| ISSCyberRiskCrawler | [ISS-Corporate](https://iss-cyber.com) | No | Scrapes data to train machine learning models. | No information. | Used to train machine learning based models to quantify cyber risk. |
| kagi\-fetcher | Kagi that fetches web content to answer user queries through Kagi AI, their suite of AI-powered tools including Assistant, Res… | Unclear at this time. | AI Assistants | Unclear at this time. | kagi-fetcher is an AI Assistant operated by Kagi that fetches web content to answer user queries through Kagi AI, their suite of AI-powered tools including Assistant, Res… More info can be found at https://knownagents.com/agents/kagi-fetcher |
| Kangaroo Bot | Unclear at this time. | Unclear at this time. | AI Data Scrapers | Unclear at this time. | Kangaroo Bot is used by the company Kangaroo LLM to download data to train AI models tailored to Australian language and culture. More info can be found at https://knownagents.com/agents/kangaroo-bot |
+| Kimi\-SearchBot | [Moonshot AI](https://www.moonshot.ai) | [Yes](https://www.kimi.ai/policies/kimi-crawlers) | AI Search Crawlers | No information provided. | Kimi-SearchBot powers Kimi's search features: it analyzes pages for relevance and builds the search index. Documented by Moonshot AI at https://www.kimi.ai/policies/kimi-crawlers |
| Kimi\-User | Moonshot AI that fetches web content on behalf of users interacting with Kimi | Unclear at this time. | AI Assistants | Unclear at this time. | Kimi-User is a web crawler operated by Moonshot AI that fetches web content on behalf of users interacting with Kimi. When a user asks Kimi to summarize an article or ans… More info can be found at https://knownagents.com/agents/kimi-user |
+| KimiBot | [Moonshot AI](https://www.moonshot.ai) | [Yes](https://www.kimi.ai/policies/kimi-crawlers) | AI Data Scrapers | No information provided. | KimiBot crawls content potentially used to train Kimi's foundation models. Documented by Moonshot AI at https://www.kimi.ai/policies/kimi-crawlers |
| KlaviyoAIBot | [Klaviyo](https://www.klaviyo.com) | [Yes](https://help.klaviyo.com/hc/en-us/articles/40496146232219) | AI Assistants | Indexes based on 'change signals' and user configuration. | KlaviyoAIBot is Klaviyo's web crawler that fetches publicly available pages from domains explicitly connected to user accounts to power the Kai Customer Agent feature. Th… More info can be found at https://knownagents.com/agents/klaviyoaibot |
| KunatoCrawler | Unclear at this time. | Unclear at this time. | Undocumented AI Agents | Unclear at this time. | Description unavailable from knownagents.com More info can be found at https://knownagents.com/agents/kunatocrawler |
| laion\-huggingface\-processor | Unclear at this time. | Unclear at this time. | AI Data Scrapers | Unclear at this time. | Description unavailable from knownagents.com More info can be found at https://knownagents.com/agents/laion-huggingface-processor |
| LAIONDownloader | [Large-scale Artificial Intelligence Open Network](https://laion.ai/) | [No](https://laion.ai/faq/) | AI tools and models for machine learning research. | Unclear at this time. | LAIONDownloader is a bot by LAION, a non-profit organization that provides datasets, tools and models to liberate machine learning research. |
| LCC | Unclear at this time. | Unclear at this time. | AI Data Scrapers | Unclear at this time. | Description unavailable from knownagents.com More info can be found at https://knownagents.com/agents/lcc |
+| Lightpanda | Anyone who downloads the Lightpanda client. Possibly being used by a [Grok-adjacent](https://github.com/lightpanda-io/browser/issues/3156#issuecomment-5217843616) organization's botnet. | At the [discretion](https://github.com/lightpanda-io/browser/blob/b04c99a9111564ebe06317f644680eda5e3ee83e/src/help.zon#L385) of Lightpanda users. | AI Data Scrapers | Defined per-user. | Lightpanda is a custom-built headless browser designed for AI and automation. |
| LinerBot | Unclear at this time. | Unclear at this time. | AI Assistants | Unclear at this time. | LinerBot is the web crawler used by Liner AI assistant to gather information from academic sources and websites to provide accurate answers with line-by-line source citat… More info can be found at https://knownagents.com/agents/linerbot |
| Linguee Bot | [Linguee](https://www.linguee.com) | No | AI powered translation service | Unclear at this time. | Linguee Bot is a web crawler used by Linguee to gather training data for its AI powered translation service. |
| LinkupBot | Unclear at this time. | Unclear at this time. | AI Search Crawlers | Unclear at this time. | Description unavailable from knownagents.com More info can be found at https://knownagents.com/agents/linkupbot |
| Manus\-User | Butterfly Effect, a company based in China | Unclear at this time. | AI Agents | Unclear at this time. | Manus-User is a browser-enabled AI agent operated by Butterfly Effect, a company based in China. It autonomously navigates websites, interprets content, and carries out m… More info can be found at https://knownagents.com/agents/manus-user |
| meta\-externalagent | [Meta](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers) | Yes | Used to train models and improve products. | No information. | "The Meta-ExternalAgent crawler crawls the web for use cases such as training AI models or improving products by indexing content directly." |
-| Meta\-ExternalAgent | Unclear at this time. | Unclear at this time. | AI Data Scrapers | Unclear at this time. | Meta-ExternalAgent is a web crawler used by Meta to download training data for its AI models and improve its products by indexing content directly. More info can be found at https://knownagents.com/agents/meta-externalagent |
-| meta\-externalfetcher | Unclear at this time. | Unclear at this time. | AI Assistants | Unclear at this time. | meta-externalfetcher is used by Meta to perform user-initiated fetches of individual links from AI assistant product functions. More info can be found at https://knownagents.com/agents/meta-externalfetcher |
-| Meta\-ExternalFetcher | Unclear at this time. | Unclear at this time. | AI Assistants | Unclear at this time. | Meta-ExternalFetcher is dispatched by Meta AI products in response to user prompts, when they need to fetch an individual links. More info can be found at https://knownagents.com/agents/meta-externalfetcher |
+| Meta\-ExternalAgent | Unclear at this time. | [Yes](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/) | AI Data Scrapers | Unclear at this time. | Meta-ExternalAgent is a web crawler used by Meta to download training data for its AI models and improve its products by indexing content directly. More info can be found at https://knownagents.com/agents/meta-externalagent |
+| meta\-externalfetcher | Unclear at this time. | [No](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/) | AI Assistants | Unclear at this time. | meta-externalfetcher is used by Meta to perform user-initiated fetches of individual links from AI assistant product functions. More info can be found at https://knownagents.com/agents/meta-externalfetcher |
+| Meta\-ExternalFetcher | Unclear at this time. | [No](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/) | AI Assistants | Unclear at this time. | Meta-ExternalFetcher is dispatched by Meta AI products in response to user prompts, when they need to fetch an individual links. More info can be found at https://knownagents.com/agents/meta-externalfetcher |
| meta\-webindexer | [Meta](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/) | Unclear at this time. | AI Assistants | Unhinged, more than 1 per second. | As per their documentation, "The Meta-WebIndexer crawler navigates the web to improve Meta AI search result quality for users. In doing so, Meta analyzes online content to enhance the relevance and accuracy of Meta AI. Allowing Meta-WebIndexer in your robots.txt file helps us cite and link to your content in Meta AI's responses." |
+| MistralAI\-Training | [Mistral AI](https://mistral.ai) | [Yes](https://docs.mistral.ai/robots/) | AI Data Scrapers | No information provided. | MistralAI-Training crawls web content to build training datasets. Documented by Mistral at https://docs.mistral.ai/robots/ |
| MistralAI\-User | Mistral | Unclear at this time. | AI Assistants | Unclear at this time. | MistralAI-User is Mistral's AI assistant bot that performs web browsing and data gathering tasks for users in Le Chat, including opening web pages and retrieving informat… More info can be found at https://knownagents.com/agents/mistralai-user |
| MistralAI\-User/1\.0 | Mistral AI | Yes | Takes action based on user prompts. | Only when prompted by a user. | MistralAI-User is for user actions in LeChat. When users ask LeChat a question, it may visit a web page to help answer and include a link to the source in its response. |
-| Mozilla\-Tabstack | [Mozilla](https://docs.tabstack.ai/trust/controlling-access) | Yes | AI Data Providers | On demand via API. | Tabstack is a web intelligence API for AI agents. It extracts structured data from web pages and makes it available to AI agents. |
+| Mozilla\-Tabstack | Mozilla that performs programmatic, AI-driven interactions with web content through Tabstack | Yes | AI Data Providers | On demand via API. | Mozilla-Tabstack is an AI agent operated by Mozilla that performs programmatic, AI-driven interactions with web content through Tabstack. More info can be found at https://knownagents.com/agents/mozilla-tabstack |
| MyCentralAIScraperBot | Unclear at this time. | Unclear at this time. | AI data scraper | Unclear at this time. | Operator and data use is unclear at this time. |
| NagetBot | Naget Inc (founded by Chris Samarinas, headquarter in Amherst, Massachusetts) | Unclear at this time. | AI data scraper | Unclear at this time. | 'Naget revolutionizes content discovery through an AI-powered ecosystem that transforms how we generate, organize, share, and discover valuable content.' (https://naget.com/) User-agent string links https://naget.ai/bot which yields 404. |
| netEstate Imprint Crawler | netEstate | Unclear at this time. | AI Data Scrapers | Unclear at this time. | netEstate Imprint Crawler is an AI data scraper operated by netEstate. If you think this is incorrect or can provide additional detail about its purpose, please contact us. More info can be found at https://knownagents.com/agents/netestate-imprint-crawler |
| newsai | Unclear at this time. | Unclear at this time. | AI data scraper | Unclear at this time. | User-agent string doen't contain an URL and there multiple sites using the newsai brand. |
| NotebookLM | Unclear at this time. | Unclear at this time. | AI Assistants | Unclear at this time. | NotebookLM is an AI-powered research and note-taking assistant that helps users synthesize information from their own uploaded sources, such as documents, transcripts, or web content. It can generate summaries, answer questions, and highlight key themes from the materials you provide, acting like a personalized research companion built on Google's Gemini model. NotebookLM fetches source URLs when users add them to their notebooks, enabling the AI to access and analyze those pages for context and insights. More info can be found at https://knownagents.com/agents/google-notebooklm |
| NovaAct | Unclear at this time. | Unclear at this time. | AI Agents | Unclear at this time. | Nova Act is an AI agent created by Amazon that can use a web browser. It can intelligently navigate and interact with websites to complete multi-step tasks on behalf of a… More info can be found at https://knownagents.com/agents/novaact |
+| OAI\-AdsBot | [OpenAI](https://openai.com) | Unclear at this time. | Validates and targets ads on ChatGPT. | Only when a page is submitted as an ad. | OAI-AdsBot visits the landing page of a page submitted as an ad on ChatGPT, to check it against OpenAI's policies and, per OpenAI, to "determine when it's most relevant to show the ad to users". OpenAI states it only visits pages submitted as ads and that the data is not used to train foundation models. Documented at https://developers.openai.com/api/docs/bots |
| OAI\-SearchBot | [OpenAI](https://openai.com) | [Yes](https://platform.openai.com/docs/bots) | Search result generation. | No information. | Crawls sites to surface as results in SearchGPT. |
| omgili | [Webz.io](https://webz.io/) | [Yes](https://webz.io/blog/web-data/what-is-the-omgili-bot-and-why-is-it-crawling-your-website/) | Data is sold. | No information. | Crawls sites for APIs used by Hootsuite, Sprinklr, NetBase, and other companies. Data also sold for research purposes or LLM training. |
| omgilibot | [Webz.io](https://webz.io/) | [Yes](https://web.archive.org/web/20170704003301/http://omgili.com/Crawler.html) | Data is sold. | No information. | Legacy user agent initially used for Omgili search engine. Unknown if still used, `omgili` agent still used by Webz.io. |
@@ -134,6 +140,7 @@
| QueritBot | Querit, a company providing a search API for large language model integration | Unclear at this time. | AI Data Providers | Unclear at this time. | QueritBot is a web crawler operated by Querit, a company providing a search API for large language model integration. This bot indexes web content to power the real-time … More info can be found at https://knownagents.com/agents/queritbot |
| QuillBot | [Quillbot](https://quillbot.com) | Unclear at this time. | Company offers AI detection, writing tools and other services. | No explicit frequency provided. | Operated by QuillBot as part of their suite of AI product offerings. |
| quillbot\.com | [Quillbot](https://quillbot.com) | Unclear at this time. | Company offers AI detection, writing tools and other services. | No explicit frequency provided. | Operated by QuillBot as part of their suite of AI product offerings. |
+| Reflectionbot | [Reflection](https://reflection.ai/) | Unclear at this time. | Undocumented AI Agents | Unclear at this time. | An undocumented crawler whose user agent links to Reflection, a company that builds AI models. |
| SBIntuitionsBot | [SB Intuitions](https://www.sbintuitions.co.jp/en/) | [Yes](https://www.sbintuitions.co.jp/en/bot/) | Uses data gathered in AI development and information analysis. | No information. | AI development and information analysis |
| Scrapy | [Zyte](https://www.zyte.com) | Unclear at this time. | Scrapes data for a variety of uses including training AI. | No information. | "AI and machine learning applications often need large amounts of quality data, and web data extraction is a fast, efficient way to build structured data sets." |
| SemrushBot\-OCOB | [Semrush](https://www.semrush.com/) | [Yes](https://www.semrush.com/bot/) | Crawls your site for ContentShake AI tool. | Roughly once every 10 seconds. | Data collected is used for the ContentShake AI tool reports. |