From 2eb931574381491da8a172579e71d15083d6b319 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=C3=96zden=20und=20Julia?= Date: Tue, 4 Aug 2026 09:11:42 +0200 Subject: [PATCH] Fill "Unclear at this time." from primary operator docs for 6 entries MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Re-submission of #248, which was merged then reverted in 80c19fc. The revert was caused by main.yml's commit-message handling, not by this data — that is fixed separately in #254. Six entries move from "Unclear at this time." to the operator's own documented value, each citing the primary source inline: Applebot Yes support.apple.com/en-us/119829#retrieval Meta-ExternalAgent Yes developers.facebook.com/docs/sharing/webmasters/web-crawlers/ Meta-ExternalFetcher No (same Meta page — user-initiated fetch, documented meta-externalfetcher No as not checking robots.txt) DuckAssistBot Yes duckduckgo.com/duckduckgo-help-pages/results/duckassistbot/ Google-Agent Yes developers.google.com/search/docs/crawling-indexing/google-common-crawlers#google-agent table-of-bot-metrics.md is regenerated and included, so `robots.py --convert` has nothing left to write and the workflow's commit step is skipped entirely. Verified: code/tests.py 13/13 · robots.py --convert exits 0 and is idempotent · robots.txt, .htaccess, the nginx/lighttpd/haproxy configs and the Caddyfile are byte-identical · the JSON diff touches only the six `respect` fields, nothing else. --- robots.json | 12 ++++++------ table-of-bot-metrics.md | 12 ++++++------ 2 files changed, 12 insertions(+), 12 deletions(-) diff --git a/robots.json b/robots.json index 9154c71..c1db729 100644 --- a/robots.json +++ b/robots.json @@ -127,7 +127,7 @@ }, "Applebot": { "operator": "Unclear at this time.", - "respect": "Unclear at this time.", + "respect": "[Yes](https://support.apple.com/en-us/119829#retrieval)", "function": "AI Search Crawlers", "frequency": "Unclear at this time.", "description": "Applebot is a web crawler used by Apple to index search results that allow the Siri AI Assistant to answer user questions. Siri's answers normally contain references to the website. More info can be found at https://knownagents.com/agents/applebot" @@ -387,7 +387,7 @@ }, "DuckAssistBot": { "operator": "Unclear at this time.", - "respect": "Unclear at this time.", + "respect": "[Yes](https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot/)", "function": "AI Assistants", "frequency": "Unclear at this time.", "description": "DuckAssistBot is a web crawler that scans websites to collect content for DuckDuckGo's AI-assisted answers feature, which generates brief responses to search queries usin\u2026 More info can be found at https://knownagents.com/agents/duckassistbot" @@ -464,7 +464,7 @@ }, "Google-Agent": { "operator": "Unclear at this time.", - "respect": "Unclear at this time.", + "respect": "[Yes](https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers#google-agent)", "function": "AI Agents", "frequency": "Unclear at this time.", "description": "Google-Agent is used by agents hosted on Google infrastructure to navigate the web and perform actions upon user request. More info can be found at https://knownagents.com/agents/google-agent" @@ -702,21 +702,21 @@ }, "Meta-ExternalAgent": { "operator": "Unclear at this time.", - "respect": "Unclear at this time.", + "respect": "[Yes](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/)", "function": "AI Data Scrapers", "frequency": "Unclear at this time.", "description": "Meta-ExternalAgent is a web crawler used by Meta to download training data for its AI models and improve its products by indexing content directly. More info can be found at https://knownagents.com/agents/meta-externalagent" }, "meta-externalfetcher": { "operator": "Unclear at this time.", - "respect": "Unclear at this time.", + "respect": "[No](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/)", "function": "AI Assistants", "frequency": "Unclear at this time.", "description": "meta-externalfetcher is used by Meta to perform user-initiated fetches of individual links from AI assistant product functions. More info can be found at https://knownagents.com/agents/meta-externalfetcher" }, "Meta-ExternalFetcher": { "operator": "Unclear at this time.", - "respect": "Unclear at this time.", + "respect": "[No](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/)", "function": "AI Assistants", "frequency": "Unclear at this time.", "description": "Meta-ExternalFetcher is dispatched by Meta AI products in response to user prompts, when they need to fetch an individual links. More info can be found at https://knownagents.com/agents/meta-externalfetcher" diff --git a/table-of-bot-metrics.md b/table-of-bot-metrics.md index 4590262..5f16c97 100644 --- a/table-of-bot-metrics.md +++ b/table-of-bot-metrics.md @@ -18,7 +18,7 @@ | anthropic\-ai | [Anthropic](https://www.anthropic.com) | Unclear at this time. | Scrapes data to train Anthropic's AI products. | No information provided. | Scrapes data to train LLMs and AI products offered by Anthropic. | | ApifyBot | Unclear at this time. | Unclear at this time. | AI Data Providers | Unclear at this time. | ApifyBot is a web scraping and data extraction crawler by Apify that collects website content for use in AI, LLMs, RAG, and automation workflows. More info can be found at https://knownagents.com/agents/apifybot | | ApifyWebsiteContentCrawler | Unclear at this time. | Unclear at this time. | AI Data Providers | Unclear at this time. | ApifyWebsiteContentCrawler is a web crawler by Apify that extracts and downloads full website content for use in AI, data analysis, and automation workflows. More info can be found at https://knownagents.com/agents/apifywebsitecontentcrawler | -| Applebot | Unclear at this time. | Unclear at this time. | AI Search Crawlers | Unclear at this time. | Applebot is a web crawler used by Apple to index search results that allow the Siri AI Assistant to answer user questions. Siri's answers normally contain references to the website. More info can be found at https://knownagents.com/agents/applebot | +| Applebot | Unclear at this time. | [Yes](https://support.apple.com/en-us/119829#retrieval) | AI Search Crawlers | Unclear at this time. | Applebot is a web crawler used by Apple to index search results that allow the Siri AI Assistant to answer user questions. Siri's answers normally contain references to the website. More info can be found at https://knownagents.com/agents/applebot | | Applebot\-Extended | [Apple](https://support.apple.com/en-us/119829#datausage) | Yes | Powers features in Siri, Spotlight, Safari, Apple Intelligence, and others. | Unclear at this time. | Apple has a secondary user agent, Applebot-Extended ... [that is] used to train Apple's foundation models powering generative AI features across Apple products, including Apple Intelligence, Services, and Developer Tools. | | Aranet\-SearchBot | Unclear at this time. | Unclear at this time. | Undocumented AI Agents | Unclear at this time. | Description unavailable from knownagents.com More info can be found at https://knownagents.com/agents/aranet-searchbot | | atlassian\-bot | [Atlassian](https://www.atlassian.com) | [Yes](https://support.atlassian.com/organization-administration/docs/connect-custom-website-to-rovo/#Editing-your-robots.txt) | AI search, assistants and agents | No information provided. | atlassian-bot is a web crawler used to index website content for its AI search, assistants and agents available in its Rovo GenAI product. | @@ -55,7 +55,7 @@ | DeepSeekBot | DeepSeek | No | Training language models and improving AI products | Unclear at this time. | DeepSeekBot is a web crawler used by DeepSeek to train its language models and improve its AI products. | | Devin | Devin AI | Yes | AI Coding Agents | Unclear at this time. | Devin is a software engineering AI assistant that can browse websites and perform web-based tasks, functioning as a collaborative AI teammate for engineering teams. More info can be found at https://knownagents.com/agents/devin | | Diffbot | [Diffbot](https://www.diffbot.com/) | At the discretion of Diffbot users. | AI Data Providers | Unclear at this time. | Diffbot is a web crawler that extracts and structures website content using AI-powered visual understanding, providing knowledge graph data for applications like market i… More info can be found at https://knownagents.com/agents/diffbot | -| DuckAssistBot | Unclear at this time. | Unclear at this time. | AI Assistants | Unclear at this time. | DuckAssistBot is a web crawler that scans websites to collect content for DuckDuckGo's AI-assisted answers feature, which generates brief responses to search queries usin… More info can be found at https://knownagents.com/agents/duckassistbot | +| DuckAssistBot | Unclear at this time. | [Yes](https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot/) | AI Assistants | Unclear at this time. | DuckAssistBot is a web crawler that scans websites to collect content for DuckDuckGo's AI-assisted answers feature, which generates brief responses to search queries usin… More info can be found at https://knownagents.com/agents/duckassistbot | | Echobot Bot | Echobox | Unclear at this time. | AI Data Scrapers | Unclear at this time. | Echobot Bot is an AI data scraper operated by Echobox. It's not currently known to be artificially intelligent or AI-related. If you think that's incorrect or can provide more detail about its purpose, please contact us. More info can be found at https://knownagents.com/agents/echobot-bot | | EchoboxBot | [Echobox](https://echobox.com) | Unclear at this time. | Data collection to support AI-powered products. | Unclear at this time. | Supports company's AI-powered social and email management products. | | ExaBot | Unclear at this time. | Unclear at this time. | AI Data Providers | Unclear at this time. | ExaBot is a web crawler that indexes web content to power Exa's AI search engine and semantic search APIs for AI applications. More info can be found at https://knownagents.com/agents/exabot | @@ -66,7 +66,7 @@ | FriendlyCrawler | Unknown | [Yes](https://imho.alex-kunz.com/2024/01/25/an-update-on-friendly-crawler) | We are using the data from the crawler to build datasets for machine learning experiments. | Unclear at this time. | Unclear who the operator is; but data is used for training/machine learning. | | GeistHaus\-PageFetcher | GeistHaus, a company developing AI systems for therapy and psychological assessment | Unclear at this time. | AI Assistants | Unclear at this time. | GeistHaus-PageFetcher is a web crawler operated by GeistHaus, a company developing AI systems for therapy and psychological assessment. This bot fetches web pages as part… More info can be found at https://knownagents.com/agents/geisthaus-pagefetcher | | Gemini\-Deep\-Research | Unclear at this time. | Unclear at this time. | AI Assistants | Unclear at this time. | Gemini-Deep-Research is the agent responsible for collecting and scanning resources used in Google Gemini's Deep Research feature, which acts as a personal research assis… More info can be found at https://knownagents.com/agents/gemini-deep-research | -| Google\-Agent | Unclear at this time. | Unclear at this time. | AI Agents | Unclear at this time. | Google-Agent is used by agents hosted on Google infrastructure to navigate the web and perform actions upon user request. More info can be found at https://knownagents.com/agents/google-agent | +| Google\-Agent | Unclear at this time. | [Yes](https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers#google-agent) | AI Agents | Unclear at this time. | Google-Agent is used by agents hosted on Google infrastructure to navigate the web and perform actions upon user request. More info can be found at https://knownagents.com/agents/google-agent | | Google\-CloudVertexBot | Google | [Yes](https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers) | Build and manage AI models for businesses employing Vertex AI | No information. | Google-CloudVertexBot crawls sites on the site owners' request when building Vertex AI Agents. | | Google\-Extended | Google | [Yes](https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers) | LLM training. | No information. | Used to train Gemini and Vertex AI generative APIs. Does not impact a site's inclusion or ranking in Google Search. | | Google\-Firebase | Google | Unclear at this time. | Used as part of AI apps developed by users of Google's Firebase AI products. | Unclear at this time. | Supports Google's Firebase AI products. | @@ -100,9 +100,9 @@ | LinkupBot | Unclear at this time. | Unclear at this time. | AI Search Crawlers | Unclear at this time. | Description unavailable from knownagents.com More info can be found at https://knownagents.com/agents/linkupbot | | Manus\-User | Butterfly Effect, a company based in China | Unclear at this time. | AI Agents | Unclear at this time. | Manus-User is a browser-enabled AI agent operated by Butterfly Effect, a company based in China. It autonomously navigates websites, interprets content, and carries out m… More info can be found at https://knownagents.com/agents/manus-user | | meta\-externalagent | [Meta](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers) | Yes | Used to train models and improve products. | No information. | "The Meta-ExternalAgent crawler crawls the web for use cases such as training AI models or improving products by indexing content directly." | -| Meta\-ExternalAgent | Unclear at this time. | Unclear at this time. | AI Data Scrapers | Unclear at this time. | Meta-ExternalAgent is a web crawler used by Meta to download training data for its AI models and improve its products by indexing content directly. More info can be found at https://knownagents.com/agents/meta-externalagent | -| meta\-externalfetcher | Unclear at this time. | Unclear at this time. | AI Assistants | Unclear at this time. | meta-externalfetcher is used by Meta to perform user-initiated fetches of individual links from AI assistant product functions. More info can be found at https://knownagents.com/agents/meta-externalfetcher | -| Meta\-ExternalFetcher | Unclear at this time. | Unclear at this time. | AI Assistants | Unclear at this time. | Meta-ExternalFetcher is dispatched by Meta AI products in response to user prompts, when they need to fetch an individual links. More info can be found at https://knownagents.com/agents/meta-externalfetcher | +| Meta\-ExternalAgent | Unclear at this time. | [Yes](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/) | AI Data Scrapers | Unclear at this time. | Meta-ExternalAgent is a web crawler used by Meta to download training data for its AI models and improve its products by indexing content directly. More info can be found at https://knownagents.com/agents/meta-externalagent | +| meta\-externalfetcher | Unclear at this time. | [No](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/) | AI Assistants | Unclear at this time. | meta-externalfetcher is used by Meta to perform user-initiated fetches of individual links from AI assistant product functions. More info can be found at https://knownagents.com/agents/meta-externalfetcher | +| Meta\-ExternalFetcher | Unclear at this time. | [No](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/) | AI Assistants | Unclear at this time. | Meta-ExternalFetcher is dispatched by Meta AI products in response to user prompts, when they need to fetch an individual links. More info can be found at https://knownagents.com/agents/meta-externalfetcher | | meta\-webindexer | [Meta](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/) | Unclear at this time. | AI Assistants | Unhinged, more than 1 per second. | As per their documentation, "The Meta-WebIndexer crawler navigates the web to improve Meta AI search result quality for users. In doing so, Meta analyzes online content to enhance the relevance and accuracy of Meta AI. Allowing Meta-WebIndexer in your robots.txt file helps us cite and link to your content in Meta AI's responses." | | MistralAI\-User | Mistral | Unclear at this time. | AI Assistants | Unclear at this time. | MistralAI-User is Mistral's AI assistant bot that performs web browsing and data gathering tasks for users in Le Chat, including opening web pages and retrieving informat… More info can be found at https://knownagents.com/agents/mistralai-user | | MistralAI\-User/1\.0 | Mistral AI | Yes | Takes action based on user prompts. | Only when prompted by a user. | MistralAI-User is for user actions in LeChat. When users ask LeChat a question, it may visit a web page to help answer and include a link to the source in its response. |