{
  "total": 34,
  "limit": 50,
  "offset": 0,
  "records": [
    {
      "fields": {
        "ai_robots_txt_function": "Content is used to train open language models.",
        "ai_robots_txt_respect": "Yes",
        "disagreement": "field: respects_robots; ai_robots_txt: Yes (uncited); operator: silent",
        "in_crawl_sweep": true,
        "note": "respects_robots null: operator does not say. No IP list, no date.",
        "operator": "Allen Institute for AI",
        "operator_doc_url": "https://allenai.org/crawler",
        "orgs": [
          "Allen Institute for AI"
        ],
        "purpose": "training",
        "purpose_operator_words": "The AI2 Bot explores certain domains to find web content. This web content is used to train open language models.",
        "respects_robots_basis": "Ai2's crawler page gives the UA string 'can be used to filter or reject traffic' but makes no robots.txt compliance statement.",
        "separate_from_search_basis": "No Ai2 search product.",
        "sites_effectively_blocking": 41,
        "sites_named_blocking": 27,
        "sites_readable": 245,
        "url": "https://allenai.org/crawler",
        "user_agent": "AI2Bot",
        "verification": []
      },
      "id": "crawler:ai2bot",
      "name": "AI2Bot",
      "notes": {
        "_org_roles": {
          "Allen Institute for AI": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "allenai.org"
        ],
        "source_urls": [
          "https://allenai.org/crawler"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.571,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "C",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://allenai.org/crawler"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Service improvement and enabling answers for Alexa users.",
        "ai_robots_txt_respect": "Yes",
        "disagreement": "field: purpose; ai_robots_txt: Service improvement and enabling answers for Alexa users; operator: may be used to train Amazon AI models",
        "in_crawl_sweep": true,
        "ip_ranges_url": "https://developer.amazon.com/amazonbot/ip-addresses/",
        "note": "Amazon ties robots.txt to a commercial carrot: allowing Amazonbot may qualify a site for Amazon Content Partners (+1% affiliate commission). first_documented null: no date.",
        "operator": "Amazon",
        "operator_doc_url": "https://developer.amazon.com/amazonbot",
        "opt_out_token_separate_from_search": false,
        "orgs": [
          "Amazon"
        ],
        "purpose": "mixed",
        "purpose_operator_words": "Amazonbot is used to improve our products and services. This helps us provide more accurate information to customers and may be used to train Amazon AI models.",
        "respects_robots": true,
        "respects_robots_basis": "Amazon: 'Automated crawling from these listed user agents respects the Robots Exclusion Protocol'; honours noarchive as 'do not use the page for model training'; no crawl-delay.",
        "separate_from_search_basis": "Amazon: 'Each user agent setting is independent of the others'; search inclusion is governed by Amzn-SearchBot.",
        "sites_effectively_blocking": 61,
        "sites_named_blocking": 50,
        "sites_readable": 245,
        "url": "https://developer.amazon.com/amazonbot",
        "user_agent": "Amazonbot",
        "verification": [
          "published_ip_ranges"
        ]
      },
      "id": "crawler:amazonbot",
      "name": "Amazonbot",
      "notes": {
        "_org_roles": {
          "Amazon": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "developer.amazon.com"
        ],
        "source_urls": [
          "https://developer.amazon.com/amazonbot"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://developer.amazon.com/amazonbot"
    },
    {
      "fields": {
        "ai_robots_txt_function": "AI Search Crawlers",
        "ai_robots_txt_respect": "Unclear at this time.",
        "in_crawl_sweep": false,
        "ip_ranges_url": "https://developer.amazon.com/amazonbot/searchbot-ip-addresses/",
        "note": "Not in our crawl_block list or ai.robots.txt lookup here. Inherits other search bots' rules when unnamed, so our wildcard 'effective' logic would misstate it.",
        "operator": "Amazon",
        "operator_doc_url": "https://developer.amazon.com/amazonbot",
        "opt_out_token_separate_from_search": true,
        "orgs": [
          "Amazon"
        ],
        "purpose": "search_index",
        "purpose_operator_words": "Amzn-SearchBot is used to improve search experiences in Amazon products and services... Amzn-SearchBot does not crawl content for generative AI model training.",
        "respects_robots": true,
        "respects_robots_basis": "Amazon: 'Automated crawling from these listed user agents respects the Robots Exclusion Protocol'; honours noarchive as 'do not use the page for model training'; no crawl-delay. If not named, it follows rules given to other search bots.",
        "separate_from_search_basis": "It is the search opt-out: allowing it makes content 'eligible to appear in search experiences such as Alexa'.",
        "url": "https://developer.amazon.com/amazonbot",
        "user_agent": "Amzn-SearchBot",
        "verification": [
          "published_ip_ranges"
        ]
      },
      "id": "crawler:amzn-searchbot",
      "name": "Amzn-SearchBot",
      "notes": {
        "_org_roles": {
          "Amazon": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "developer.amazon.com"
        ],
        "source_urls": [
          "https://developer.amazon.com/amazonbot"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://developer.amazon.com/amazonbot"
    },
    {
      "fields": {
        "ai_robots_txt_function": "AI Assistants",
        "ai_robots_txt_respect": "Unclear at this time.",
        "in_crawl_sweep": false,
        "ip_ranges_url": "https://developer.amazon.com/amazonbot/live-ip-addresses/",
        "note": "Not in our crawl_block list.",
        "operator": "Amazon",
        "operator_doc_url": "https://developer.amazon.com/amazonbot",
        "orgs": [
          "Amazon"
        ],
        "purpose": "user_fetch",
        "purpose_operator_words": "Amzn-User supports user actions, such as responding to Alexa queries that require up-to-date information... Amzn-User does not crawl content for generative AI model training.",
        "respects_robots": false,
        "respects_robots_basis": "Amazon: 'Because actions taken by Amzn-User can be initiated by a user, it may not follow all robots.txt directives.'",
        "separate_from_search_basis": "Not stated.",
        "url": "https://developer.amazon.com/amazonbot",
        "user_agent": "Amzn-User",
        "verification": [
          "published_ip_ranges"
        ]
      },
      "id": "crawler:amzn-user",
      "name": "Amzn-User",
      "notes": {
        "_org_roles": {
          "Amazon": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "developer.amazon.com"
        ],
        "source_urls": [
          "https://developer.amazon.com/amazonbot"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.857,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://developer.amazon.com/amazonbot"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Scrapes data to train Anthropic's AI products.",
        "ai_robots_txt_respect": "Unclear at this time.",
        "in_crawl_sweep": true,
        "note": "Legacy token: 55 sites still name it in a Disallow. purpose null/unknown and respects_robots null because the operator no longer documents it; ai.robots.txt itself says respect 'Unclear at this time'. Blocking it likely does nothing, which is itself worth telling publishers, but that is an inference and not recorded as fact.",
        "operator": "Anthropic",
        "orgs": [
          "Anthropic"
        ],
        "purpose": null,
        "respects_robots_basis": "Anthropic's current doc names exactly three bots (ClaudeBot, Claude-User, Claude-SearchBot); anthropic-ai is not among them, so the operator makes no statement.",
        "separate_from_search_basis": "Operator does not document this token.",
        "sites_effectively_blocking": 68,
        "sites_named_blocking": 55,
        "sites_readable": 245,
        "url": "https://github.com/ai-robots-txt/ai.robots.txt/blob/48abe91f99318da1968ce0e038b731fc38051b8f/robots.json",
        "user_agent": "anthropic-ai",
        "verification": []
      },
      "id": "crawler:anthropic-ai",
      "name": "anthropic-ai",
      "notes": {
        "_org_roles": {
          "Anthropic": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "github.com"
        ],
        "source_urls": [
          "https://github.com/ai-robots-txt/ai.robots.txt/blob/48abe91f99318da1968ce0e038b731fc38051b8f/robots.json"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.429,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source",
          "sparse"
        ],
        "grade": "C",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://github.com/ai-robots-txt/ai.robots.txt/blob/48abe91f99318da1968ce0e038b731fc38051b8f/robots.json"
    },
    {
      "fields": {
        "ai_robots_txt_function": "AI Search Crawlers",
        "ai_robots_txt_respect": "[Yes](https://support.apple.com/en-us/119829#retrieval)",
        "in_crawl_sweep": false,
        "ip_ranges_url": "https://search.developer.apple.com/applebot.json",
        "note": "Not in our crawl_block list. Reverse DNS *.applebot.apple.com.",
        "operator": "Apple",
        "operator_doc_url": "https://support.apple.com/en-us/119829",
        "opt_out_token_separate_from_search": true,
        "orgs": [
          "Apple"
        ],
        "purpose": "mixed",
        "purpose_operator_words": "The data crawled by Applebot is used to power various features, such as the search technology integrated into many user experiences in Apple's ecosystem including Spotlight, Siri, and Safari... may also be used to help train Apple foundation models.",
        "respects_robots": true,
        "respects_robots_basis": "Apple: 'In addition to following all robots.txt rules and directives, Apple has a secondary user agent, Applebot-Extended'. Operator statement.",
        "separate_from_search_basis": "Applebot is the search crawler; blocking it removes Spotlight/Siri/Safari search inclusion.",
        "url": "https://support.apple.com/en-us/119829",
        "user_agent": "Applebot",
        "verification": [
          "published_ip_ranges",
          "reverse_dns"
        ]
      },
      "id": "crawler:applebot",
      "name": "Applebot",
      "notes": {
        "_org_roles": {
          "Apple": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "support.apple.com"
        ],
        "source_urls": [
          "https://support.apple.com/en-us/119829"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://support.apple.com/en-us/119829"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Powers features in Siri, Spotlight, Safari, Apple Intelligence, and others.",
        "ai_robots_txt_respect": "Yes",
        "disagreement": "field: purpose; ai_robots_txt: Powers features in Siri, Spotlight, Safari, Apple Intelligence (function text describes Applebot, not the Extended token); operator: training opt-out token only; AI answers with links are controlled separately by nosnippet",
        "in_crawl_sweep": true,
        "note": "Apple adds a third control: blocking Applebot-Extended does NOT stop Applebot crawled content being used as AI answer context; that needs the nosnippet meta tag, and isAccessibleForFree:false also excludes a page. first_documented null: page 'Published Date: September 04, 2026' is a revision date. ip_ranges_url null: token does not fetch; Applebot's list applies.",
        "operator": "Apple",
        "operator_doc_url": "https://support.apple.com/en-us/119829",
        "opt_out_token_separate_from_search": false,
        "orgs": [
          "Apple"
        ],
        "purpose": "training",
        "purpose_operator_words": "With Applebot-Extended, web publishers can choose to opt out of their website content being used to train Apple's general purpose foundation models powering generative AI features across Apple products, including Apple Intelligence, Services, and Developer Tools.",
        "respects_robots": true,
        "respects_robots_basis": "Apple: 'Applebot-Extended does not crawl webpages... is only used to determine how to use the data crawled by the Applebot user agent.' It is a control token read from robots.txt.",
        "separate_from_search_basis": "Apple: 'Webpages that disallow Applebot-Extended can still be included in search results.' and 'Site rules for Applebot-Extended are not considered in ranking for Search.'",
        "sites_effectively_blocking": 67,
        "sites_named_blocking": 56,
        "sites_readable": 245,
        "url": "https://support.apple.com/en-us/119829",
        "user_agent": "Applebot-Extended",
        "verification": [
          "not_applicable_control_token"
        ]
      },
      "id": "crawler:applebot-extended",
      "name": "Applebot-Extended",
      "notes": {
        "_org_roles": {
          "Apple": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "support.apple.com"
        ],
        "source_urls": [
          "https://support.apple.com/en-us/119829"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://support.apple.com/en-us/119829"
    },
    {
      "fields": {
        "ai_robots_txt_function": "LLM training.",
        "ai_robots_txt_respect": "No",
        "disagreement": "field: respects_robots; ai_robots_txt: No (no source cited); operator: silent",
        "in_crawl_sweep": true,
        "note": "The second most blocked crawler in our rows (84 effective) has no operator documentation at all. ai.robots.txt's 'No' is uncited, so it is not recorded as fact. purpose 'LLM training' per ai.robots.txt only.",
        "operator": "ByteDance",
        "orgs": [
          "ByteDance"
        ],
        "purpose": null,
        "respects_robots_basis": "No operator page found: bytespider.bytedance.com does not resolve (DNS ENOTFOUND on 2026-09-17).",
        "separate_from_search_basis": "No operator documentation.",
        "sites_effectively_blocking": 84,
        "sites_named_blocking": 71,
        "sites_readable": 245,
        "url": "https://github.com/ai-robots-txt/ai.robots.txt/blob/48abe91f99318da1968ce0e038b731fc38051b8f/robots.json",
        "user_agent": "Bytespider",
        "verification": []
      },
      "id": "crawler:bytespider",
      "name": "Bytespider",
      "notes": {
        "_org_roles": {
          "ByteDance": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "github.com"
        ],
        "source_urls": [
          "https://github.com/ai-robots-txt/ai.robots.txt/blob/48abe91f99318da1968ce0e038b731fc38051b8f/robots.json"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.429,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source",
          "sparse"
        ],
        "grade": "C",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://github.com/ai-robots-txt/ai.robots.txt/blob/48abe91f99318da1968ce0e038b731fc38051b8f/robots.json"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Provides open crawl dataset, used for many purposes, including Machine Learning/AI.",
        "ai_robots_txt_respect": "[Yes](https://commoncrawl.org/ccbot)",
        "in_crawl_sweep": true,
        "ip_ranges_url": "https://index.commoncrawl.org/ccbot.json",
        "note": "The most blocked crawler in our rows (88 effective). Purpose 'mixed' because the corpus is general-purpose and downstream AI training is by third parties. Blocking stops future crawls, not copies already in past archives. Common Crawl warns of UA spoofing; reverse DNS *.crawl.commoncrawl.org (IPv4 only).",
        "operator": "Common Crawl",
        "operator_doc_url": "https://commoncrawl.org/ccbot",
        "orgs": [
          "Common Crawl"
        ],
        "purpose": "mixed",
        "purpose_operator_words": "Common Crawl is a non-profit foundation founded with the goal of democratizing access to web information by producing and maintaining an open repository of web crawl data",
        "respects_robots": true,
        "respects_robots_basis": "Common Crawl: 'To prevent Common Crawl from crawling your website, include the following in your robots.txt: User-agent: CCBot Disallow: /'.",
        "separate_from_search_basis": "Common Crawl runs no search product; not applicable, left null.",
        "sites_effectively_blocking": 88,
        "sites_named_blocking": 75,
        "sites_readable": 245,
        "url": "https://commoncrawl.org/ccbot",
        "user_agent": "CCBot",
        "verification": [
          "published_ip_ranges",
          "reverse_dns"
        ]
      },
      "id": "crawler:ccbot",
      "name": "CCBot",
      "notes": {
        "_org_roles": {
          "Common Crawl": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "commoncrawl.org"
        ],
        "source_urls": [
          "https://commoncrawl.org/ccbot"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.857,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://commoncrawl.org/ccbot"
    },
    {
      "fields": {
        "ai_robots_txt_function": "AI Assistants",
        "ai_robots_txt_respect": "Yes",
        "disagreement": "field: respects_robots; ai_robots_txt: Yes; operator: robots.txt rules may not apply; cloudflare_measured: complied in Cloudflare's 2025 test",
        "in_crawl_sweep": true,
        "ip_ranges_url": "https://openai.com/chatgpt-user.json",
        "measured_by_third_party": "{\"source\": \"https://blog.cloudflare.com/perplexity-is-using-stealth-undeclared-crawlers-to-evade-website-no-crawl-directives/\", \"claim\": \"Cloudflare test published 2025-08-04: 'ChatGPT-User fetched the robots file and stopped crawling when it was disallowed.' Measured behaviour is stricter than the operator's own doc.\"}",
        "note": "respects_robots is false on the operator's words; Cloudflare's measurement says it obeyed in one test. Both kept. first_documented null: no date on page.",
        "operator": "OpenAI",
        "operator_doc_url": "https://developers.openai.com/api/docs/bots",
        "opt_out_token_separate_from_search": false,
        "orgs": [
          "OpenAI"
        ],
        "purpose": "user_fetch",
        "purpose_operator_words": "When users ask ChatGPT or a CustomGPT a question, it may visit a web page with a ChatGPT-User agent... ChatGPT-User is not used for crawling the web in an automatic fashion.",
        "respects_robots": false,
        "respects_robots_basis": "OpenAI doc: 'Because these actions are initiated by a user, robots.txt rules may not apply.' Operator reserves the right to ignore robots.txt.",
        "separate_from_search_basis": "OpenAI doc: 'ChatGPT-User is not used to determine whether content may appear in Search. Please use OAI-SearchBot in robots.txt for managing Search opt outs'.",
        "sites_effectively_blocking": 49,
        "sites_named_blocking": 41,
        "sites_readable": 245,
        "url": "https://developers.openai.com/api/docs/bots",
        "user_agent": "ChatGPT-User",
        "verification": [
          "published_ip_ranges"
        ]
      },
      "id": "crawler:chatgpt-user",
      "name": "ChatGPT-User",
      "notes": {
        "_org_roles": {
          "OpenAI": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.openai.com"
        ],
        "source_urls": [
          "https://developers.openai.com/api/docs/bots"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://developers.openai.com/api/docs/bots"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Claude-SearchBot navigates the web to improve search result quality for users. It analyzes online content specifically to enhance the relevance and accuracy of search responses.",
        "ai_robots_txt_respect": "[Yes](https://support.anthropic.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler)",
        "in_crawl_sweep": false,
        "ip_ranges_url": "https://claude.com/crawling/bots.json",
        "note": "Not in our crawl_block list of 22, so no join counts: a gap in the sweep, since it is Anthropic's search opt-out. first_documented null: no date.",
        "operator": "Anthropic",
        "operator_doc_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
        "opt_out_token_separate_from_search": true,
        "orgs": [
          "Anthropic"
        ],
        "purpose": "search_index",
        "purpose_operator_words": "Claude-SearchBot navigates the web to improve search result quality for users. It analyzes online content specifically to enhance the relevance and accuracy of search responses.",
        "respects_robots": true,
        "respects_robots_basis": "Anthropic doc: 'Anthropic's Bots respect do not crawl signals by honoring industry standard directives in robots.txt' and supports Crawl-delay. Operator statement.",
        "separate_from_search_basis": "Anthropic: 'Disabling Claude-SearchBot on your site prevents our system from indexing your content for search optimization'.",
        "url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
        "user_agent": "Claude-SearchBot",
        "verification": [
          "published_ip_ranges"
        ]
      },
      "id": "crawler:claude-searchbot",
      "name": "Claude-SearchBot",
      "notes": {
        "_org_roles": {
          "Anthropic": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "support.claude.com"
        ],
        "source_urls": [
          "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
    },
    {
      "fields": {
        "ai_robots_txt_function": "AI Assistants",
        "ai_robots_txt_respect": "[Yes](https://support.anthropic.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler)",
        "in_crawl_sweep": true,
        "ip_ranges_url": "https://claude.com/crawling/bots.json",
        "note": "first_documented null: no first publication date on page.",
        "operator": "Anthropic",
        "operator_doc_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
        "orgs": [
          "Anthropic"
        ],
        "purpose": "user_fetch",
        "purpose_operator_words": "When individuals ask questions to Claude, it may access websites using a Claude-User agent.",
        "respects_robots": true,
        "respects_robots_basis": "Anthropic doc: 'Anthropic's Bots respect do not crawl signals by honoring industry standard directives in robots.txt' and supports Crawl-delay. Operator statement. Unlike OpenAI and Perplexity, the respect statement covers the user-initiated fetcher too.",
        "separate_from_search_basis": "Anthropic: 'Disabling Claude-User ... may reduce your site's visibility for user-directed web search.' It does not say whether Claude-SearchBot indexing is affected, so not established.",
        "sites_effectively_blocking": 43,
        "sites_named_blocking": 32,
        "sites_readable": 245,
        "url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
        "user_agent": "Claude-User",
        "verification": [
          "published_ip_ranges"
        ]
      },
      "id": "crawler:claude-user",
      "name": "Claude-User",
      "notes": {
        "_org_roles": {
          "Anthropic": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "support.claude.com"
        ],
        "source_urls": [
          "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.857,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Scrapes data to train Anthropic's AI products.",
        "ai_robots_txt_respect": "[Yes](https://support.anthropic.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler)",
        "in_crawl_sweep": true,
        "ip_ranges_url": "https://claude.com/crawling/bots.json",
        "measured_by_third_party": "{\"source\": \"https://blog.cloudflare.com/ai-crawler-traffic-by-purpose-and-industry/\", \"claim\": \"Cloudflare, week of 2025-08-01 to 08-07: Anthropic crawl-to-referral ratio about 50,000:1 across all industries, the highest of the named operators.\"}",
        "note": "first_documented null: page shows an update date (April 7, 2026) not a first publication date. bots.json is one list for all Anthropic bots (creationTime 2026-08-18). Anthropic warns IP blocking may stop it reading robots.txt.",
        "operator": "Anthropic",
        "operator_doc_url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
        "orgs": [
          "Anthropic"
        ],
        "purpose": "training",
        "purpose_operator_words": "ClaudeBot helps enhance the utility and safety of our generative AI models by collecting web content that could potentially contribute to their training.",
        "respects_robots": true,
        "respects_robots_basis": "Anthropic doc: 'Anthropic's Bots respect do not crawl signals by honoring industry standard directives in robots.txt' and supports Crawl-delay. Operator statement.",
        "separate_from_search_basis": "Anthropic lists three separate bots and says blocking Claude-SearchBot reduces search visibility, but never states that blocking ClaudeBot leaves search unaffected, so not established.",
        "sites_effectively_blocking": 79,
        "sites_named_blocking": 68,
        "sites_readable": 245,
        "url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler",
        "user_agent": "ClaudeBot",
        "verification": [
          "published_ip_ranges"
        ]
      },
      "id": "crawler:claudebot",
      "name": "ClaudeBot",
      "notes": {
        "_org_roles": {
          "Anthropic": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "support.claude.com"
        ],
        "source_urls": [
          "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.857,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Retrieves data to provide responses to user-initiated prompts.",
        "ai_robots_txt_respect": "Unclear at this time.",
        "disagreement": "field: existence; ai_robots_txt: cohere-ai retrieves data for user prompts; operator: no bots in use; sample opt-out token is 'Coherebot', not cohere-ai",
        "in_crawl_sweep": true,
        "note": "58 of our sites block a token the operator does not acknowledge, and Cohere's own sample robots.txt names a different token (Coherebot).",
        "operator": "Cohere",
        "operator_doc_url": "https://docs.cohere.com/docs/cohere-web-crawlers",
        "orgs": [
          "Cohere"
        ],
        "purpose": null,
        "respects_robots_basis": "Cohere's crawler doc says it does 'not use Cohere bots or user agents for the purpose of crawling or scraping web content to train generative AI foundation models at this time' and its bot table reads N/A; it does not mention cohere-ai.",
        "separate_from_search_basis": "Not documented.",
        "sites_effectively_blocking": 58,
        "sites_named_blocking": 44,
        "sites_readable": 245,
        "url": "https://docs.cohere.com/docs/cohere-web-crawlers",
        "user_agent": "cohere-ai",
        "verification": []
      },
      "id": "crawler:cohere-ai",
      "name": "cohere-ai",
      "notes": {
        "_org_roles": {
          "Cohere": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "docs.cohere.com"
        ],
        "source_urls": [
          "https://docs.cohere.com/docs/cohere-web-crawlers"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.429,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source",
          "sparse"
        ],
        "grade": "C",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://docs.cohere.com/docs/cohere-web-crawlers"
    },
    {
      "fields": {
        "ai_robots_txt_function": "AI Data Scrapers",
        "ai_robots_txt_respect": "Unclear at this time.",
        "disagreement": "field: existence; ai_robots_txt: operator 'Cohere to download training data'; operator: no training crawler 'at this time'",
        "in_crawl_sweep": false,
        "note": "Not in our crawl_block list. Direct conflict between the list and the operator; neither measured.",
        "operator": "Cohere",
        "operator_doc_url": "https://docs.cohere.com/docs/cohere-web-crawlers",
        "orgs": [
          "Cohere"
        ],
        "purpose": null,
        "respects_robots_basis": "Cohere's doc denies operating training crawlers at this time.",
        "separate_from_search_basis": "Not documented.",
        "url": "https://docs.cohere.com/docs/cohere-web-crawlers",
        "user_agent": "cohere-training-data-crawler",
        "verification": []
      },
      "id": "crawler:cohere-training-data-crawler",
      "name": "cohere-training-data-crawler",
      "notes": {
        "_org_roles": {
          "Cohere": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "docs.cohere.com"
        ],
        "source_urls": [
          "https://docs.cohere.com/docs/cohere-web-crawlers"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.429,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source",
          "sparse"
        ],
        "grade": "C",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://docs.cohere.com/docs/cohere-web-crawlers"
    },
    {
      "fields": {
        "ai_robots_txt_function": "AI Data Providers",
        "ai_robots_txt_respect": "At the discretion of Diffbot users.",
        "disagreement": "field: respects_robots; ai_robots_txt: At the discretion of Diffbot users; function 'AI Data Providers'; operator: respects by default, override by agreement; says not used for AI training",
        "in_crawl_sweep": true,
        "note": "Diffbot's knowledge graph is sold to AI builders; 'not used for AI training' refers to Diffbot's own use, as summarised by the fetcher, verify verbatim before ingest. Also documents Diffbot-User for user-initiated fetches. No IP list, so null. Purpose left unknown: a claim that Diffbot crawls for its Knowledge Graph and not for AI training came from a summarising fetch and is not on the cited page (checked verbatim 2026-09-17).",
        "operator": "Diffbot",
        "operator_doc_url": "https://www.diffbot.com/docs/crawl/faq/robots-txt",
        "orgs": [
          "Diffbot"
        ],
        "purpose": null,
        "respects_robots": true,
        "respects_robots_basis": "Diffbot: 'By default Diffbot's web crawls adhere to a site's robots.txt instructions, including the disallow and crawl-delay directives', but can be overridden 'typically because of a partnership or agreement' with the crawled site.",
        "separate_from_search_basis": "Diffbot's search is sold to customers; no statement on separate tokens.",
        "sites_effectively_blocking": 67,
        "sites_named_blocking": 54,
        "sites_readable": 245,
        "url": "https://www.diffbot.com/docs/crawl/faq/robots-txt",
        "user_agent": "Diffbot",
        "verification": []
      },
      "id": "crawler:diffbot",
      "name": "Diffbot",
      "notes": {
        "_org_roles": {
          "Diffbot": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "diffbot.com"
        ],
        "source_urls": [
          "https://www.diffbot.com/docs/crawl/faq/robots-txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.571,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "C",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://www.diffbot.com/docs/crawl/faq/robots-txt"
    },
    {
      "fields": {
        "ai_robots_txt_function": "AI Assistants",
        "ai_robots_txt_respect": "[Yes](https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot/)",
        "in_crawl_sweep": false,
        "ip_ranges_url": "https://duckduckgo.com/duckassistbot.json",
        "note": "Not in our crawl_block list. The cleanest operator statement of the separate-token question in the set.",
        "operator": "DuckDuckGo",
        "operator_doc_url": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot/",
        "opt_out_token_separate_from_search": false,
        "orgs": [
          "DuckDuckGo"
        ],
        "purpose": "user_fetch",
        "purpose_operator_words": "DuckAssistBot is a web crawler for DuckDuckGo Search that crawls pages in real-time for our AI-assisted answers, which prominently cite their sources. This data is not used in any way to train AI models.",
        "respects_robots": true,
        "respects_robots_basis": "DuckDuckGo: after a disallow 'the change will take effect after 72 hours and DuckAssistBot will stop crawling your site.'",
        "separate_from_search_basis": "DuckDuckGo: 'Opting out of DuckAssistBot does not impact organic search rankings and will not affect whether or not websites appear in our search results.'",
        "url": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot",
        "user_agent": "DuckAssistBot",
        "verification": [
          "published_ip_ranges"
        ]
      },
      "id": "crawler:duckassistbot",
      "name": "DuckAssistBot",
      "notes": {
        "_org_roles": {
          "DuckDuckGo": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "duckduckgo.com"
        ],
        "source_urls": [
          "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Training language models",
        "ai_robots_txt_respect": "[Yes](https://developers.facebook.com/docs/sharing/bot/)",
        "disagreement": "field: purpose; ai_robots_txt: Training language models (speech recognition), respect Yes, citing a Meta URL that no longer describes it; operator: no current documentation",
        "in_crawl_sweep": true,
        "note": "Legacy token: 42 of our sites still disallow it. purpose/respects null because the operator page no longer carries it.",
        "operator": "Meta",
        "orgs": [
          "Meta"
        ],
        "purpose": null,
        "respects_robots_basis": "Meta's current crawler pages do not mention FacebookBot (docs/sharing/bot now redirects to the web-crawlers page, which covers facebookexternalhit, meta-externalagent, meta-externalfetcher).",
        "separate_from_search_basis": "Not documented by operator.",
        "sites_effectively_blocking": 54,
        "sites_named_blocking": 42,
        "sites_readable": 245,
        "url": "https://github.com/ai-robots-txt/ai.robots.txt/blob/48abe91f99318da1968ce0e038b731fc38051b8f/robots.json",
        "user_agent": "FacebookBot",
        "verification": []
      },
      "id": "crawler:facebookbot",
      "name": "FacebookBot",
      "notes": {
        "_org_roles": {
          "Meta": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "github.com"
        ],
        "source_urls": [
          "https://github.com/ai-robots-txt/ai.robots.txt/blob/48abe91f99318da1968ce0e038b731fc38051b8f/robots.json"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.429,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source",
          "sparse"
        ],
        "grade": "C",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://github.com/ai-robots-txt/ai.robots.txt/blob/48abe91f99318da1968ce0e038b731fc38051b8f/robots.json"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Build and manage AI models for businesses employing Vertex AI",
        "ai_robots_txt_respect": "[Yes](https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers)",
        "in_crawl_sweep": false,
        "ip_ranges_url": "https://developers.google.com/static/crawling/ipranges/common-crawlers.json",
        "note": "Not in our crawl_block list. purpose is site-owner-requested fetch for agents, closest to user_fetch.",
        "operator": "Google",
        "operator_doc_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
        "opt_out_token_separate_from_search": false,
        "orgs": [
          "Google"
        ],
        "purpose": "user_fetch",
        "purpose_operator_words": "Crawling preferences addressed to the Google-CloudVertexBot user agent affect crawls requested by the site owners' for building Vertex AI Agents.",
        "respects_robots": true,
        "respects_robots_basis": "Listed with its own robots.txt token on Google's common crawlers page. Operator statement.",
        "separate_from_search_basis": "Google: 'It has no effect on Google Search or other products.'",
        "url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
        "user_agent": "Google-CloudVertexBot",
        "verification": [
          "published_ip_ranges",
          "reverse_dns"
        ]
      },
      "id": "crawler:google-cloudvertexbot",
      "name": "Google-CloudVertexBot",
      "notes": {
        "_org_roles": {
          "Google": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.google.com"
        ],
        "source_urls": [
          "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers"
    },
    {
      "fields": {
        "ai_robots_txt_function": "LLM training.",
        "ai_robots_txt_respect": "[Yes](https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers)",
        "disagreement": "field: purpose; ai_robots_txt: LLM training; operator: training AND grounding in Gemini Apps / Vertex AI",
        "first_documented": "2023-09-28",
        "in_crawl_sweep": true,
        "note": "ip_ranges_url null because the token never crawls; fetching is done by Googlebot (common-crawlers.json). first_documented from Google's announcement post of 2023-09-28, which then named Bard and Vertex AI.",
        "operator": "Google",
        "operator_doc_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
        "opt_out_token_separate_from_search": false,
        "orgs": [
          "Google"
        ],
        "purpose": "mixed",
        "purpose_operator_words": "Google-Extended is a standalone product token that web publishers can use to manage whether content Google crawls from their sites may be used for training future generations of Gemini models that power Gemini Apps and Vertex AI API for Gemini and for grounding (providing content from the Google Search index to the model at prompt time...) in Gemini Apps and Grounding with Google Search on Vertex AI.",
        "respects_robots": true,
        "respects_robots_basis": "It is a robots.txt control token, not a fetcher: 'Google-Extended doesn't have a separate HTTP request user agent string... the robots.txt user-agent token is used in a control capacity.'",
        "separate_from_search_basis": "Google doc: 'Google-Extended does not impact a site's inclusion in Google Search nor is it used as a ranking signal in Google Search.' AI Overviews are served from Googlebot and are not controlled by this token.",
        "sites_effectively_blocking": 67,
        "sites_named_blocking": 58,
        "sites_readable": 245,
        "url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
        "user_agent": "Google-Extended",
        "verification": [
          "not_applicable_control_token"
        ]
      },
      "id": "crawler:google-extended",
      "name": "Google-Extended",
      "notes": {
        "_org_roles": {
          "Google": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.google.com"
        ],
        "source_urls": [
          "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Scrapes data.",
        "ai_robots_txt_respect": "[Yes](https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers)",
        "in_crawl_sweep": false,
        "ip_ranges_url": "https://developers.google.com/static/crawling/ipranges/common-crawlers.json",
        "note": "Not in our crawl_block list. purpose 'mixed' because Google does not bound its uses; training use is neither stated nor excluded. Google also documents experimental Web Bot Auth for its crawlers.",
        "operator": "Google",
        "operator_doc_url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
        "opt_out_token_separate_from_search": false,
        "orgs": [
          "Google"
        ],
        "purpose": "mixed",
        "purpose_operator_words": "GoogleOther is the generic crawler that may be used by various product teams for fetching publicly accessible content from sites. For example, it may be used for one-off crawls for internal research and development.",
        "respects_robots": true,
        "respects_robots_basis": "Listed with a robots.txt token on Google's common crawlers page, which says common crawlers obey robots.txt. Operator statement.",
        "separate_from_search_basis": "Google: 'Crawling preferences addressed to the GoogleOther user agent don't affect any specific product.'",
        "url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
        "user_agent": "GoogleOther",
        "verification": [
          "published_ip_ranges",
          "reverse_dns"
        ]
      },
      "id": "crawler:googleother",
      "name": "GoogleOther",
      "notes": {
        "_org_roles": {
          "Google": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.google.com"
        ],
        "source_urls": [
          "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Scrapes data to train OpenAI's products.",
        "ai_robots_txt_respect": "Yes",
        "in_crawl_sweep": true,
        "ip_ranges_url": "https://openai.com/gptbot.json",
        "measured_by_third_party": "{\"source\": \"https://blog.cloudflare.com/ai-crawler-traffic-by-purpose-and-industry/\", \"claim\": \"Cloudflare, week of 2025-08-01 to 08-07: OpenAI crawl-to-referral ratio about 887:1 across all industries; ClaudeBot and GPTBot together nearly half of observed AI crawling.\"}",
        "note": "first_documented null: the operator page carries no publication date. OpenAI also says 'If your site has allowed both bots, we may use the results from just one crawl for both use cases', so one fetch can serve training and search. Web Bot Auth not mentioned on this page.",
        "operator": "OpenAI",
        "operator_doc_url": "https://developers.openai.com/api/docs/bots",
        "opt_out_token_separate_from_search": false,
        "orgs": [
          "OpenAI"
        ],
        "purpose": "training",
        "purpose_operator_words": "GPTBot is used to make our generative AI foundation models more useful and safe. It is used to crawl content that may be used in training our generative AI foundation models.",
        "respects_robots": true,
        "respects_robots_basis": "OpenAI doc: 'Disallowing GPTBot indicates a site's content should not be used in training generative AI foundation models.' Operator statement, not measured.",
        "separate_from_search_basis": "OpenAI doc: 'Each setting is independent of the others - for example, a webmaster can allow OAI-SearchBot in order to appear in search results while disallowing GPTBot'.",
        "sites_effectively_blocking": 74,
        "sites_named_blocking": 66,
        "sites_readable": 245,
        "url": "https://developers.openai.com/api/docs/bots",
        "user_agent": "GPTBot",
        "verification": [
          "published_ip_ranges"
        ]
      },
      "id": "crawler:gptbot",
      "name": "GPTBot",
      "notes": {
        "_org_roles": {
          "OpenAI": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.openai.com"
        ],
        "source_urls": [
          "https://developers.openai.com/api/docs/bots"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://developers.openai.com/api/docs/bots"
    },
    {
      "fields": {
        "ai_robots_txt_function": "ImageSiftBot is a web crawler that scrapes the internet for publicly available images to support their suite of web intelligence products",
        "ai_robots_txt_respect": "[Yes](https://imagesift.com/about)",
        "in_crawl_sweep": true,
        "note": "SURPRISE for our own data: 'If there is no rule targeting ImagesiftBot, but there is a rule targeting Googlebot, then ImagesiftBot will follow the Googlebot directives', even under 'User-agent: * Disallow: /'. Our effective_blocks counts wildcard disallows as blocking it (49 vs 36 named); sites with an open Googlebot group are not in fact blocked. Page now branded 'Imagesift by Hive'. purpose search_index: reverse image search index for intelligence products, training use neither stated nor excluded.",
        "operator": "ImageSift",
        "operator_doc_url": "https://imagesift.com/about",
        "orgs": [
          "ImageSift"
        ],
        "purpose": "search_index",
        "purpose_operator_words": "ImageSiftBot is a web crawler that scrapes the internet for publicly available images to support our suite of web intelligence products",
        "respects_robots": true,
        "respects_robots_basis": "ImageSift: 'Standard directives in robots.txt that target ImagesiftBot are respected.' Supports crawl-delay.",
        "separate_from_search_basis": "No consumer search product.",
        "sites_effectively_blocking": 49,
        "sites_named_blocking": 36,
        "sites_readable": 245,
        "url": "https://imagesift.com/about",
        "user_agent": "ImagesiftBot",
        "verification": []
      },
      "id": "crawler:imagesiftbot",
      "name": "ImagesiftBot",
      "notes": {
        "_org_roles": {
          "ImageSift": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "imagesift.com"
        ],
        "source_urls": [
          "https://imagesift.com/about"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.714,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://imagesift.com/about"
    },
    {
      "fields": {
        "ai_robots_txt_function": "AI Data Scrapers",
        "ai_robots_txt_respect": "[Yes](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/)",
        "disagreement": "field: respects_robots; ai_robots_txt: Yes (links to Meta's page); operator: page makes no such statement at retrieval",
        "in_crawl_sweep": true,
        "note": "Operator page returned HTTP 400 to curl and was read via a fetcher; Meta's page gave no IP list, ASN or date, so ip_ranges_url and first_documented null. The robots.txt token Meta documents is Meta-ExternalAgent (case-insensitive match).",
        "operator": "Meta",
        "operator_doc_url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
        "orgs": [
          "Meta"
        ],
        "purpose": "mixed",
        "purpose_operator_words": "crawls the web for use cases such as training foundation AI models or improving products by indexing content directly",
        "respects_robots_basis": "Meta's crawler page states no robots.txt policy for this agent.",
        "separate_from_search_basis": "Meta does not document a web search product tied to this agent.",
        "sites_effectively_blocking": 68,
        "sites_named_blocking": 56,
        "sites_readable": 245,
        "url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers",
        "user_agent": "meta-externalagent",
        "verification": []
      },
      "id": "crawler:meta-externalagent",
      "name": "meta-externalagent",
      "notes": {
        "_org_roles": {
          "Meta": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.facebook.com"
        ],
        "source_urls": [
          "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.571,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "C",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers"
    },
    {
      "fields": {
        "ai_robots_txt_function": "AI Assistants",
        "ai_robots_txt_respect": "[No](https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/)",
        "in_crawl_sweep": false,
        "note": "Not in our crawl_block list. Note the purpose includes 'evaluating and improving agentic AI capabilities', i.e. user fetches can feed product improvement.",
        "operator": "Meta",
        "operator_doc_url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
        "orgs": [
          "Meta"
        ],
        "purpose": "user_fetch",
        "purpose_operator_words": "fetches individual links at a user's request and supports product functions such as evaluating and improving agentic AI capabilities",
        "respects_robots": false,
        "respects_robots_basis": "Meta: it 'may bypass robots.txt rules'.",
        "separate_from_search_basis": "No Meta search product documented.",
        "url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers",
        "user_agent": "Meta-ExternalFetcher",
        "verification": []
      },
      "id": "crawler:meta-externalfetcher",
      "name": "Meta-ExternalFetcher",
      "notes": {
        "_org_roles": {
          "Meta": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.facebook.com"
        ],
        "source_urls": [
          "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.714,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Indexes web content for Mistral AI's search, used to answer questions in Le Chat. Per Mistral, not used for model training.",
        "ai_robots_txt_respect": "Yes",
        "in_crawl_sweep": false,
        "ip_ranges_url": "https://mistral.ai/mistralai-index-ips.json",
        "note": "Not in our crawl_block list.",
        "operator": "Mistral AI",
        "operator_doc_url": "https://docs.mistral.ai/robots/",
        "opt_out_token_separate_from_search": true,
        "orgs": [
          "Mistral AI"
        ],
        "purpose": "search_index",
        "purpose_operator_words": "MistralAI-Index is for automated crawling of the web for indexing purposes only. It indexes content for Mistral search... Content crawled by MistralAI-Index is not used for generative AI training of any kind.",
        "respects_robots_basis": "No explicit robots.txt compliance sentence for this agent.",
        "separate_from_search_basis": "It is Mistral's search index crawler.",
        "url": "https://docs.mistral.ai/robots",
        "user_agent": "MistralAI-Index",
        "verification": [
          "published_ip_ranges"
        ]
      },
      "id": "crawler:mistralai-index",
      "name": "MistralAI-Index",
      "notes": {
        "_org_roles": {
          "Mistral AI": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "docs.mistral.ai"
        ],
        "source_urls": [
          "https://docs.mistral.ai/robots"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.857,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://docs.mistral.ai/robots"
    },
    {
      "fields": {
        "ai_robots_txt_function": "AI Data Scrapers",
        "ai_robots_txt_respect": "[Yes](https://docs.mistral.ai/robots/)",
        "in_crawl_sweep": false,
        "note": "Not in our crawl_block list. No IP list published for the training crawler, unlike its user and index agents.",
        "operator": "Mistral AI",
        "operator_doc_url": "https://docs.mistral.ai/robots/",
        "opt_out_token_separate_from_search": false,
        "orgs": [
          "Mistral AI"
        ],
        "purpose": "training",
        "purpose_operator_words": "MistralAI-Training crawls web content to help build datasets for training Mistral generative AI models. Webmasters can disallow this user agent in their robots.txt file.",
        "respects_robots": true,
        "respects_robots_basis": "Mistral: 'Webmasters can disallow this user agent in their robots.txt file.'",
        "separate_from_search_basis": "Mistral: 'This crawler is not used for search indexing or to answer live user queries in Vibe.'",
        "url": "https://docs.mistral.ai/robots",
        "user_agent": "MistralAI-Training",
        "verification": []
      },
      "id": "crawler:mistralai-training",
      "name": "MistralAI-Training",
      "notes": {
        "_org_roles": {
          "Mistral AI": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "docs.mistral.ai"
        ],
        "source_urls": [
          "https://docs.mistral.ai/robots"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.857,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://docs.mistral.ai/robots"
    },
    {
      "fields": {
        "ai_robots_txt_function": "AI Assistants",
        "ai_robots_txt_respect": "Unclear at this time.",
        "in_crawl_sweep": false,
        "ip_ranges_url": "https://mistral.ai/mistralai-user-ips.json",
        "note": "Not in our crawl_block list. Product renamed Le Chat to Vibe on the operator page; ai.robots.txt still says Le Chat.",
        "operator": "Mistral AI",
        "operator_doc_url": "https://docs.mistral.ai/robots/",
        "orgs": [
          "Mistral AI"
        ],
        "purpose": "user_fetch",
        "purpose_operator_words": "MistralAI-User is for user actions in Vibe. When users ask Vibe a question, it may visit a web page... It is not used for crawling the web in any automatic fashion, nor to crawl content for generative AI training.",
        "respects_robots_basis": "Mistral says the agent 'governs which sites these user requests can be made to' but states no robots.txt compliance commitment.",
        "separate_from_search_basis": "Not stated.",
        "url": "https://docs.mistral.ai/robots",
        "user_agent": "MistralAI-User",
        "verification": [
          "published_ip_ranges"
        ]
      },
      "id": "crawler:mistralai-user",
      "name": "MistralAI-User",
      "notes": {
        "_org_roles": {
          "Mistral AI": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "docs.mistral.ai"
        ],
        "source_urls": [
          "https://docs.mistral.ai/robots"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.714,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://docs.mistral.ai/robots"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Search result generation.",
        "ai_robots_txt_respect": "[Yes](https://platform.openai.com/docs/bots)",
        "disagreement": "field: purpose; ai_robots_txt: Crawls sites to surface as results in SearchGPT (product name is stale); operator: ChatGPT search features",
        "in_crawl_sweep": true,
        "ip_ranges_url": "https://openai.com/searchbot.json",
        "note": "first_documented null: no date on operator page. The least blocked OpenAI agent in our rows: publishers block the training bot and keep the one that sends traffic.",
        "operator": "OpenAI",
        "operator_doc_url": "https://developers.openai.com/api/docs/bots",
        "opt_out_token_separate_from_search": true,
        "orgs": [
          "OpenAI"
        ],
        "purpose": "search_index",
        "purpose_operator_words": "OAI-SearchBot is used to surface websites in search results in ChatGPT's search features. Sites that are opted out of OAI-SearchBot will not be shown in ChatGPT search answers, though can still appear as navigational links.",
        "respects_robots": true,
        "respects_robots_basis": "OpenAI doc describes OAI-SearchBot as a robots.txt tag and says robots.txt changes take ~24 hours to apply for search. Operator statement.",
        "separate_from_search_basis": "Blocking OAI-SearchBot is the search opt-out itself: 'Sites that are opted out of OAI-SearchBot will not be shown in ChatGPT search answers'.",
        "sites_effectively_blocking": 38,
        "sites_named_blocking": 30,
        "sites_readable": 245,
        "url": "https://developers.openai.com/api/docs/bots",
        "user_agent": "OAI-SearchBot",
        "verification": [
          "published_ip_ranges"
        ]
      },
      "id": "crawler:oai-searchbot",
      "name": "OAI-SearchBot",
      "notes": {
        "_org_roles": {
          "OpenAI": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.openai.com"
        ],
        "source_urls": [
          "https://developers.openai.com/api/docs/bots"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://developers.openai.com/api/docs/bots"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Data is sold.",
        "ai_robots_txt_respect": "[Yes](https://webz.io/blog/web-data/what-is-the-omgili-bot-and-why-is-it-crawling-your-website/)",
        "first_documented": "2017-12-28",
        "in_crawl_sweep": true,
        "note": "Superseded in 2024 by the 'Webzio Duo' (Webzio + Webzio-Extended) per Webz.io's own post; 60 of our sites still block omgili. Data is resold, including for LLM training per ai.robots.txt. No IP list.",
        "operator": "Webz.io",
        "operator_doc_url": "https://webz.io/blog/web-data/what-is-the-omgili-bot-and-why-is-it-crawling-your-website/",
        "orgs": [
          "Webz.io"
        ],
        "purpose": "mixed",
        "purpose_operator_words": "the Omgili Bot is a web crawler we developed a decade ago to power the (now discontinued) Omgili search engine. Today this bot powers Webz.io, a web crawling service used by the world's leading media monitors and research institutes",
        "respects_robots": true,
        "respects_robots_basis": "Webz.io (2017): 'you can tell us directly or through your robots.txt file... and always comply with these requests.'",
        "separate_from_search_basis": "Omgili search engine is discontinued; not applicable.",
        "sites_effectively_blocking": 60,
        "sites_named_blocking": 46,
        "sites_readable": 245,
        "url": "https://webz.io/blog/web-data/what-is-the-omgili-bot-and-why-is-it-crawling-your-website",
        "user_agent": "omgili",
        "verification": []
      },
      "id": "crawler:omgili",
      "name": "omgili",
      "notes": {
        "_org_roles": {
          "Webz.io": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "webz.io"
        ],
        "source_urls": [
          "https://webz.io/blog/web-data/what-is-the-omgili-bot-and-why-is-it-crawling-your-website"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.714,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://webz.io/blog/web-data/what-is-the-omgili-bot-and-why-is-it-crawling-your-website"
    },
    {
      "fields": {
        "ai_robots_txt_function": "AI Assistants",
        "ai_robots_txt_respect": "[No](https://docs.perplexity.ai/guides/bots)",
        "in_crawl_sweep": true,
        "ip_ranges_url": "https://www.perplexity.com/perplexity-user.json",
        "measured_by_third_party": "{\"source\": \"https://blog.cloudflare.com/perplexity-is-using-stealth-undeclared-crawlers-to-evade-website-no-crawl-directives/\", \"claim\": \"Cloudflare, 2025-08-04: declared Perplexity-User/1.0 traffic of 20-25m daily requests.\"}",
        "note": "Also odd: Perplexity says 'Perplexity-User controls which sites these user requests can access' and in the next line that it 'generally ignores robots.txt rules'. A robots.txt block on it is a signal only.",
        "operator": "Perplexity AI",
        "operator_doc_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
        "orgs": [
          "Perplexity AI"
        ],
        "purpose": "user_fetch",
        "purpose_operator_words": "When users ask Perplexity a question, it might visit a web page to help provide an accurate answer and include a link to the page in its response... It is not used for web crawling or to collect content for training AI foundation models.",
        "respects_robots": false,
        "respects_robots_basis": "Perplexity doc: 'Since a user requested the fetch, this fetcher generally ignores robots.txt rules.' So the Disallow our 28 sites wrote for it is, per the operator, not honoured.",
        "separate_from_search_basis": "Operator does not say whether blocking it affects PerplexityBot search inclusion.",
        "sites_effectively_blocking": 41,
        "sites_named_blocking": 28,
        "sites_readable": 245,
        "url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
        "user_agent": "Perplexity-User",
        "verification": [
          "published_ip_ranges"
        ]
      },
      "id": "crawler:perplexity-user",
      "name": "Perplexity-User",
      "notes": {
        "_org_roles": {
          "Perplexity AI": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "docs.perplexity.ai"
        ],
        "source_urls": [
          "https://docs.perplexity.ai/docs/resources/perplexity-crawlers"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.857,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Search result generation.",
        "ai_robots_txt_respect": "[Yes](https://docs.perplexity.ai/guides/bots)",
        "disagreement": "field: respects_robots; operator: declared bots honour robots.txt; cloudflare_measured: undeclared Perplexity traffic evaded no-crawl directives",
        "in_crawl_sweep": true,
        "ip_ranges_url": "https://www.perplexity.com/perplexitybot.json",
        "measured_by_third_party": "{\"source\": \"https://blog.cloudflare.com/perplexity-is-using-stealth-undeclared-crawlers-to-evade-website-no-crawl-directives/\", \"claim\": \"Cloudflare, 2025-08-04: Perplexity also used an undeclared crawler impersonating Chrome on macOS (3-6m daily requests) from rotating IPs/ASNs, and returned content from fresh test domains that disallowed it; Cloudflare de-listed Perplexity as a verified bot.\"}",
        "note": "Operator doc and third-party measurement conflict; both kept. The measured evasion was by undeclared agents, not the PerplexityBot string itself. first_documented null: no date.",
        "operator": "Perplexity AI",
        "operator_doc_url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
        "opt_out_token_separate_from_search": true,
        "orgs": [
          "Perplexity AI"
        ],
        "purpose": "search_index",
        "purpose_operator_words": "PerplexityBot is designed to surface and link websites in search results on Perplexity. It is not used to crawl content for AI foundation models.",
        "respects_robots": true,
        "respects_robots_basis": "Perplexity doc presents it as a robots.txt tag and says changes take up to 24 hours. Operator statement.",
        "separate_from_search_basis": "It is the search crawler: 'To ensure your site appears in search results, we recommend allowing PerplexityBot'.",
        "sites_effectively_blocking": 61,
        "sites_named_blocking": 51,
        "sites_readable": 245,
        "url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers",
        "user_agent": "PerplexityBot",
        "verification": [
          "published_ip_ranges"
        ]
      },
      "id": "crawler:perplexitybot",
      "name": "PerplexityBot",
      "notes": {
        "_org_roles": {
          "Perplexity AI": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "docs.perplexity.ai"
        ],
        "source_urls": [
          "https://docs.perplexity.ai/docs/resources/perplexity-crawlers"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://docs.perplexity.ai/docs/resources/perplexity-crawlers"
    },
    {
      "fields": {
        "ai_robots_txt_function": "Scrapes data for use in training LLMs.",
        "ai_robots_txt_respect": "Unclear at this time.",
        "disagreement": "field: purpose; ai_robots_txt: Scrapes data for use in training LLMs; respect Unclear; operator: silent",
        "in_crawl_sweep": true,
        "note": "51 of our sites block an operator that publishes nothing about the bot.",
        "operator": "Timpi",
        "orgs": [
          "Timpi"
        ],
        "purpose": null,
        "respects_robots_basis": "No operator crawler page: timpi.io/crawler returns 404 and timpi.io home page does not mention the bot.",
        "separate_from_search_basis": "Not documented.",
        "sites_effectively_blocking": 51,
        "sites_named_blocking": 37,
        "sites_readable": 245,
        "url": "https://github.com/ai-robots-txt/ai.robots.txt/blob/48abe91f99318da1968ce0e038b731fc38051b8f/robots.json",
        "user_agent": "Timpibot",
        "verification": []
      },
      "id": "crawler:timpibot",
      "name": "Timpibot",
      "notes": {
        "_org_roles": {
          "Timpi": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "github.com"
        ],
        "source_urls": [
          "https://github.com/ai-robots-txt/ai.robots.txt/blob/48abe91f99318da1968ce0e038b731fc38051b8f/robots.json"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.429,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source",
          "sparse"
        ],
        "grade": "C",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://github.com/ai-robots-txt/ai.robots.txt/blob/48abe91f99318da1968ce0e038b731fc38051b8f/robots.json"
    },
    {
      "fields": {
        "ai_robots_txt_function": "AI Data Scrapers",
        "ai_robots_txt_respect": "Unclear at this time.",
        "disagreement": "field: operator; ai_robots_txt: Unclear at this time; operator: Webz.io (own blog post)",
        "first_documented": "2024-07-17",
        "in_crawl_sweep": true,
        "note": "Works like Google-Extended: an AI-use flag over data the Webzio crawler collects. The Webzio crawler itself is not in our list of 22. Source is a marketing blog aimed at data buyers, not a crawler spec page.",
        "operator": "Webz.io",
        "operator_doc_url": "https://webz.io/blog/company/from-omgilibot-to-the-webzbot-duo-a-powerful-leap-for-ethical-and-comprehensive-data-collection/",
        "orgs": [
          "Webz.io"
        ],
        "purpose": "training",
        "purpose_operator_words": "Webzio-extended ... takes the data collected by Webzio and performs a critical function: ethical validation... tagging the data as usable or not usable for AI/ML training.",
        "respects_robots": true,
        "respects_robots_basis": "Webz.io: 'The Webzio Duo meticulously adheres to robots.txt exclusions.'",
        "separate_from_search_basis": "Webz.io runs no consumer search; not applicable.",
        "sites_effectively_blocking": 43,
        "sites_named_blocking": 29,
        "sites_readable": 245,
        "url": "https://webz.io/blog/company/from-omgilibot-to-the-webzbot-duo-a-powerful-leap-for-ethical-and-comprehensive-data-collection",
        "user_agent": "Webzio-Extended",
        "verification": [
          "not_applicable_control_token"
        ]
      },
      "id": "crawler:webzio-extended",
      "name": "Webzio-Extended",
      "notes": {
        "_org_roles": {
          "Webz.io": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "webz.io"
        ],
        "source_urls": [
          "https://webz.io/blog/company/from-omgilibot-to-the-webzbot-duo-a-powerful-leap-for-ethical-and-comprehensive-data-collection"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.857,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "crawler",
      "url": "https://webz.io/blog/company/from-omgilibot-to-the-webzbot-duo-a-powerful-leap-for-ethical-and-comprehensive-data-collection"
    }
  ],
  "_meta": {
    "source": "Blomega Data Refinery",
    "url": "https://data.blomega.com",
    "publisher": "Blomega",
    "publisher_url": "https://blomegalab.com",
    "wikidata": "Q141048865",
    "license": "CC BY 4.0",
    "license_url": "https://creativecommons.org/licenses/by/4.0/",
    "cite_as": "Blomega Data Refinery (https://data.blomega.com), CC BY 4.0. Cite the registry and the record id.",
    "attribution_required": true
  }
}
