{
  "total": 256,
  "limit": 50,
  "offset": 0,
  "records": [
    {
      "fields": {
        "allows_all": true,
        "bucket": "stock_media",
        "domain": "500px.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow. tdmrep.json HTTP 200",
        "readable": true,
        "robots_bytes": 100,
        "url": "https://500px.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:500px-com",
      "name": "500px.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "500px.com"
        ],
        "source_urls": [
          "https://500px.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://500px.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "ai_company",
        "domain": "adobe.com",
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "defendant",
          "licensee"
        ],
        "robots_bytes": 10199,
        "url": "https://adobe.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:adobe-com",
      "name": "adobe.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "adobe.com"
        ],
        "source_urls": [
          "https://adobe.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.727,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://adobe.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "news_publisher",
        "domain": "advancelocal.com",
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "plaintiff"
        ],
        "robots_bytes": 178,
        "sued": [
          "Cohere"
        ],
        "url": "https://advancelocal.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:advancelocal-com",
      "name": "advancelocal.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "advancelocal.com"
        ],
        "source_urls": [
          "https://advancelocal.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.727,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://advancelocal.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "music",
        "domain": "afm.org",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "plaintiff"
        ],
        "robots_bytes": 302,
        "sued": [
          "Atlantic Recording Corp.",
          "Universal Music Group",
          "Warner Records"
        ],
        "url": "https://afm.org/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:afm-org",
      "name": "afm.org",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "afm.org"
        ],
        "source_urls": [
          "https://afm.org/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://afm.org/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "news_publisher",
        "domain": "afp.com",
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "licensor"
        ],
        "robots_bytes": 2027,
        "signed_licensees": [
          "Mistral AI"
        ],
        "url": "https://www.afp.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:afp-com",
      "name": "afp.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "afp.com"
        ],
        "source_urls": [
          "https://www.afp.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.727,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://www.afp.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "ImageSift"
        ],
        "blocks": [
          "ImagesiftBot"
        ],
        "bucket": "ecommerce",
        "domain": "airbnb.com",
        "effective_blocked_operators": [
          "ImageSift"
        ],
        "effective_blocks": [
          "ImagesiftBot"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "named_but_allowed": [
          "GPTBot",
          "ChatGPT-User",
          "OAI-SearchBot",
          "ClaudeBot",
          "anthropic-ai",
          "PerplexityBot",
          "meta-externalagent",
          "cohere-ai"
        ],
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 24175,
        "url": "https://airbnb.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:airbnb-com",
      "name": "airbnb.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "airbnb.com"
        ],
        "source_urls": [
          "https://airbnb.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://airbnb.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "stock_media",
        "domain": "alamy.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "named_but_allowed": [
          "GPTBot",
          "ClaudeBot",
          "CCBot",
          "Bytespider",
          "Applebot-Extended",
          "PerplexityBot",
          "Perplexity-User",
          "AI2Bot"
        ],
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 11706,
        "url": "https://alamy.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:alamy-com",
      "name": "alamy.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "alamy.com"
        ],
        "source_urls": [
          "https://alamy.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://alamy.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "Google",
          "OpenAI"
        ],
        "blocks": [
          "GPTBot",
          "Google-Extended"
        ],
        "bucket": "ecommerce",
        "domain": "alibaba.com",
        "effective_blocked_operators": [
          "Google",
          "OpenAI"
        ],
        "effective_blocks": [
          "GPTBot",
          "Google-Extended"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow. tdmrep.json HTTP 200",
        "readable": true,
        "robots_bytes": 5406,
        "url": "https://alibaba.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:alibaba-com",
      "name": "alibaba.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "alibaba.com"
        ],
        "source_urls": [
          "https://alibaba.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://alibaba.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "ecommerce",
        "domain": "aliexpress.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow. tdmrep.json HTTP 200",
        "readable": true,
        "robots_bytes": 1971,
        "url": "https://aliexpress.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:aliexpress-com",
      "name": "aliexpress.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "aliexpress.com"
        ],
        "source_urls": [
          "https://aliexpress.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://aliexpress.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "Anthropic",
          "ByteDance",
          "Cohere",
          "OpenAI",
          "Perplexity AI"
        ],
        "blocks": [
          "GPTBot",
          "ChatGPT-User",
          "ClaudeBot",
          "anthropic-ai",
          "Bytespider",
          "PerplexityBot",
          "cohere-ai"
        ],
        "bucket": "news_publisher",
        "domain": "aljazeera.com",
        "effective_blocked_operators": [
          "Anthropic",
          "ByteDance",
          "Cohere",
          "OpenAI",
          "Perplexity AI"
        ],
        "effective_blocks": [
          "GPTBot",
          "ChatGPT-User",
          "ClaudeBot",
          "anthropic-ai",
          "Bytespider",
          "PerplexityBot",
          "cohere-ai"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 1834,
        "url": "https://aljazeera.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:aljazeera-com",
      "name": "aljazeera.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "aljazeera.com"
        ],
        "source_urls": [
          "https://aljazeera.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://aljazeera.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "Allen Institute for AI",
          "Anthropic",
          "ByteDance",
          "Cohere",
          "Common Crawl",
          "Diffbot",
          "Google",
          "ImageSift",
          "Meta",
          "OpenAI",
          "Perplexity AI",
          "Timpi",
          "Webz.io"
        ],
        "blocks": [
          "GPTBot",
          "ChatGPT-User",
          "OAI-SearchBot",
          "ClaudeBot",
          "Claude-User",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "PerplexityBot",
          "Perplexity-User",
          "meta-externalagent",
          "cohere-ai",
          "Diffbot",
          "ImagesiftBot",
          "omgili",
          "Timpibot",
          "AI2Bot",
          "Webzio-Extended"
        ],
        "bucket": "ecommerce",
        "domain": "amazon.com",
        "effective_blocked_operators": [
          "Allen Institute for AI",
          "Anthropic",
          "ByteDance",
          "Cohere",
          "Common Crawl",
          "Diffbot",
          "Google",
          "ImageSift",
          "Meta",
          "OpenAI",
          "Perplexity AI",
          "Timpi",
          "Webz.io"
        ],
        "effective_blocks": [
          "GPTBot",
          "ChatGPT-User",
          "OAI-SearchBot",
          "ClaudeBot",
          "Claude-User",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "PerplexityBot",
          "Perplexity-User",
          "meta-externalagent",
          "cohere-ai",
          "Diffbot",
          "ImagesiftBot",
          "omgili",
          "Timpibot",
          "AI2Bot",
          "Webzio-Extended"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "defendant",
          "licensee"
        ],
        "robots_bytes": 7887,
        "url": "https://amazon.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:amazon-com",
      "name": "amazon.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "amazon.com"
        ],
        "source_urls": [
          "https://amazon.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://amazon.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "news_publisher",
        "domain": "aninews.in",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "plaintiff"
        ],
        "robots_bytes": 212,
        "sued": [
          "OpenAI"
        ],
        "url": "https://aninews.in/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:aninews-in",
      "name": "aninews.in",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "aninews.in"
        ],
        "source_urls": [
          "https://aninews.in/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://aninews.in/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "ai_company",
        "domain": "anthropic.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "defendant"
        ],
        "robots_bytes": 71,
        "url": "https://anthropic.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:anthropic-com",
      "name": "anthropic.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "anthropic.com"
        ],
        "source_urls": [
          "https://anthropic.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://anthropic.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "Amazon",
          "Anthropic",
          "Apple",
          "Cohere",
          "Common Crawl",
          "OpenAI",
          "Perplexity AI",
          "Timpi"
        ],
        "blocks": [
          "GPTBot",
          "ClaudeBot",
          "Claude-User",
          "anthropic-ai",
          "CCBot",
          "Amazonbot",
          "Applebot-Extended",
          "PerplexityBot",
          "cohere-ai",
          "Timpibot"
        ],
        "bucket": "news_publisher",
        "domain": "apnews.com",
        "effective_blocked_operators": [
          "Amazon",
          "Anthropic",
          "Apple",
          "Cohere",
          "Common Crawl",
          "OpenAI",
          "Perplexity AI",
          "Timpi"
        ],
        "effective_blocks": [
          "GPTBot",
          "ClaudeBot",
          "Claude-User",
          "anthropic-ai",
          "CCBot",
          "Amazonbot",
          "Applebot-Extended",
          "PerplexityBot",
          "cohere-ai",
          "Timpibot"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "licensor"
        ],
        "robots_bytes": 1293,
        "signed_licensees": [
          "Google",
          "Microsoft",
          "OpenAI"
        ],
        "url": "https://apnews.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:apnews-com",
      "name": "apnews.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "apnews.com"
        ],
        "source_urls": [
          "https://apnews.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://apnews.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "ai_company",
        "domain": "apple.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "defendant",
          "licensee"
        ],
        "robots_bytes": 1018,
        "url": "https://apple.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:apple-com",
      "name": "apple.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "apple.com"
        ],
        "source_urls": [
          "https://apple.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://apple.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "ugc_platform",
        "domain": "archive.org",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 238,
        "url": "https://archive.org/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:archive-org",
      "name": "archive.org",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "archive.org"
        ],
        "source_urls": [
          "https://archive.org/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://archive.org/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "Common Crawl",
          "OpenAI"
        ],
        "blocks": [
          "GPTBot",
          "ChatGPT-User",
          "CCBot"
        ],
        "bucket": "ugc_platform",
        "domain": "archiveofourown.org",
        "effective_blocked_operators": [
          "Common Crawl",
          "OpenAI"
        ],
        "effective_blocks": [
          "GPTBot",
          "ChatGPT-User",
          "CCBot"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 872,
        "url": "https://archiveofourown.org/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:archiveofourown-org",
      "name": "archiveofourown.org",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "archiveofourown.org"
        ],
        "source_urls": [
          "https://archiveofourown.org/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://archiveofourown.org/robots.txt"
    },
    {
      "fields": {
        "bucket": "news_publisher",
        "domain": "arstechnica.com",
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow. robots.txt could not be read from five client profiles over two hostnames (statuses {'chrome': 405, 'curl': 403, 'firefox-h1': 405, 'chrome-full': 405, 'named': 405}); the edge refuses unrecognised clients, which gates crawlers harder than robots.txt does Not readable: the site's edge refused every client profile tried. This is recorded as its own state, never as allowing every crawler.",
        "readable": false,
        "url": "https://arstechnica.com/robots.txt"
      },
      "id": "crawl_block:arstechnica-com",
      "name": "arstechnica.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "arstechnica.com"
        ],
        "source_urls": [
          "https://arstechnica.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.455,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source",
          "sparse"
        ],
        "grade": "C",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://arstechnica.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "stock_media",
        "domain": "artstation.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow. tdmrep.json HTTP 200",
        "readable": true,
        "robots_bytes": 927,
        "url": "https://artstation.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:artstation-com",
      "name": "artstation.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "artstation.com"
        ],
        "source_urls": [
          "https://artstation.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://artstation.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "academic_publisher",
        "domain": "arxiv.org",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 5854,
        "url": "https://arxiv.org/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:arxiv-org",
      "name": "arxiv.org",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "arxiv.org"
        ],
        "source_urls": [
          "https://arxiv.org/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://arxiv.org/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "music",
        "domain": "ascap.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 371,
        "url": "https://ascap.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:ascap-com",
      "name": "ascap.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "ascap.com"
        ],
        "source_urls": [
          "https://ascap.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://ascap.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "trade_body",
        "domain": "authorsguild.org",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "plaintiff"
        ],
        "robots_bytes": 248,
        "sued": [
          "Microsoft",
          "OpenAI"
        ],
        "url": "https://authorsguild.org/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:authorsguild-org",
      "name": "authorsguild.org",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "authorsguild.org"
        ],
        "source_urls": [
          "https://authorsguild.org/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://authorsguild.org/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "ugc_platform",
        "domain": "automattic.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "licensor"
        ],
        "robots_bytes": 628,
        "signed_licensees": [
          "OpenAI and Midjourney"
        ],
        "url": "https://automattic.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:automattic-com",
      "name": "automattic.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "automattic.com"
        ],
        "source_urls": [
          "https://automattic.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://automattic.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "news_publisher",
        "domain": "axelspringer.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "licensor"
        ],
        "robots_bytes": 121,
        "signed_licensees": [
          "OpenAI"
        ],
        "url": "https://axelspringer.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:axelspringer-com",
      "name": "axelspringer.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "axelspringer.com"
        ],
        "source_urls": [
          "https://axelspringer.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://axelspringer.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "Amazon",
          "ByteDance",
          "Common Crawl",
          "Diffbot",
          "ImageSift",
          "Meta"
        ],
        "blocks": [
          "CCBot",
          "Bytespider",
          "Amazonbot",
          "FacebookBot",
          "Diffbot",
          "ImagesiftBot"
        ],
        "bucket": "news_publisher",
        "domain": "axios.com",
        "effective_blocked_operators": [
          "Amazon",
          "ByteDance",
          "Common Crawl",
          "Diffbot",
          "ImageSift",
          "Meta"
        ],
        "effective_blocks": [
          "CCBot",
          "Bytespider",
          "Amazonbot",
          "FacebookBot",
          "Diffbot",
          "ImagesiftBot"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "named_but_allowed": [
          "Google-Extended"
        ],
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "licensor"
        ],
        "robots_bytes": 645,
        "signed_licensees": [
          "OpenAI"
        ],
        "url": "https://axios.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:axios-com",
      "name": "axios.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "axios.com"
        ],
        "source_urls": [
          "https://axios.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://axios.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "Amazon",
          "Anthropic",
          "ByteDance",
          "Common Crawl",
          "Diffbot",
          "Google",
          "ImageSift",
          "Meta",
          "OpenAI",
          "Webz.io"
        ],
        "blocks": [
          "GPTBot",
          "ClaudeBot",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "Amazonbot",
          "meta-externalagent",
          "FacebookBot",
          "Diffbot",
          "ImagesiftBot",
          "omgili"
        ],
        "bucket": "ugc_platform",
        "domain": "bandcamp.com",
        "effective_blocked_operators": [
          "Amazon",
          "Anthropic",
          "ByteDance",
          "Common Crawl",
          "Diffbot",
          "Google",
          "ImageSift",
          "Meta",
          "OpenAI",
          "Webz.io"
        ],
        "effective_blocks": [
          "GPTBot",
          "ClaudeBot",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "Amazonbot",
          "meta-externalagent",
          "FacebookBot",
          "Diffbot",
          "ImagesiftBot",
          "omgili"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 1053,
        "url": "https://bandcamp.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:bandcamp-com",
      "name": "bandcamp.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "bandcamp.com"
        ],
        "source_urls": [
          "https://bandcamp.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://bandcamp.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "Amazon",
          "Anthropic",
          "Apple",
          "ByteDance",
          "Cohere",
          "Common Crawl",
          "Diffbot",
          "Google",
          "Meta",
          "OpenAI",
          "Perplexity AI",
          "Webz.io"
        ],
        "blocks": [
          "GPTBot",
          "ChatGPT-User",
          "OAI-SearchBot",
          "ClaudeBot",
          "anthropic-ai",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "Amazonbot",
          "Applebot-Extended",
          "PerplexityBot",
          "Perplexity-User",
          "meta-externalagent",
          "cohere-ai",
          "Diffbot",
          "omgili"
        ],
        "bucket": "news_publisher",
        "domain": "bbc.co.uk",
        "effective_blocked_operators": [
          "Amazon",
          "Anthropic",
          "Apple",
          "ByteDance",
          "Cohere",
          "Common Crawl",
          "Diffbot",
          "Google",
          "Meta",
          "OpenAI",
          "Perplexity AI",
          "Webz.io"
        ],
        "effective_blocks": [
          "GPTBot",
          "ChatGPT-User",
          "OAI-SearchBot",
          "ClaudeBot",
          "anthropic-ai",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "Amazonbot",
          "Applebot-Extended",
          "PerplexityBot",
          "Perplexity-User",
          "meta-externalagent",
          "cohere-ai",
          "Diffbot",
          "omgili"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 4907,
        "url": "https://bbc.co.uk/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:bbc-co-uk",
      "name": "bbc.co.uk",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "bbc.co.uk"
        ],
        "source_urls": [
          "https://bbc.co.uk/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://bbc.co.uk/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "Allen Institute for AI",
          "Anthropic",
          "Apple",
          "ByteDance",
          "Common Crawl",
          "Diffbot",
          "Google",
          "ImageSift",
          "Meta",
          "OpenAI",
          "Timpi",
          "Webz.io"
        ],
        "blocks": [
          "GPTBot",
          "ClaudeBot",
          "anthropic-ai",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "Applebot-Extended",
          "meta-externalagent",
          "FacebookBot",
          "Diffbot",
          "ImagesiftBot",
          "omgili",
          "Timpibot",
          "AI2Bot",
          "Webzio-Extended"
        ],
        "bucket": "ugc_platform",
        "domain": "behance.net",
        "effective_blocked_operators": [
          "Allen Institute for AI",
          "Anthropic",
          "Apple",
          "ByteDance",
          "Common Crawl",
          "Diffbot",
          "Google",
          "ImageSift",
          "Meta",
          "OpenAI",
          "Timpi",
          "Webz.io"
        ],
        "effective_blocks": [
          "GPTBot",
          "ClaudeBot",
          "anthropic-ai",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "Applebot-Extended",
          "meta-externalagent",
          "FacebookBot",
          "Diffbot",
          "ImagesiftBot",
          "omgili",
          "Timpibot",
          "AI2Bot",
          "Webzio-Extended"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "licensee"
        ],
        "robots_bytes": 1317,
        "url": "https://behance.net/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:behance-net",
      "name": "behance.net",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "behance.net"
        ],
        "source_urls": [
          "https://behance.net/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://behance.net/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "music",
        "domain": "believe.com",
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "licensor"
        ],
        "robots_bytes": 189,
        "signed_licensees": [
          "Suno"
        ],
        "url": "https://believe.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:believe-com",
      "name": "believe.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "believe.com"
        ],
        "source_urls": [
          "https://believe.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.727,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://believe.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "ecommerce",
        "domain": "bestbuy.com",
        "http_status": 200,
        "measured_at": "2026-09-17",
        "named_but_allowed": [
          "GPTBot",
          "ChatGPT-User",
          "OAI-SearchBot",
          "ClaudeBot",
          "Claude-User",
          "anthropic-ai",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "Amazonbot",
          "PerplexityBot",
          "Perplexity-User"
        ],
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 15304,
        "url": "https://bestbuy.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:bestbuy-com",
      "name": "bestbuy.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "bestbuy.com"
        ],
        "source_urls": [
          "https://bestbuy.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.727,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://bestbuy.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "Amazon",
          "Anthropic",
          "Apple",
          "ByteDance",
          "Cohere",
          "Common Crawl",
          "Diffbot",
          "Google",
          "Meta",
          "OpenAI",
          "Perplexity AI"
        ],
        "blocks": [
          "GPTBot",
          "ChatGPT-User",
          "OAI-SearchBot",
          "ClaudeBot",
          "Claude-User",
          "anthropic-ai",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "Amazonbot",
          "Applebot-Extended",
          "PerplexityBot",
          "meta-externalagent",
          "cohere-ai",
          "Diffbot"
        ],
        "bucket": "news_publisher",
        "domain": "bloomberg.com",
        "effective_blocked_operators": [
          "Amazon",
          "Anthropic",
          "Apple",
          "ByteDance",
          "Cohere",
          "Common Crawl",
          "Diffbot",
          "Google",
          "Meta",
          "OpenAI",
          "Perplexity AI"
        ],
        "effective_blocks": [
          "GPTBot",
          "ChatGPT-User",
          "OAI-SearchBot",
          "ClaudeBot",
          "Claude-User",
          "anthropic-ai",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "Amazonbot",
          "Applebot-Extended",
          "PerplexityBot",
          "meta-externalagent",
          "cohere-ai",
          "Diffbot"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 7048,
        "url": "https://bloomberg.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:bloomberg-com",
      "name": "bloomberg.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "bloomberg.com"
        ],
        "source_urls": [
          "https://bloomberg.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://bloomberg.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "music",
        "domain": "bmg.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "named_but_allowed": [
          "GPTBot",
          "CCBot",
          "PerplexityBot"
        ],
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "licensor"
        ],
        "robots_bytes": 2059,
        "signed_licensees": [
          "Suno"
        ],
        "url": "https://bmg.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:bmg-com",
      "name": "bmg.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "bmg.com"
        ],
        "source_urls": [
          "https://bmg.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://bmg.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "music",
        "domain": "bmi.com",
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow. /robots.txt returns the site's HTML shell, not a robots file: under RFC 9309 an absent robots.txt places no restriction on any crawler",
        "readable": true,
        "robots_bytes": 0,
        "url": "https://bmi.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:bmi-com",
      "name": "bmi.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "bmi.com"
        ],
        "source_urls": [
          "https://bmi.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.727,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://bmi.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "ecommerce",
        "domain": "booking.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 39491,
        "url": "https://booking.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:booking-com",
      "name": "booking.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "booking.com"
        ],
        "source_urls": [
          "https://booking.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://booking.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "ai_company",
        "domain": "brave.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "defendant",
          "plaintiff"
        ],
        "robots_bytes": 53,
        "sued": [
          "Brave Software"
        ],
        "url": "https://brave.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:brave-com",
      "name": "brave.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "brave.com"
        ],
        "source_urls": [
          "https://brave.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://brave.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "ai_company",
        "domain": "bria.ai",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "named_but_allowed": [
          "GPTBot",
          "ChatGPT-User",
          "OAI-SearchBot",
          "ClaudeBot",
          "anthropic-ai",
          "Google-Extended",
          "Amazonbot",
          "Applebot-Extended",
          "PerplexityBot",
          "meta-externalagent",
          "FacebookBot",
          "cohere-ai"
        ],
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "licensee"
        ],
        "robots_bytes": 543,
        "url": "https://bria.ai/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:bria-ai",
      "name": "bria.ai",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "bria.ai"
        ],
        "source_urls": [
          "https://bria.ai/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://bria.ai/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "data_broker",
        "content_signal": {
          "ai-input": "yes",
          "ai-train": "yes",
          "search": "yes"
        },
        "domain": "brightdata.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "named_but_allowed": [
          "GPTBot",
          "ChatGPT-User",
          "ClaudeBot",
          "Google-Extended",
          "CCBot",
          "Applebot-Extended",
          "PerplexityBot"
        ],
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "defendant"
        ],
        "robots_bytes": 609,
        "url": "https://brightdata.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:brightdata-com",
      "name": "brightdata.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "brightdata.com"
        ],
        "source_urls": [
          "https://brightdata.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://brightdata.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "Anthropic",
          "ByteDance",
          "Common Crawl",
          "Diffbot",
          "Timpi"
        ],
        "blocks": [
          "ClaudeBot",
          "anthropic-ai",
          "CCBot",
          "Bytespider",
          "Diffbot",
          "Timpibot"
        ],
        "bucket": "news_publisher",
        "domain": "businessinsider.com",
        "effective_blocked_operators": [
          "Anthropic",
          "ByteDance",
          "Common Crawl",
          "Diffbot",
          "Timpi"
        ],
        "effective_blocks": [
          "ClaudeBot",
          "anthropic-ai",
          "CCBot",
          "Bytespider",
          "Diffbot",
          "Timpibot"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "licensor"
        ],
        "robots_bytes": 2231,
        "signed_licensees": [
          "OpenAI"
        ],
        "url": "https://businessinsider.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:businessinsider-com",
      "name": "businessinsider.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "businessinsider.com"
        ],
        "source_urls": [
          "https://businessinsider.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://businessinsider.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "OpenAI"
        ],
        "blocks": [
          "ChatGPT-User"
        ],
        "bucket": "academic_publisher",
        "domain": "cambridge.org",
        "effective_blocked_operators": [
          "OpenAI"
        ],
        "effective_blocks": [
          "ChatGPT-User"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 2785,
        "url": "https://cambridge.org/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:cambridge-org",
      "name": "cambridge.org",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "cambridge.org"
        ],
        "source_urls": [
          "https://cambridge.org/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://cambridge.org/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "government",
        "domain": "canada.ca",
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 2730,
        "url": "https://canada.ca/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:canada-ca",
      "name": "canada.ca",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "canada.ca"
        ],
        "source_urls": [
          "https://canada.ca/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.727,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://canada.ca/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "Allen Institute for AI",
          "Anthropic",
          "Apple",
          "ByteDance",
          "Common Crawl",
          "Diffbot",
          "ImageSift",
          "Meta",
          "OpenAI",
          "Timpi",
          "Webz.io"
        ],
        "blocks": [
          "GPTBot",
          "ClaudeBot",
          "anthropic-ai",
          "CCBot",
          "Bytespider",
          "Applebot-Extended",
          "meta-externalagent",
          "FacebookBot",
          "Diffbot",
          "ImagesiftBot",
          "omgili",
          "Timpibot",
          "AI2Bot",
          "Webzio-Extended"
        ],
        "bucket": "stock_media",
        "domain": "canva.com",
        "effective_blocked_operators": [
          "Allen Institute for AI",
          "Anthropic",
          "Apple",
          "ByteDance",
          "Common Crawl",
          "Diffbot",
          "ImageSift",
          "Meta",
          "OpenAI",
          "Timpi",
          "Webz.io"
        ],
        "effective_blocks": [
          "GPTBot",
          "ClaudeBot",
          "anthropic-ai",
          "CCBot",
          "Bytespider",
          "Applebot-Extended",
          "meta-externalagent",
          "FacebookBot",
          "Diffbot",
          "ImagesiftBot",
          "omgili",
          "Timpibot",
          "AI2Bot",
          "Webzio-Extended"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "named_but_allowed": [
          "PerplexityBot",
          "Perplexity-User"
        ],
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 6438,
        "url": "https://canva.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:canva-com",
      "name": "canva.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "canva.com"
        ],
        "source_urls": [
          "https://canva.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://canva.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "ai_company",
        "domain": "causaly.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "licensee"
        ],
        "robots_bytes": 78,
        "url": "https://causaly.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:causaly-com",
      "name": "causaly.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "causaly.com"
        ],
        "source_urls": [
          "https://causaly.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://causaly.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "OpenAI"
        ],
        "blocks": [
          "GPTBot"
        ],
        "bucket": "news_publisher",
        "domain": "cbsnews.com",
        "effective_blocked_operators": [
          "OpenAI"
        ],
        "effective_blocks": [
          "GPTBot"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 950,
        "url": "https://cbsnews.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:cbsnews-com",
      "name": "cbsnews.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "cbsnews.com"
        ],
        "source_urls": [
          "https://cbsnews.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://cbsnews.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "government",
        "domain": "cdc.gov",
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 1699,
        "url": "https://cdc.gov/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:cdc-gov",
      "name": "cdc.gov",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "cdc.gov"
        ],
        "source_urls": [
          "https://cdc.gov/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.727,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://cdc.gov/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "government",
        "domain": "census.gov",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 1162,
        "url": "https://census.gov/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:census-gov",
      "name": "census.gov",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "census.gov"
        ],
        "source_urls": [
          "https://census.gov/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://census.gov/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "Allen Institute for AI",
          "Amazon",
          "Anthropic",
          "Apple",
          "ByteDance",
          "Common Crawl",
          "Diffbot",
          "Google",
          "Meta",
          "OpenAI",
          "Perplexity AI",
          "Timpi",
          "Webz.io"
        ],
        "blocks": [
          "GPTBot",
          "ChatGPT-User",
          "OAI-SearchBot",
          "ClaudeBot",
          "Claude-User",
          "anthropic-ai",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "Amazonbot",
          "Applebot-Extended",
          "PerplexityBot",
          "Perplexity-User",
          "meta-externalagent",
          "FacebookBot",
          "Diffbot",
          "omgili",
          "Timpibot",
          "AI2Bot",
          "Webzio-Extended"
        ],
        "bucket": "news_publisher",
        "domain": "chicagotribune.com",
        "effective_blocked_operators": [
          "Allen Institute for AI",
          "Amazon",
          "Anthropic",
          "Apple",
          "ByteDance",
          "Common Crawl",
          "Diffbot",
          "Google",
          "Meta",
          "OpenAI",
          "Perplexity AI",
          "Timpi",
          "Webz.io"
        ],
        "effective_blocks": [
          "GPTBot",
          "ChatGPT-User",
          "OAI-SearchBot",
          "ClaudeBot",
          "Claude-User",
          "anthropic-ai",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "Amazonbot",
          "Applebot-Extended",
          "PerplexityBot",
          "Perplexity-User",
          "meta-externalagent",
          "FacebookBot",
          "Diffbot",
          "omgili",
          "Timpibot",
          "AI2Bot",
          "Webzio-Extended"
        ],
        "has_tdm_reservation": true,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow. [{\"location\": \"/\", \"tdm-reservation\": 1}]",
        "readable": true,
        "robots_bytes": 4434,
        "url": "https://chicagotribune.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:chicagotribune-com",
      "name": "chicagotribune.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "chicagotribune.com"
        ],
        "source_urls": [
          "https://chicagotribune.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://chicagotribune.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "ai_company",
        "domain": "clearview.ai",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow. tdmrep.json HTTP 400",
        "readable": true,
        "registry_roles": [
          "defendant"
        ],
        "robots_bytes": 490,
        "url": "https://clearview.ai/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:clearview-ai",
      "name": "clearview.ai",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "clearview.ai"
        ],
        "source_urls": [
          "https://clearview.ai/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://clearview.ai/robots.txt"
    },
    {
      "fields": {
        "allows_all": false,
        "blocked_operators": [
          "Allen Institute for AI",
          "Amazon",
          "Anthropic",
          "Apple",
          "ByteDance",
          "Cohere",
          "Common Crawl",
          "Diffbot",
          "Google",
          "ImageSift",
          "Meta",
          "OpenAI",
          "Perplexity AI",
          "Timpi",
          "Webz.io"
        ],
        "blocks": [
          "GPTBot",
          "ChatGPT-User",
          "OAI-SearchBot",
          "ClaudeBot",
          "Claude-User",
          "anthropic-ai",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "Amazonbot",
          "Applebot-Extended",
          "PerplexityBot",
          "Perplexity-User",
          "FacebookBot",
          "cohere-ai",
          "Diffbot",
          "ImagesiftBot",
          "omgili",
          "Timpibot",
          "AI2Bot",
          "Webzio-Extended"
        ],
        "bucket": "news_publisher",
        "domain": "cnn.com",
        "effective_blocked_operators": [
          "Allen Institute for AI",
          "Amazon",
          "Anthropic",
          "Apple",
          "ByteDance",
          "Cohere",
          "Common Crawl",
          "Diffbot",
          "Google",
          "ImageSift",
          "Meta",
          "OpenAI",
          "Perplexity AI",
          "Timpi",
          "Webz.io"
        ],
        "effective_blocks": [
          "GPTBot",
          "ChatGPT-User",
          "OAI-SearchBot",
          "ClaudeBot",
          "Claude-User",
          "anthropic-ai",
          "Google-Extended",
          "CCBot",
          "Bytespider",
          "Amazonbot",
          "Applebot-Extended",
          "PerplexityBot",
          "Perplexity-User",
          "FacebookBot",
          "cohere-ai",
          "Diffbot",
          "ImagesiftBot",
          "omgili",
          "Timpibot",
          "AI2Bot",
          "Webzio-Extended"
        ],
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "robots_bytes": 3456,
        "url": "https://cnn.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:cnn-com",
      "name": "cnn.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "cnn.com"
        ],
        "source_urls": [
          "https://cnn.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://cnn.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "ai_company",
        "domain": "cohere.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "defendant"
        ],
        "robots_bytes": 328,
        "url": "https://cohere.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:cohere-com",
      "name": "cohere.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "cohere.com"
        ],
        "source_urls": [
          "https://cohere.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://cohere.com/robots.txt"
    },
    {
      "fields": {
        "allows_all": true,
        "bucket": "music",
        "domain": "concord.com",
        "has_tdm_reservation": false,
        "http_status": 200,
        "measured_at": "2026-09-17",
        "note": "Measured from this hostname's own robots.txt on 2026-09-17. Blocked here means a Disallow: / for that crawler's user-agent; effective adds crawlers covered by a wildcard Disallow.",
        "readable": true,
        "registry_roles": [
          "plaintiff"
        ],
        "robots_bytes": 130,
        "sued": [
          "Anthropic",
          "Benjamin Mann",
          "Dario Amodei"
        ],
        "url": "https://concord.com/robots.txt",
        "wildcard_disallow_root": false
      },
      "id": "crawl_block:concord-com",
      "name": "concord.com",
      "notes": {
        "_measured_at_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "concord.com"
        ],
        "source_urls": [
          "https://concord.com/robots.txt"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 45
      },
      "type": "crawl_block",
      "url": "https://concord.com/robots.txt"
    }
  ],
  "_meta": {
    "source": "Blomega Data Refinery",
    "url": "https://data.blomega.com",
    "publisher": "Blomega",
    "publisher_url": "https://blomegalab.com",
    "wikidata": "Q141048865",
    "license": "CC BY 4.0",
    "license_url": "https://creativecommons.org/licenses/by/4.0/",
    "cite_as": "Blomega Data Refinery (https://data.blomega.com), CC BY 4.0. Cite the registry and the record id.",
    "attribution_required": true
  }
}
