{
  "total": 41,
  "limit": 50,
  "offset": 0,
  "records": [
    {
      "fields": {
        "controls": "Adobe commits not to train Firefly on user content; trains on licensed (Adobe Stock) and public domain content",
        "default_state": null,
        "does_not_cover": "Adobe Stock contributors' submissions are used for training (compensated); this is a policy commitment, not a user toggle; content analysis toggle not verified (helpx page returned 403)",
        "honoured_by": [
          "Adobe (self-declared, Terms of Use 2.2F and 4.3C2)"
        ],
        "jurisdiction": "global",
        "legal_force": "contractual",
        "mechanism": "platform-toggle",
        "name": "Adobe no training on customer content (Firefly)",
        "note": "Not an opt-out mechanism strictly: a no-training-by-default commitment.",
        "operator": "Adobe",
        "orgs": [
          "Adobe"
        ],
        "url": "https://www.adobe.com/ai/overview/firefly/gen-ai-approach.html"
      },
      "id": "opt_out:adobe-no-training-on-customer-content-firefly",
      "name": "Adobe no training on customer content (Firefly)",
      "notes": {
        "_org_roles": {
          "Adobe": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "adobe.com"
        ],
        "source_urls": [
          "https://www.adobe.com/ai/overview/firefly/gen-ai-approach.html"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://www.adobe.com/ai/overview/firefly/gen-ai-approach.html"
    },
    {
      "fields": {
        "controls": "crawl for Amazon products, may be used to train Amazon AI models",
        "default_state": "opt-out",
        "does_not_cover": "Amzn-User user-initiated fetches may not follow all robots.txt rules; data already collected",
        "honoured_by": [
          "Amazon (self-declared)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "Amazonbot robots.txt token",
        "note": "Amzn-SearchBot and Amzn-User state they do not crawl for generative AI training.",
        "operator": "Amazon",
        "orgs": [
          "Amazon"
        ],
        "url": "https://developer.amazon.com/amazonbot"
      },
      "id": "opt_out:amazonbot-robots-txt-token",
      "name": "Amazonbot robots.txt token",
      "notes": {
        "_org_roles": {
          "Amazon": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "developer.amazon.com"
        ],
        "source_urls": [
          "https://developer.amazon.com/amazonbot"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://developer.amazon.com/amazonbot"
    },
    {
      "fields": {
        "controls": "use of Applebot-crawled content to train Apple foundation models (Apple Intelligence)",
        "default_state": "opt-out",
        "does_not_cover": "Does not stop Applebot crawling or Spotlight/Siri/Safari search inclusion; Apple does not address removal of previously crawled data",
        "honoured_by": [
          "Apple (self-declared)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "Applebot-Extended robots.txt token",
        "note": "Applebot-Extended does not crawl; it is a usage-control token only.",
        "operator": "Apple",
        "orgs": [
          "Apple"
        ],
        "url": "https://support.apple.com/en-us/119829"
      },
      "id": "opt_out:applebot-extended-robots-txt-token",
      "name": "Applebot-Extended robots.txt token",
      "notes": {
        "_org_roles": {
          "Apple": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "support.apple.com"
        ],
        "source_urls": [
          "https://support.apple.com/en-us/119829"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://support.apple.com/en-us/119829"
    },
    {
      "fields": {
        "controls": "training crawl (reported use for LLMs including Doubao)",
        "default_state": "opt-out",
        "does_not_cover": "No operator documentation read that commits to honouring robots.txt; data already collected",
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "Bytespider robots.txt token",
        "note": "Known Agents (formerly Dark Visitors) lists it as expected to follow robots.txt, citing a ByteDance site that was not read here. Unverified: treat as not documented.",
        "operator": "ByteDance",
        "orgs": [
          "ByteDance"
        ],
        "url": "https://knownagents.com/agents/bytespider"
      },
      "id": "opt_out:bytespider-robots-txt-token",
      "name": "Bytespider robots.txt token",
      "notes": {
        "_org_roles": {
          "ByteDance": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "knownagents.com"
        ],
        "source_urls": [
          "https://knownagents.com/agents/bytespider"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://knownagents.com/agents/bytespider"
    },
    {
      "fields": {
        "controls": "single request to delete and stop sale of personal information across 600+ registered data brokers, reprocessed every 45 days",
        "default_state": "opt-out",
        "does_not_cover": "First-party data held by companies that are not data brokers (including most AI labs and platforms); not a copyright or training opt-out; page does not mention AI",
        "effective": "2026-08-01",
        "honoured_by": [
          "Registered data brokers (legal obligation)"
        ],
        "jurisdiction": "US-CA",
        "legal_force": "enforceable-law",
        "mechanism": "regulatory-request",
        "name": "California DROP (Delete Act, SB 362)",
        "note": "Residents could submit from 2026-01-01; brokers must begin processing 2026-08-01. Relevant only where AI training data is bought from brokers.",
        "operator": "California Privacy Protection Agency",
        "orgs": [
          "California Privacy Protection Agency"
        ],
        "url": "https://privacy.ca.gov/drop"
      },
      "id": "opt_out:california-drop-delete-act-sb-362",
      "name": "California DROP (Delete Act, SB 362)",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "California Privacy Protection Agency": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "privacy.ca.gov"
        ],
        "source_urls": [
          "https://privacy.ca.gov/drop"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://privacy.ca.gov/drop"
    },
    {
      "fields": {
        "controls": "per-asset permissions for data mining, AI inference, AI training, generative AI training (allowed, notAllowed, constrained)",
        "default_state": null,
        "does_not_cover": "C2PA removed its own training-mining assertion in spec v2.0, so older c2pa.* labels are obsolete; metadata is easily stripped; no model developer documentation read commits to honouring it",
        "effective": "2025-05-16",
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "standard",
        "name": "CAWG training and data mining assertion (successor to C2PA do-not-train)",
        "note": "Removal confirmed at https://spec.c2pa.org/specifications/specifications/2.1/specs/C2PA_Specification.html. Constrained is treated as notAllowed absent further terms, which makes it a licensing hook.",
        "operator": [
          "Creator Assertions Working Group",
          "used inside C2PA manifests"
        ],
        "orgs": [
          "Creator Assertions Working Group",
          "used inside C2PA manifests"
        ],
        "url": "https://cawg.io/training-and-data-mining/1.1"
      },
      "id": "opt_out:cawg-training-and-data-mining-assertion-successor-to-c2pa-do-not-train",
      "name": "CAWG training and data mining assertion (successor to C2PA do-not-train)",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "Creator Assertions Working Group": [
            "operator"
          ],
          "used inside C2PA manifests": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "cawg.io"
        ],
        "source_urls": [
          "https://cawg.io/training-and-data-mining/1.1"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://cawg.io/training-and-data-mining/1.1"
    },
    {
      "fields": {
        "controls": "open web archive crawl, redistributed publicly and widely reused as AI training data",
        "default_state": "opt-out",
        "does_not_cover": "Past crawl snapshots already published and copied by downstream users; page does not address retroactive removal; downstream AI developers who already downloaded archives",
        "honoured_by": [
          "Common Crawl (self-declared)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "CCBot robots.txt token",
        "note": "Critical upstream token: blocking CCBot is the only way to exit future Common Crawl releases, which feed many model datasets. Spoofed CCBot agents exist; verify by reverse DNS.",
        "operator": "Common Crawl Foundation",
        "orgs": [
          "Common Crawl Foundation"
        ],
        "url": "https://commoncrawl.org/ccbot"
      },
      "id": "opt_out:ccbot-robots-txt-token",
      "name": "CCBot robots.txt token",
      "notes": {
        "_org_roles": {
          "Common Crawl Foundation": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "commoncrawl.org"
        ],
        "source_urls": [
          "https://commoncrawl.org/ccbot"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://commoncrawl.org/ccbot"
    },
    {
      "fields": {
        "controls": "user-initiated fetch (actions in ChatGPT, custom GPTs)",
        "default_state": "opt-out",
        "does_not_cover": "OpenAI itself says robots.txt may not apply because fetches are user-initiated; not a training control",
        "honoured_by": [
          "OpenAI (partial: operator says robots.txt rules may not apply; Cloudflare test 2025-08-04 observed it fetch robots.txt and stop when disallowed)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "ChatGPT-User robots.txt token",
        "note": "Operator documentation and third-party test differ in strength of claim. Test source: https://blog.cloudflare.com/perplexity-is-using-stealth-undeclared-crawlers-to-evade-website-no-crawl-directives/",
        "operator": "OpenAI",
        "orgs": [
          "OpenAI"
        ],
        "url": "https://developers.openai.com/api/docs/bots"
      },
      "id": "opt_out:chatgpt-user-robots-txt-token",
      "name": "ChatGPT-User robots.txt token",
      "notes": {
        "_org_roles": {
          "OpenAI": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.openai.com"
        ],
        "source_urls": [
          "https://developers.openai.com/api/docs/bots"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://developers.openai.com/api/docs/bots"
    },
    {
      "fields": {
        "controls": "search index (Claude search result quality)",
        "default_state": "opt-out",
        "does_not_cover": "Training (ClaudeBot); already-indexed content",
        "honoured_by": [
          "Anthropic (self-declared)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "Claude-SearchBot robots.txt token",
        "note": "Blocking may reduce visibility in Claude search answers.",
        "operator": "Anthropic",
        "orgs": [
          "Anthropic"
        ],
        "url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
      },
      "id": "opt_out:claude-searchbot-robots-txt-token",
      "name": "Claude-SearchBot robots.txt token",
      "notes": {
        "_org_roles": {
          "Anthropic": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "support.claude.com"
        ],
        "source_urls": [
          "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
    },
    {
      "fields": {
        "controls": "user-initiated fetch (pages retrieved when a user directs Claude)",
        "default_state": "opt-out",
        "does_not_cover": "Training (ClaudeBot) and search indexing (Claude-SearchBot) are separate tokens",
        "honoured_by": [
          "Anthropic (self-declared: its bots honour robots.txt)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "Claude-User robots.txt token",
        "note": "Unlike OpenAI, Perplexity and Meta, Anthropic's page does not carve user-initiated fetches out of robots.txt compliance.",
        "operator": "Anthropic",
        "orgs": [
          "Anthropic"
        ],
        "url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
      },
      "id": "opt_out:claude-user-robots-txt-token",
      "name": "Claude-User robots.txt token",
      "notes": {
        "_org_roles": {
          "Anthropic": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "support.claude.com"
        ],
        "source_urls": [
          "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
    },
    {
      "fields": {
        "controls": "training crawl",
        "default_state": "opt-out",
        "does_not_cover": "Anthropic frames the signal as excluding the site's future materials from training; no statement on removing already-collected data; does not govern Claude-User or Claude-SearchBot",
        "honoured_by": [
          "Anthropic (self-declared)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "ClaudeBot robots.txt token",
        "note": "Anthropic also supports the non-standard Crawl-delay extension and publishes crawler IPs for verification.",
        "operator": "Anthropic",
        "orgs": [
          "Anthropic"
        ],
        "url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
      },
      "id": "opt_out:claudebot-robots-txt-token",
      "name": "ClaudeBot robots.txt token",
      "notes": {
        "_org_roles": {
          "Anthropic": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "support.claude.com"
        ],
        "source_urls": [
          "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
    },
    {
      "fields": {
        "controls": "blocks known and fingerprinted AI crawlers at the network edge, including ones that ignore robots.txt",
        "default_state": "off",
        "does_not_cover": "Only sites behind Cloudflare; ML detection is imperfect; user-initiated agents and headless browsers can blur categories; does nothing for data already collected",
        "effective": "2024-07-03",
        "honoured_by": [
          "Enforced technically by Cloudflare on its customers' zones"
        ],
        "jurisdiction": "global",
        "legal_force": "contractual",
        "mechanism": "network-block",
        "name": "Cloudflare AI Scrapers and Crawlers block",
        "note": "Available on all plans including free. On 2025-07-01 Cloudflare announced default blocking of AI crawlers for new domains (https://blog.cloudflare.com/content-independence-day-no-ai-crawl-without-compensation/), so default_state depends on domain age.",
        "operator": "Cloudflare",
        "orgs": [
          "Cloudflare"
        ],
        "url": "https://blog.cloudflare.com/declaring-your-aindependence-block-ai-bots-scrapers-and-crawlers-with-a-single-click"
      },
      "id": "opt_out:cloudflare-ai-scrapers-and-crawlers-block",
      "name": "Cloudflare AI Scrapers and Crawlers block",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "Cloudflare": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "blog.cloudflare.com"
        ],
        "source_urls": [
          "https://blog.cloudflare.com/declaring-your-aindependence-block-ai-bots-scrapers-and-crawlers-with-a-single-click"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://blog.cloudflare.com/declaring-your-aindependence-block-ai-bots-scrapers-and-crawlers-with-a-single-click"
    },
    {
      "fields": {
        "controls": "search, ai-input (real-time generative answers), ai-train preferences",
        "default_state": "on",
        "does_not_cover": "Cloudflare says signals are preferences, not countermeasures, and some companies may ignore them; ai-input left unset by default; content already collected",
        "effective": "2025-09-24",
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "standard",
        "name": "Cloudflare Content Signals Policy (Content-Signal in robots.txt)",
        "note": "Default search=yes, ai-train=no applied to managed robots.txt on about 3.8M domains. Policy text asserts the signals are express Article 4 reservations, which would give them legal weight in the EU: that is Cloudflare's claim, untested in court.",
        "operator": "Cloudflare",
        "orgs": [
          "Cloudflare"
        ],
        "url": "https://blog.cloudflare.com/content-signals-policy"
      },
      "id": "opt_out:cloudflare-content-signals-policy-content-signal-in-robots-txt",
      "name": "Cloudflare Content Signals Policy (Content-Signal in robots.txt)",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "Cloudflare": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "blog.cloudflare.com"
        ],
        "source_urls": [
          "https://blog.cloudflare.com/content-signals-policy"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://blog.cloudflare.com/content-signals-policy"
    },
    {
      "fields": {
        "controls": "per-crawler allow, charge (HTTP 402 with price headers) or block",
        "default_state": "off",
        "does_not_cover": "Private beta; only sites on Cloudflare; flat domain-wide price, no training vs inference distinction; unregistered crawlers are simply blocked, not paid",
        "effective": "2025-07-01",
        "honoured_by": [
          "Crawlers registered with Cloudflare using Web Bot Auth signatures"
        ],
        "jurisdiction": "global",
        "legal_force": "contractual",
        "mechanism": "network-block",
        "name": "Cloudflare Pay per crawl",
        "note": "The only mechanism in this set that turns an opt-out into a payment route at crawl time.",
        "operator": "Cloudflare",
        "orgs": [
          "Cloudflare"
        ],
        "url": "https://blog.cloudflare.com/introducing-pay-per-crawl"
      },
      "id": "opt_out:cloudflare-pay-per-crawl",
      "name": "Cloudflare Pay per crawl",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "Cloudflare": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "blog.cloudflare.com"
        ],
        "source_urls": [
          "https://blog.cloudflare.com/introducing-pay-per-crawl"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://blog.cloudflare.com/introducing-pay-per-crawl"
    },
    {
      "fields": {
        "controls": "obliges GPAI providers to identify and respect Article 4(3) reservations with state-of-the-art technology and publish a training content summary",
        "default_state": null,
        "does_not_cover": "Does not itself create a new opt-out; transitional rules for models placed before 2025-08-02 not verified; does not require removal from already-trained models",
        "effective": "2025-08-02",
        "honoured_by": [
          "Binding on GPAI model providers placing models on the EU market"
        ],
        "jurisdiction": "EU",
        "legal_force": "enforceable-law",
        "mechanism": "legal-reservation",
        "name": "EU AI Act Article 53(1)(c)-(d) GPAI copyright policy and training summary",
        "note": "This is what converts crawler tokens into a legally relevant signal in the EU.",
        "operator": "European Union",
        "orgs": [
          "European Union"
        ],
        "url": "https://artificialintelligenceact.eu/article/53"
      },
      "id": "opt_out:eu-ai-act-article-53-1-c-d-gpai-copyright-policy-and-training-summary",
      "name": "EU AI Act Article 53(1)(c)-(d) GPAI copyright policy and training summary",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "European Union": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "artificialintelligenceact.eu"
        ],
        "source_urls": [
          "https://artificialintelligenceact.eu/article/53"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://artificialintelligenceact.eu/article/53"
    },
    {
      "fields": {
        "controls": "signatories commit to crawlers that follow robots.txt and other standardised reservation protocols, not circumvent paywalls, exclude piracy domains, provide a rightsholder complaint point",
        "default_state": null,
        "does_not_cover": "Non-signatories; voluntary instrument (a compliance route for Article 53, not the law itself); data already collected",
        "effective": "2025-07-10",
        "honoured_by": [
          "Signatories (list not verified in this pass)"
        ],
        "jurisdiction": "EU",
        "legal_force": "voluntary",
        "mechanism": "regulatory-request",
        "name": "EU GPAI Code of Practice, Copyright chapter",
        "note": "Complaint mechanism is the closest thing to an enforcement channel for rights holders against signatories.",
        "operator": "European Commission AI Office",
        "orgs": [
          "European Commission AI Office"
        ],
        "url": "https://code-of-practice.ai/?section=copyright"
      },
      "id": "opt_out:eu-gpai-code-of-practice-copyright-chapter",
      "name": "EU GPAI Code of Practice, Copyright chapter",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "European Commission AI Office": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "code-of-practice.ai"
        ],
        "source_urls": [
          "https://code-of-practice.ai/?section=copyright"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://code-of-practice.ai/?section=copyright"
    },
    {
      "fields": {
        "controls": "commercial text and data mining of lawfully accessible works, including AI training, where rightholder has expressly reserved rights",
        "default_state": "opt-out",
        "does_not_cover": "Research organisations under Article 3 (no opt-out); works mined before the reservation was made; training performed outside the EU; unclear which machine-readable formats qualify (robots.txt vs TDMRep vs natural-language terms)",
        "effective": "2021-06-07",
        "honoured_by": [
          "Binding on anyone relying on the Article 4 exception in the EU; GPAI providers must identify and respect it under AI Act Article 53(1)(c)"
        ],
        "jurisdiction": "EU",
        "legal_force": "enforceable-law",
        "mechanism": "legal-reservation",
        "name": "EU TDM rights reservation (DSM Directive Article 4(3))",
        "note": "For online content the reservation must be machine-readable (metadata, website terms). Transposition deadline 6 June 2021. Effective date is the day after the deadline; national dates vary. Kneschke v LAION (Hamburg, Sept 2024) is the first major test case; details not verified here.",
        "operator": [
          "European Union (Directive (EU) 2019",
          "790)"
        ],
        "orgs": [
          "European Union (Directive (EU) 2019",
          "790)"
        ],
        "url": "https://eur-lex.europa.eu/eli/dir/2019/790/oj/eng"
      },
      "id": "opt_out:eu-tdm-rights-reservation-dsm-directive-article-4-3",
      "name": "EU TDM rights reservation (DSM Directive Article 4(3))",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "790)": [
            "operator"
          ],
          "European Union (Directive (EU) 2019": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "eur-lex.europa.eu"
        ],
        "source_urls": [
          "https://eur-lex.europa.eu/eli/dir/2019/790/oj/eng"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://eur-lex.europa.eu/eli/dir/2019/790/oj/eng"
    },
    {
      "fields": {
        "controls": "processing of personal data under legitimate interests, including AI training by any controller relying on that basis",
        "default_state": "opt-out",
        "does_not_cover": "Only personal data, not copyright in works; controller may continue if it shows compelling legitimate grounds; unlearning from trained models is technically unresolved",
        "effective": "2018-05-25",
        "honoured_by": [
          "Binding on controllers subject to GDPR"
        ],
        "jurisdiction": "EU",
        "legal_force": "enforceable-law",
        "mechanism": "regulatory-request",
        "name": "GDPR Article 21 right to object (general)",
        "note": "The universal regulatory lever for individuals; platform forms (Meta, LinkedIn) are implementations of it.",
        "operator": [
          "European Union",
          "mirrored in UK GDPR"
        ],
        "orgs": [
          "European Union",
          "mirrored in UK GDPR"
        ],
        "url": "https://gdpr-info.eu/art-21-gdpr"
      },
      "id": "opt_out:gdpr-article-21-right-to-object-general",
      "name": "GDPR Article 21 right to object (general)",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "European Union": [
            "operator"
          ],
          "mirrored in UK GDPR": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "gdpr-info.eu"
        ],
        "source_urls": [
          "https://gdpr-info.eu/art-21-gdpr"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://gdpr-info.eu/art-21-gdpr"
    },
    {
      "fields": {
        "controls": "crawls requested by site owners for building Vertex AI Agents",
        "default_state": "opt-out",
        "does_not_cover": "Not a general training opt-out; no effect on Search or other products",
        "honoured_by": [
          "Google (self-declared)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "Google-CloudVertexBot robots.txt token",
        "note": "Narrow scope.",
        "operator": "Google",
        "orgs": [
          "Google"
        ],
        "url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers"
      },
      "id": "opt_out:google-cloudvertexbot-robots-txt-token",
      "name": "Google-CloudVertexBot robots.txt token",
      "notes": {
        "_org_roles": {
          "Google": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.google.com"
        ],
        "source_urls": [
          "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers"
    },
    {
      "fields": {
        "controls": "use of crawled content for training future Gemini models and for grounding in Gemini Apps and Vertex AI Grounding with Google Search",
        "default_state": "opt-out",
        "does_not_cover": "Does not crawl separately (reuses Googlebot); does NOT affect Google Search, AI Overviews or AI Mode, which are controlled only via Googlebot, nosnippet, max-snippet or noindex; models already trained",
        "honoured_by": [
          "Google (self-declared)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "Google-Extended robots.txt token",
        "note": "Biggest known gap in the ecosystem: opting out of AI Overviews requires limiting Search snippets or indexing. See https://developers.google.com/search/docs/appearance/ai-features.",
        "operator": "Google",
        "orgs": [
          "Google"
        ],
        "url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers"
      },
      "id": "opt_out:google-extended-robots-txt-token",
      "name": "Google-Extended robots.txt token",
      "notes": {
        "_org_roles": {
          "Google": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.google.com"
        ],
        "source_urls": [
          "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers"
    },
    {
      "fields": {
        "controls": "display of content in AI Overviews and AI Mode (same controls as Search snippets)",
        "default_state": "opt-out",
        "does_not_cover": "Cannot opt out of AI features while keeping full Search snippets; recrawl can take days to months; does not govern Gemini training (Google-Extended)",
        "honoured_by": [
          "Google (self-declared)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "Google search snippet controls (nosnippet, data-nosnippet, max-snippet, noindex) for AI Overviews and AI Mode",
        "note": "Google states AI is integral to Search, so Googlebot directives are the control.",
        "operator": "Google",
        "orgs": [
          "Google"
        ],
        "url": "https://developers.google.com/search/docs/appearance/ai-features"
      },
      "id": "opt_out:google-search-snippet-controls-nosnippet-data-nosnippet-max-snippet-noindex-for-ai-overviews-and-ai-mode",
      "name": "Google search snippet controls (nosnippet, data-nosnippet, max-snippet, noindex) for AI Overviews and AI Mode",
      "notes": {
        "_org_roles": {
          "Google": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.google.com"
        ],
        "source_urls": [
          "https://developers.google.com/search/docs/appearance/ai-features"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://developers.google.com/search/docs/appearance/ai-features"
    },
    {
      "fields": {
        "controls": "training crawl (generative AI foundation models)",
        "default_state": "opt-out",
        "does_not_cover": "Data already collected before the disallow; content reaching OpenAI via third-party datasets or licensed sources; does not affect ChatGPT search (separate OAI-SearchBot token) or user-initiated fetches (ChatGPT-User)",
        "honoured_by": [
          "OpenAI (self-declared)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "GPTBot robots.txt token",
        "note": "OpenAI states disallowing GPTBot signals content should not be used in training. No statement on retroactive removal.",
        "operator": "OpenAI",
        "orgs": [
          "OpenAI"
        ],
        "url": "https://developers.openai.com/api/docs/bots"
      },
      "id": "opt_out:gptbot-robots-txt-token",
      "name": "GPTBot robots.txt token",
      "notes": {
        "_org_roles": {
          "OpenAI": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.openai.com"
        ],
        "source_urls": [
          "https://developers.openai.com/api/docs/bots"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://developers.openai.com/api/docs/bots"
    },
    {
      "fields": {
        "controls": "standard vocabulary for AI use preferences attached via robots.txt and HTTP headers, with reconciliation rules",
        "default_state": null,
        "does_not_cover": "Charter explicitly excludes technical enforcement; not yet an RFC; no retroactive effect",
        "effective": "2026-08",
        "jurisdiction": "global",
        "legal_force": "proposed",
        "mechanism": "standard",
        "name": "IETF AI Preferences (aipref) vocabulary and attachment",
        "note": "Milestone: submission to IESG targeted August 2026; status after that not verified.",
        "operator": "IETF aipref working group",
        "orgs": [
          "IETF aipref working group"
        ],
        "url": "https://datatracker.ietf.org/wg/aipref/about"
      },
      "id": "opt_out:ietf-ai-preferences-aipref-vocabulary-and-attachment",
      "name": "IETF AI Preferences (aipref) vocabulary and attachment",
      "notes": {
        "_effective_precision": "month",
        "_org_roles": {
          "IETF aipref working group": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "datatracker.ietf.org"
        ],
        "source_urls": [
          "https://datatracker.ietf.org/wg/aipref/about"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://datatracker.ietf.org/wg/aipref/about"
    },
    {
      "fields": {
        "controls": "directory of AI agents, crawlers and scrapers with auto-generated robots.txt, agent analytics and identification API",
        "default_state": null,
        "does_not_cover": "Covers crawler identity and robots.txt only; no legal force, platform toggles, regulatory routes, or data-already-collected gap",
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "network-block",
        "name": "Known Agents (formerly Dark Visitors) robots.txt agent directory",
        "note": "Included as the reference tracker, not as an opt-out. darkvisitors.com/agents now 301-redirects to knownagents.com/agents.",
        "operator": "Known Agents",
        "orgs": [
          "Known Agents"
        ],
        "url": "https://knownagents.com/agents"
      },
      "id": "opt_out:known-agents-formerly-dark-visitors-robots-txt-agent-directory",
      "name": "Known Agents (formerly Dark Visitors) robots.txt agent directory",
      "notes": {
        "_org_roles": {
          "Known Agents": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "knownagents.com"
        ],
        "source_urls": [
          "https://knownagents.com/agents"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.727,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://knownagents.com/agents"
    },
    {
      "fields": {
        "controls": "use of member data by LinkedIn and affiliates (including Microsoft) to train content-generating AI models",
        "default_state": null,
        "does_not_cover": "Does not affect training that already took place; feedback data and non-generative models need the separate Data Processing Objection form",
        "effective": "2025-11-03",
        "honoured_by": [
          "LinkedIn (self-declared)"
        ],
        "jurisdiction": "global",
        "legal_force": "contractual",
        "mechanism": "platform-toggle",
        "name": "LinkedIn Data for Generative AI Improvement setting",
        "note": "Default not stated on page read. EU/EEA/Switzerland and UK members have extra protections; effective date is the European regional privacy notice update.",
        "operator": "LinkedIn",
        "orgs": [
          "LinkedIn"
        ],
        "url": "https://www.linkedin.com/help/linkedin/answer/a5538339"
      },
      "id": "opt_out:linkedin-data-for-generative-ai-improvement-setting",
      "name": "LinkedIn Data for Generative AI Improvement setting",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "LinkedIn": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "linkedin.com"
        ],
        "source_urls": [
          "https://www.linkedin.com/help/linkedin/answer/a5538339"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://www.linkedin.com/help/linkedin/answer/a5538339"
    },
    {
      "fields": {
        "controls": "use of an EU/UK adult's public Facebook/Instagram posts, comments and Meta AI interactions for training Meta AI models",
        "default_state": "opt-out",
        "does_not_cover": "Private messages and under-18 data already excluded; announcement does not say objection removes data already used; content about you posted by others needs the separate third-party data form; non-EU/UK users have no equivalent right",
        "effective": "2025-04-14",
        "honoured_by": [
          "Meta (states it honours all objections received)"
        ],
        "jurisdiction": "EU",
        "legal_force": "enforceable-law",
        "mechanism": "regulatory-request",
        "name": "Meta AI training objection (GDPR Article 21 route)",
        "note": "Legal force comes from GDPR Article 21(1) (processing based on legitimate interests must stop unless compelling grounds): https://gdpr-info.eu/art-21-gdpr/. The form itself requires login and was not readable. UK coverage asserted in general knowledge, not verified.",
        "operator": "Meta Platforms",
        "orgs": [
          "Meta Platforms"
        ],
        "url": "https://about.fb.com/news/2025/04/making-ai-work-harder-for-europeans"
      },
      "id": "opt_out:meta-ai-training-objection-gdpr-article-21-route",
      "name": "Meta AI training objection (GDPR Article 21 route)",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "Meta Platforms": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "about.fb.com"
        ],
        "source_urls": [
          "https://about.fb.com/news/2025/04/making-ai-work-harder-for-europeans"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://about.fb.com/news/2025/04/making-ai-work-harder-for-europeans"
    },
    {
      "fields": {
        "controls": "training crawl (foundation AI models) and product indexing",
        "default_state": "opt-out",
        "does_not_cover": "Meta's page does not explicitly state robots.txt compliance for this agent; Meta-ExternalFetcher (user-initiated) may bypass robots.txt; data already collected; content posted on Meta's own platforms (governed by privacy objection routes)",
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "Meta-ExternalAgent robots.txt token",
        "note": "honoured_by left empty because the page read does not explicitly commit. Meta advises allowing up to 24 hours for robots.txt changes.",
        "operator": "Meta",
        "orgs": [
          "Meta"
        ],
        "url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers"
      },
      "id": "opt_out:meta-externalagent-robots-txt-token",
      "name": "Meta-ExternalAgent robots.txt token",
      "notes": {
        "_org_roles": {
          "Meta": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.facebook.com"
        ],
        "source_urls": [
          "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers"
    },
    {
      "fields": {
        "controls": "user-initiated fetch and agentic AI navigation",
        "default_state": null,
        "does_not_cover": "Meta states it may bypass robots.txt",
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "Meta-ExternalFetcher token",
        "note": "Identification only.",
        "operator": "Meta",
        "orgs": [
          "Meta"
        ],
        "url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers"
      },
      "id": "opt_out:meta-externalfetcher-token",
      "name": "Meta-ExternalFetcher token",
      "notes": {
        "_org_roles": {
          "Meta": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.facebook.com"
        ],
        "source_urls": [
          "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.727,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers"
    },
    {
      "fields": {
        "controls": "signal that content is not authorised for AI datasets",
        "default_state": "opt-out",
        "does_not_cover": "No crawler operator documentation read commits to honouring it; DeviantArt acknowledges it cannot technically prevent scraping; content already in datasets",
        "effective": "2022-11-11",
        "honoured_by": [
          "DeviantArt (sets it by default on its own pages)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "standard",
        "name": "noai / noimageai directives (meta robots and X-Robots-Tag)",
        "note": "DeviantArt states third parties must exclude tagged content, a platform policy claim not law.",
        "operator": "DeviantArt",
        "orgs": [
          "DeviantArt"
        ],
        "url": "https://www.deviantart.com/team/journal/UPDATE-All-Deviations-Are-Opted-Out-of-AI-Datasets-934500371"
      },
      "id": "opt_out:noai-noimageai-directives-meta-robots-and-x-robots-tag",
      "name": "noai / noimageai directives (meta robots and X-Robots-Tag)",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "DeviantArt": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "deviantart.com"
        ],
        "source_urls": [
          "https://www.deviantart.com/team/journal/UPDATE-All-Deviations-Are-Opted-Out-of-AI-Datasets-934500371"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://www.deviantart.com/team/journal/UPDATE-All-Deviations-Are-Opted-Out-of-AI-Datasets-934500371"
    },
    {
      "fields": {
        "controls": "use of page for model training while allowing crawl",
        "default_state": "opt-out",
        "does_not_cover": "Only Amazon agents documented; page-level only",
        "honoured_by": [
          "Amazon (self-declared for Amazonbot, Amzn-SearchBot, Amzn-User)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "noarchive meta tag as AI-training signal (Amazon)",
        "note": "Reuses a legacy search directive with a new AI meaning.",
        "operator": "Amazon",
        "orgs": [
          "Amazon"
        ],
        "url": "https://developer.amazon.com/amazonbot"
      },
      "id": "opt_out:noarchive-meta-tag-as-ai-training-signal-amazon",
      "name": "noarchive meta tag as AI-training signal (Amazon)",
      "notes": {
        "_org_roles": {
          "Amazon": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "developer.amazon.com"
        ],
        "source_urls": [
          "https://developer.amazon.com/amazonbot"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://developer.amazon.com/amazonbot"
    },
    {
      "fields": {
        "controls": "NOARCHIVE: exclusion from Bing Chat answers and from training Microsoft generative AI foundation models; NOCACHE: only URL, title, snippet used",
        "default_state": "opt-out",
        "does_not_cover": "Default (no tag) permits training use; announcement dates from 2023 and product names have since changed (Copilot); models already trained",
        "effective": "2023-09-22",
        "honoured_by": [
          "Microsoft Bing (self-declared)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "NOCACHE and NOARCHIVE meta tags for Bing Chat / Microsoft generative AI",
        "note": "Check current Copilot documentation before relying on 2023 semantics.",
        "operator": "Microsoft",
        "orgs": [
          "Microsoft"
        ],
        "url": "https://blogs.bing.com/webmaster/september-2023/Announcing-new-options-for-webmasters-to-control-usage-of-their-content-in-Bing-Chat"
      },
      "id": "opt_out:nocache-and-noarchive-meta-tags-for-bing-chat-microsoft-generative-ai",
      "name": "NOCACHE and NOARCHIVE meta tags for Bing Chat / Microsoft generative AI",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "Microsoft": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "blogs.bing.com"
        ],
        "source_urls": [
          "https://blogs.bing.com/webmaster/september-2023/Announcing-new-options-for-webmasters-to-control-usage-of-their-content-in-Bing-Chat"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://blogs.bing.com/webmaster/september-2023/Announcing-new-options-for-webmasters-to-control-usage-of-their-content-in-Bing-Chat"
    },
    {
      "fields": {
        "controls": "search index (appearance in ChatGPT search answers)",
        "default_state": "opt-out",
        "does_not_cover": "Training (governed by GPTBot); user-initiated fetches (ChatGPT-User)",
        "honoured_by": [
          "OpenAI (self-declared)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "OAI-SearchBot robots.txt token",
        "note": "OpenAI states opted-out sites are not shown in ChatGPT search answers; changes take about 24 hours. Blocking it trades visibility for control, separate from the training decision.",
        "operator": "OpenAI",
        "orgs": [
          "OpenAI"
        ],
        "url": "https://developers.openai.com/api/docs/bots"
      },
      "id": "opt_out:oai-searchbot-robots-txt-token",
      "name": "OAI-SearchBot robots.txt token",
      "notes": {
        "_org_roles": {
          "OpenAI": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "developers.openai.com"
        ],
        "source_urls": [
          "https://developers.openai.com/api/docs/bots"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://developers.openai.com/api/docs/bots"
    },
    {
      "fields": {
        "controls": "user-initiated fetch",
        "default_state": null,
        "does_not_cover": "Operator states it generally ignores robots.txt, so the token is identification only, not an opt-out",
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "Perplexity-User token",
        "note": "Blocking requires network-level controls (WAF, Cloudflare).",
        "operator": "Perplexity",
        "orgs": [
          "Perplexity"
        ],
        "url": "https://docs.perplexity.ai/guides/bots"
      },
      "id": "opt_out:perplexity-user-token",
      "name": "Perplexity-User token",
      "notes": {
        "_org_roles": {
          "Perplexity": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "docs.perplexity.ai"
        ],
        "source_urls": [
          "https://docs.perplexity.ai/guides/bots"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.727,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://docs.perplexity.ai/guides/bots"
    },
    {
      "fields": {
        "controls": "search index (surfacing sites in Perplexity results)",
        "default_state": "opt-out",
        "does_not_cover": "Perplexity-User fetches, which Perplexity says generally ignore robots.txt; undeclared crawling observed by Cloudflare",
        "honoured_by": [
          "Perplexity (self-declared)"
        ],
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "robots-token",
        "name": "PerplexityBot robots.txt token",
        "note": "Perplexity states its bots are not used to collect content for training foundation models. Cloudflare test (2025-08-04) reported undeclared crawlers impersonating Chrome evading no-crawl directives, and delisted Perplexity as a verified bot: https://blog.cloudflare.com/perplexity-is-using-stealth-undeclared-crawlers-to-evade-website-no-crawl-directives/",
        "operator": "Perplexity",
        "orgs": [
          "Perplexity"
        ],
        "url": "https://docs.perplexity.ai/guides/bots"
      },
      "id": "opt_out:perplexitybot-robots-txt-token",
      "name": "PerplexityBot robots.txt token",
      "notes": {
        "_org_roles": {
          "Perplexity": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "docs.perplexity.ai"
        ],
        "source_urls": [
          "https://docs.perplexity.ai/guides/bots"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://docs.perplexity.ai/guides/bots"
    },
    {
      "fields": {
        "controls": "machine-readable licensing terms (free, attribution, subscription, pay-per-crawl, pay-per-inference) in robots.txt, HTML, headers, RSS, media",
        "default_state": null,
        "does_not_cover": "No AI developer commitment to honour it listed on the site; licensing enforcement depends on CDN partners or contracts",
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "standard",
        "name": "RSL (Really Simple Licensing)",
        "note": "RSL 1.0 published; backers include Cloudflare, Akamai, Fastly, Reddit, Yahoo, Ziff Davis, O'Reilly, Creative Commons.",
        "operator": [
          "RSL Collective",
          "RSL Internet Collective"
        ],
        "orgs": [
          "RSL Collective",
          "RSL Internet Collective"
        ],
        "url": "https://rslstandard.org/"
      },
      "id": "opt_out:rsl-really-simple-licensing",
      "name": "RSL (Really Simple Licensing)",
      "notes": {
        "_org_roles": {
          "RSL Collective": [
            "operator"
          ],
          "RSL Internet Collective": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "rslstandard.org"
        ],
        "source_urls": [
          "https://rslstandard.org/"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.727,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "internal-prose-removed",
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://rslstandard.org/"
    },
    {
      "fields": {
        "controls": "site-level and asset-level AI training preferences; registry of opted-out works",
        "default_state": null,
        "does_not_cover": "Could not verify current status: spawning.ai showed an under-maintenance page on 2026-09-16; honouring parties not verified",
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "standard",
        "name": "Spawning ai.txt and Do Not Train registry",
        "note": "Record kept to flag a possibly defunct mechanism. Re-verify before publishing; historical claims of Stability AI and Hugging Face honouring the registry were not confirmed from a readable source.",
        "operator": "Spawning",
        "orgs": [
          "Spawning"
        ],
        "url": "https://spawning.ai/"
      },
      "id": "opt_out:spawning-ai-txt-and-do-not-train-registry",
      "name": "Spawning ai.txt and Do Not Train registry",
      "notes": {
        "_org_roles": {
          "Spawning": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "spawning.ai"
        ],
        "source_urls": [
          "https://spawning.ai/"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.727,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://spawning.ai/"
    },
    {
      "fields": {
        "controls": "adds about 25 named AI bots to the site's robots.txt disallow list",
        "default_state": "off",
        "does_not_cover": "Only cooperative crawlers; no retroactive removal; no page-level control",
        "jurisdiction": "global",
        "legal_force": "voluntary",
        "mechanism": "platform-toggle",
        "name": "Squarespace Block known artificial intelligence crawlers",
        "note": "Squarespace defaults it off to preserve chatbot referral traffic.",
        "operator": "Squarespace",
        "orgs": [
          "Squarespace"
        ],
        "url": "https://support.squarespace.com/hc/en-us/articles/360022347072"
      },
      "id": "opt_out:squarespace-block-known-artificial-intelligence-crawlers",
      "name": "Squarespace Block known artificial intelligence crawlers",
      "notes": {
        "_org_roles": {
          "Squarespace": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "support.squarespace.com"
        ],
        "source_urls": [
          "https://support.squarespace.com/hc/en-us/articles/360022347072"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://support.squarespace.com/hc/en-us/articles/360022347072"
    },
    {
      "fields": {
        "controls": "machine-readable EU Article 4 reservation (tdm-reservation) plus pointer to licensing terms (tdm-policy)",
        "default_state": "opt-out",
        "does_not_cover": "Is a W3C Community Group report, not a W3C Standard; no crawler operator documentation read commits to reading it; content mined before publication",
        "effective": "2024-05-10",
        "jurisdiction": "EU",
        "legal_force": "voluntary",
        "mechanism": "standard",
        "name": "TDM Reservation Protocol (TDMRep)",
        "note": "Expressible via /.well-known/tdmrep.json, HTTP headers, HTML meta, EPUB and PDF XMP. Its legal force derives from Article 4(3), not from the protocol. Carries a licensing URL, so it supports the get-paid route.",
        "operator": "W3C TDM Reservation Protocol Community Group",
        "orgs": [
          "W3C TDM Reservation Protocol Community Group"
        ],
        "url": "https://www.w3.org/community/reports/tdmrep/CG-FINAL-tdmrep-20240510"
      },
      "id": "opt_out:tdm-reservation-protocol-tdmrep",
      "name": "TDM Reservation Protocol (TDMRep)",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "W3C TDM Reservation Protocol Community Group": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "w3.org"
        ],
        "source_urls": [
          "https://www.w3.org/community/reports/tdmrep/CG-FINAL-tdmrep-20240510"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.909,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://www.w3.org/community/reports/tdmrep/CG-FINAL-tdmrep-20240510"
    },
    {
      "fields": {
        "controls": "would have allowed commercial TDM unless rights reserved",
        "default_state": null,
        "does_not_cover": "Not law: March 2026 report says a broad exception with opt-out is no longer the preferred way forward; current law (CDPA s29A) permits TDM only for non-commercial research, so there is no statutory opt-out to exercise",
        "effective": "2026-03-18",
        "jurisdiction": "UK",
        "legal_force": "proposed",
        "mechanism": "legal-reservation",
        "name": "UK TDM opt-out exception (proposed, abandoned)",
        "note": "In the UK commercial AI training on copyright works has no exception, so the lever is ordinary copyright, not an opt-out. Government will support market-led standards.",
        "operator": "UK Government",
        "orgs": [
          "UK Government"
        ],
        "url": "https://www.gov.uk/government/publications/report-and-impact-assessment-on-copyright-and-artificial-intelligence/report-on-copyright-and-artificial-intelligence"
      },
      "id": "opt_out:uk-tdm-opt-out-exception-proposed-abandoned",
      "name": "UK TDM opt-out exception (proposed, abandoned)",
      "notes": {
        "_effective_precision": "day",
        "_org_roles": {
          "UK Government": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "gov.uk"
        ],
        "source_urls": [
          "https://www.gov.uk/government/publications/report-and-impact-assessment-on-copyright-and-artificial-intelligence/report-on-copyright-and-artificial-intelligence"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://www.gov.uk/government/publications/report-and-impact-assessment-on-copyright-and-artificial-intelligence/report-on-copyright-and-artificial-intelligence"
    },
    {
      "fields": {
        "controls": "removes site from WordPress.com's third-party content and research partner sharing (including AI) and adds AI bots to robots.txt disallow",
        "default_state": null,
        "does_not_cover": "Robots.txt part depends on AI platforms honouring it; per-site setting; no statement on data already shared",
        "honoured_by": [
          "Automattic (for its own partner sharing)"
        ],
        "jurisdiction": "global",
        "legal_force": "contractual",
        "mechanism": "platform-toggle",
        "name": "WordPress.com Prevent third-party sharing",
        "note": "Two layers: a contractual exclusion from Automattic's own data deals, plus a voluntary crawler request. Tumblr has a similar setting but its help page 404ed and was not verified.",
        "operator": "Automattic",
        "orgs": [
          "Automattic"
        ],
        "url": "https://wordpress.com/support/privacy-settings/make-your-website-public"
      },
      "id": "opt_out:wordpress-com-prevent-third-party-sharing",
      "name": "WordPress.com Prevent third-party sharing",
      "notes": {
        "_org_roles": {
          "Automattic": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "wordpress.com"
        ],
        "source_urls": [
          "https://wordpress.com/support/privacy-settings/make-your-website-public"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://wordpress.com/support/privacy-settings/make-your-website-public"
    },
    {
      "fields": {
        "controls": "whether channel videos may be used by approved third-party companies to train AI models",
        "default_state": "off",
        "does_not_cover": "Default off means no third-party permission, but it does not stop scraping by parties outside the programme; page does not cover Google's own training; no statement on data already used; changes can take up to 7 days to reflect in the Data API",
        "jurisdiction": "global",
        "legal_force": "contractual",
        "mechanism": "platform-toggle",
        "name": "YouTube third-party training setting",
        "note": "This is an opt-IN to licensing, not an opt-out. List of eligible companies not shown on page read.",
        "operator": "YouTube",
        "orgs": [
          "YouTube"
        ],
        "url": "https://support.google.com/youtube/answer/15509945"
      },
      "id": "opt_out:youtube-third-party-training-setting",
      "name": "YouTube third-party training setting",
      "notes": {
        "_org_roles": {
          "YouTube": [
            "operator"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery-explore",
        "source_hosts": [
          "support.google.com"
        ],
        "source_urls": [
          "https://support.google.com/youtube/answer/15509945"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.818,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 90
      },
      "type": "opt_out",
      "url": "https://support.google.com/youtube/answer/15509945"
    }
  ],
  "_meta": {
    "source": "Blomega Data Refinery",
    "url": "https://data.blomega.com",
    "publisher": "Blomega",
    "publisher_url": "https://blomegalab.com",
    "wikidata": "Q141048865",
    "license": "CC BY 4.0",
    "license_url": "https://creativecommons.org/licenses/by/4.0/",
    "cite_as": "Blomega Data Refinery (https://data.blomega.com), CC BY 4.0. Cite the registry and the record id.",
    "attribution_required": true
  }
}
