{
  "total": 76,
  "limit": 50,
  "offset": 0,
  "records": [
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Adobe_Firefly_2026_07_22.pdf",
        "cutoff_date": "2025-09",
        "data_sources_named": [
          "OpenImages v7",
          "Flickr.com"
        ],
        "developer": "Adobe",
        "eu_training_summary_url": "https://www.adobe.com/cc-shared/assets/pdf/trust-center/ungated/whitepapers/creative-cloud/adobe-firefly-image-model-5-training-set.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Adobe Firefly Image Model 5",
        "note": "Release date basis: EU placement date stated in the summary. No crawler (crawler_named null because none was used). Licensed image and video Yes, but Adobe Stock is not named even though the registry carries the Adobe Stock contributors to Adobe deal (2023-09). Synthetic data from 'Firefly Image Model 5' and an 'Internal Adobe model'. Not a Code of Practice signatory; 'TDM exception was not relied on'. Registry also holds Dorcus v. Adobe Inc. weights=closed from product-only availability, not re-verified.",
        "orgs": [
          "Adobe"
        ],
        "release_date": "2025-10-28",
        "status_note": "2.3 'Data crawled and scraped from online sources': 'Were crawlers used by the provider or on behalf of? No'. Purposes and behaviour both 'N/A'. The content field says 'Adobe did not crawl online sources. Instead, we searched specifically for content licensed under CC0 ... within select sites'. 'Summary of the most relevant domain names crawled' is 'Flickr.com'. No crawler or user agent is named anywhere in the summary.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Adobe did not crawl online sources. Instead, we searched specifically for content licensed under CC0 (license dedicating content to the public domain) or in the public domain in limited amounts within select sites",
        "update_note": "updated 2026-09-18 from raw.githubusercontent.com: Names a domain (Flickr.com) and the OpenImages v7 dataset, never a crawler. Declares it did not crawl. crawler_named deliberately left unset: the summary names no crawler. names a third-party corpus but no crawler. developer URL https://www.adobe.com/cc-shared/assets/pdf/trust-center/ungated/whitepapers/creative-cloud/adobe-firefly-image-model-5-t",
        "url": "https://www.adobe.com/cc-shared/assets/pdf/trust-center/ungated/whitepapers/creative-cloud/adobe-firefly-image-model-5-training-set.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": false,
        "weights": "closed"
      },
      "id": "model:adobe-firefly-image-model-5",
      "name": "Adobe Firefly Image Model 5",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Adobe": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "adobe.com",
          "raw.githubusercontent.com"
        ],
        "source_urls": [
          "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Adobe_Firefly_2026_07_22.pdf",
          "https://www.adobe.com/cc-shared/assets/pdf/trust-center/ungated/whitepapers/creative-cloud/adobe-firefly-image-model-5-training-set.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.923,
        "confidence": "high",
        "corroborated": true,
        "flags": [],
        "grade": "A",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 2,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://www.adobe.com/cc-shared/assets/pdf/trust-center/ungated/whitepapers/creative-cloud/adobe-firefly-image-model-5-training-set.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Nova_2_Lite_2026_08_03.pdf",
        "crawler_named": [
          "Amazonbot"
        ],
        "cutoff_date": "2025-10",
        "data_sources_named": [
          "Amazonbot (crawler)",
          "Amazon Nova Premier (synthetic data generator)"
        ],
        "developer": "Amazon",
        "eu_training_summary_url": "https://docs.aws.amazon.com/ai/responsible-ai/nova-2-lite/overview.html",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Amazon Nova 2 Lite",
        "note": "Release date basis: EU placement date stated in the summary. Licensed Text, image, video and audio: Yes, unnamed; private datasets 'subject to confidentiality terms'. Registry holds New York Times, Conde Nast and Hearst, and Shutterstock deals with Amazon (NYT deal 2025-05-29 predates the October 2025 cutoff) plus Rogers v. Amazon.com; none named. Crawl period November 2023 to December 2024. weights=closed from Bedrock-only availability, not re-verified this session.",
        "orgs": [
          "Amazon"
        ],
        "release_date": "2025-12-02",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Nova 2 Lite was trained on text content that included reference materials, technical documentation, source code, and general web content.",
        "url": "https://docs.aws.amazon.com/ai/responsible-ai/nova-2-lite/overview.html",
        "uses_interaction_data": true,
        "uses_other_service_data": true,
        "weights": "closed"
      },
      "id": "model:amazon-nova-2-lite",
      "name": "Amazon Nova 2 Lite",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Amazon": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "docs.aws.amazon.com"
        ],
        "source_urls": [
          "https://docs.aws.amazon.com/ai/responsible-ai/nova-2-lite/overview.html"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://docs.aws.amazon.com/ai/responsible-ai/nova-2-lite/overview.html"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Apertus_1_5_2026_08_05.pdf",
        "cutoff_date": "2026-04-28",
        "data_sources_named": [
          "HuggingFaceFW/fineweb-2",
          "HuggingFaceTB/dclm-edu",
          "nvidia/Nemotron-CC-v2.1",
          "HuggingFaceFW/finePDFs-edu",
          "joelniklaus/Multi_Legal_Pile",
          "nvidia/Nemotron-Pretraining-Code-v1",
          "mozilla/CommonVoice24",
          "speechcolab/gigaspeech",
          "MLCommons/peoples_speech",
          "facebookresearch/voxpopuli",
          "k2-fsa/libriheavy",
          "facebook/omnilingual-asr-corpus",
          "mlfoundations/MINT-1T",
          "UCSC-VLAA/Recap-DataComp-1B",
          "dclure/laion-aesthetics-12m-umap",
          "mvp-lab/LLaVA-OneVision-1.5-Mid-Training-85M",
          "pixmo-cap"
        ],
        "developer": "Swiss AI Initiative",
        "eu_training_summary_url": "https://huggingface.co/swiss-ai/Apertus-v1.5-70B/blob/main/Apertus_1_5_EU_Public_Summary.pdf",
        "licensed_data_declared": "no",
        "licensed_data_named": false,
        "name": "Apertus v1.5 (8B, 70B, Instruct)",
        "note": "Release date basis: EU placement date stated in the summary. Control case: the only summary in the sample that names its datasets, each with a license (17 trillion tokens stated). licensed_data_named=false because no licensed data exists to name (2.2.1 No, 2.2.2 No). Retroactive robots.txt opt-out filtering since 2013. data_sources_named is a partial list (text, audio, image sections read, list continues). Crawlers Yes, no name given. weights=open verified via Hugging Face API: swiss-ai/Apertus-v1.5-70B, apache-2.0, gated=auto (click-through).",
        "orgs": [
          "Swiss AI Initiative"
        ],
        "release_date": "2026-07-14",
        "status_note": "2.3 'Data crawled and scraped from online sources': 'Were crawlers used by the provider or on behalf of? [x] Yes'. Every sub-field of 2.3 (crawler name/identifier, purposes, behaviour, period, domains) is absent from the document: the template jumps straight from the Yes tick to 2.4 User data. 3.1 says removals were applied for websites that opted out 'by specifying at least one of the common AI crawlers, at the time of January 2025', without naming any of them.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Public text-only datasets derived mainly from web documents written in over 1000 languages, while making significant efforts were made to respect consent",
        "update_note": "updated 2026-09-18 from raw.githubusercontent.com: Contradiction worth recording: the v1.5 summary ticks Yes to crawler use and then omits the whole crawler block, while the earlier Apertus v1 summary ticked No. Names only datasets (fineweb-2, dclm-edu, Nemotron-CC-v2.1 and others). crawler_named deliberately left unset: the summary names no crawler. names a third-party corpus but no crawler. deve",
        "url": "https://huggingface.co/swiss-ai/Apertus-v1.5-70B/blob/main/Apertus_1_5_EU_Public_Summary.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": false,
        "weights": "open"
      },
      "id": "model:apertus-v1-5-8b-70b-instruct",
      "name": "Apertus v1.5 (8B, 70B, Instruct)",
      "notes": {
        "_cutoff_date_precision": "day",
        "_org_roles": {
          "Swiss AI Initiative": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "huggingface.co",
          "raw.githubusercontent.com"
        ],
        "source_urls": [
          "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Apertus_1_5_2026_08_05.pdf",
          "https://huggingface.co/swiss-ai/Apertus-v1.5-70B/blob/main/Apertus_1_5_EU_Public_Summary.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.923,
        "confidence": "high",
        "corroborated": true,
        "flags": [],
        "grade": "A",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 2,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://huggingface.co/swiss-ai/Apertus-v1.5-70B/blob/main/Apertus_1_5_EU_Public_Summary.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Apertus_2025_11_12.pdf",
        "cutoff_date": "2024-03",
        "data_sources_named": [
          "HuggingFaceFW/fineweb-edu",
          "epfml/FineWeb-HQ",
          "HuggingFaceTB/dclm-edu",
          "epfml/FineWeb2-HQ",
          "HuggingFaceFW/fineweb-2",
          "bigcode/the-stack-dedup",
          "common-pile/stackv2_edu_filtered",
          "HuggingFaceTB/finemath",
          "LLM360/MegaMath",
          "Wikipedia",
          "HuggingFaceTB/smoltalk2",
          "utter-project/EuroBlocks-SFT-Synthetic-1124",
          "allenai/tulu-3-sft-olmo-2-mixture-0225",
          "DataProvenanceInitiative/Commercial-Flan-Collection-Chain-Of-Thought",
          "Project Gutenberg"
        ],
        "developer": "Swiss AI Initiative",
        "eu_training_summary_url": "https://huggingface.co/swiss-ai/Apertus-70B-2509/blob/main/Apertus_EU_Public_Summary.pdf",
        "licensed_data_declared": "no",
        "licensed_data_named": false,
        "name": "Apertus v1 (8B, 70B, Instruct)",
        "note": "Release date basis: EU placement date stated in the summary ('Sept 2nd, 2025'). First Apertus release, distinct from the carried Apertus v1.5 row. Summary V1, last update 01/09/2025. Ticks are '☒' glyphs in the text layer: 2.2.1 No, 2.2.2 No, crawlers No, user data No/No, synthetic No, other sources No, Code of Practice signatory No. 15 trillion tokens, knowledge cutoff 'Main pretraining dataset knowledge cutoff is 03/2024', with later math and post-training data. Synthetic No although the post-training mix includes third-party synthetic sets (EuroBlocks-SFT-Synthetic), reported under public datasets. Retroactive AI-crawler opt-out filtering to January 2025. weights=open verified via Hugging Face API: swiss-ai/Apertus-70B-2509 public, not gated, apache-2.0. Nulls: crawler_named: crawlers answered No.",
        "orgs": [
          "Swiss AI Initiative"
        ],
        "release_date": "2025-09-02",
        "status_note": "2.3 'Data crawled and scraped from online sources': 'Were crawlers used by the provider or on behalf of? [x] No'. No crawler sub-fields follow. 3.1 refers to websites that opted out 'by specifying at least one of the common AI crawlers, at the time of January 2025' without naming one.",
        "synthetic_data_stated": false,
        "training_data_disclosure": "Public text-only datasets derived mainly from web documents, in over 1000 languages.",
        "update_note": "updated 2026-09-18 from huggingface.co: No crawler named. Web content reaches the model through third-party corpora (fineweb-edu, FineWeb2-HQ, DCLM-edu, the-stack-dedup). crawler_named deliberately left unset: the summary names no crawler. names a third-party corpus but no crawler.",
        "url": "https://huggingface.co/swiss-ai/Apertus-70B-2509/blob/main/Apertus_EU_Public_Summary.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": false,
        "weights": "open"
      },
      "id": "model:apertus-v1-8b-70b-instruct",
      "name": "Apertus v1 (8B, 70B, Instruct)",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Swiss AI Initiative": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "huggingface.co"
        ],
        "source_urls": [
          "https://huggingface.co/swiss-ai/Apertus-70B-2509/blob/main/Apertus_EU_Public_Summary.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://huggingface.co/swiss-ai/Apertus-70B-2509/blob/main/Apertus_EU_Public_Summary.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Pleias_Baguettotron_2026_09_08.pdf",
        "cutoff_date": "2025-11",
        "data_sources_named": [
          "SYNTH (PleIAs/SYNTH)",
          "Structured Wikipedia (Wikimedia Enterprise)",
          "Wikibooks",
          "AI-MO/Kimina-Prover-Promptset",
          "Qwen 3 8B (synthetic data generator)",
          "DeepSeek-Prover-V2-7B (synthetic data generator)"
        ],
        "developer": "Pleias",
        "eu_training_summary_url": "https://pleias.ai/training-content/baguettotron-monad",
        "licensed_data_declared": "no",
        "licensed_data_named": false,
        "name": "Baguettotron and Monad",
        "note": "Release date basis: weights publication and announcement date stated in the summary's placement field (10 November 2025); no separate EU placement date given. Summary version 1.0, last update 04/08/2026. Ticks are '☒' glyphs: 2.2.1 No, 2.2.2 No, crawlers No, user data No/No, synthetic Yes, other sources Yes (130 staff-written documents amplified about 10,000 times), Code of Practice signatory Yes. Entire corpus is synthetic: SYNTH, 79.6 million samples, about 75B tokens, generated from 62,555 seed documents (Wikipedia vital articles, Wikibooks). Cutoff '11/2025'. weights=open verified via Hugging Face API: PleIAs/Baguettotron and PleIAs/Monad public, not gated, apache-2.0. Nulls: crawler_named: crawlers answered No.",
        "orgs": [
          "Pleias"
        ],
        "release_date": "2025-11-10",
        "status_note": "2.3: 'Were crawlers used by the provider or on behalf of? [x] No'. Crawler name field: 'Not applicable. No crawler was used by PLEIAS or on its behalf for the training of these models.' Behaviour field: seed material came from 'machine-readable dumps published by the Wikimedia Foundation through its Wikimedia Enterprise service and from the official Wikimedia API ... Obtaining data through them does not involve crawling or scraping'. Additional comments: 'The absence of any crawling is a deliberate design property of these models'.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Synthetic instructional and reasoning text, generated in its entirety by AI models from a fixed set of openly licensed encyclopaedic seed documents.",
        "update_note": "updated 2026-09-18 from pleias.ai: One of the clearest no-crawler declarations in the sample; names Wikimedia Enterprise and the Wikimedia API as bulk interfaces rather than a crawler. crawler_named deliberately left unset: the summary names no crawler. names a third-party corpus but no crawler.",
        "url": "https://pleias.ai/training-content/baguettotron-monad",
        "uses_interaction_data": false,
        "uses_other_service_data": false,
        "weights": "open"
      },
      "id": "model:baguettotron-and-monad",
      "name": "Baguettotron and Monad",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Pleias": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "pleias.ai"
        ],
        "source_urls": [
          "https://pleias.ai/training-content/baguettotron-monad"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://pleias.ai/training-content/baguettotron-monad"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Bielik_3_2026_01_12.pdf",
        "crawler_named": [
          "Speakleash"
        ],
        "cutoff_date": "2025-05",
        "data_sources_named": [
          "HPLT v2.0",
          "CulturaX",
          "FineWeb-2",
          "FineWeb-Edu",
          "Common Crawl",
          "SlimPajama-627B",
          "Wikipedia",
          "Biblioteka Nauki (Science Library)",
          "Korpus Dyskursu Parlamentarnego",
          "Europeana",
          "itwiz.pl (licensor)",
          "Speakleash (crawler)",
          "DeepSeek v3 (synthetic data generator)",
          "Bielik v2.3 11B (synthetic data generator)"
        ],
        "developer": "SpeakLeash",
        "eu_training_summary_url": "https://bielik.ai/downloads/Bielik%2011B%20v3%20EU%20Public%20Summary.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": true,
        "name": "Bielik v3 11B Instruct",
        "note": "Release date basis: EU placement date stated in the summary ('31.12.2025'). 2.2.1 'X Yes' confirmed on the rendered page 4 image, and names the licensor: 'Agreements were concluded with the itwiz.pl editorial office.' (about ten years of ITWiz articles delivered as DOCX). 2.2.2 No, crawlers Yes (identifier 'Speakleash', 11/2022 to 05/2025, .gov and .bip domains plus thematic forums), user data No/No, synthetic Yes. Model dependency Mistral 7B v0.2. Cutoff stated 'May 2025', though section 2.1 says public datasets cover data acquired 'between 2022 and November 2025'. Summary version 1.0, 'Last update: None'. weights=open verified via Hugging Face API: speakleash/Bielik-11B-v3.0-Instruct public, gated=auto, apache-2.0.",
        "orgs": [
          "SpeakLeash"
        ],
        "release_date": "2025-12-31",
        "synthetic_data_stated": true,
        "training_data_disclosure": "legal and official documents (e.g., court rulings, legal acts, regulations), scientific texts (from the Science Library), press publications (from commercially licensed sources), web content from public domains and thematic forums",
        "url": "https://bielik.ai/downloads/Bielik%2011B%20v3%20EU%20Public%20Summary.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": false,
        "weights": "open"
      },
      "id": "model:bielik-v3-11b-instruct",
      "name": "Bielik v3 11B Instruct",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "SpeakLeash": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "bielik.ai"
        ],
        "source_urls": [
          "https://bielik.ai/downloads/Bielik%2011B%20v3%20EU%20Public%20Summary.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://bielik.ai/downloads/Bielik%2011B%20v3%20EU%20Public%20Summary.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Bria_32_2026_01_12.pdf",
        "cutoff_date": "2025-06",
        "developer": "Bria AI",
        "eu_training_summary_url": "https://drive.google.com/file/d/13BhHQhd7vcDArPmpBrl_b2FpdBqkYU4n/view",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Bria 3.2",
        "note": "Release date basis: EU placement date stated in the summary. The one fully-licensed image model in the sample (479 million images, no public datasets, no crawling, synthetic No) and still names no partner, although the registry carries Getty Images to Bria AI and Envato to Bria AI deals. Bria does disclose it uses third-party AI captions for enrichment, which the template excludes from 'synthetic'. weights null: Hugging Face API for briaai/BRIA-3.2 returned 401 (gated or moved), not established. Summary hosted on Google Drive, a fragile location. Nulls: weights: not established, the model is not published on Hugging Face under the developer's account and no second party states it",
        "orgs": [
          "Bria AI"
        ],
        "release_date": "2025-06-10",
        "status_note": "2.3 'Data Crawled and Scraped from Online Sources': 'Q: Were crawlers used by the provider or on behalf of? A: No'. An annex headed 'No Web-Crawling Policy' adds 'Bria does not and will not engage in web-crawling activities or utilize publicly [available web data]'. No crawler is named and no third-party dataset is named anywhere in the summary.",
        "synthetic_data_stated": false,
        "training_data_disclosure": "Bria's training data is sourced exclusively through commercial licensing agreements with data partners globally",
        "update_note": "updated 2026-09-18 from raw.githubusercontent.com: Fully licensed image model. Names neither a crawler nor a corpus nor a licensor. crawler_named deliberately left unset: the summary names no crawler. developer URL https://drive.google.com/file/d/13BhHQhd7vcDArPmpBrl_b2FpdBqkYU4n/view returned HTTP 404 on 2026-09-18, read from the AI Accountability Lab archive copy https://raw.githubusercontent.co",
        "url": "https://drive.google.com/file/d/13BhHQhd7vcDArPmpBrl_b2FpdBqkYU4n/view",
        "uses_interaction_data": false,
        "uses_other_service_data": false
      },
      "id": "model:bria-3-2",
      "name": "Bria 3.2",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Bria AI": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "drive.google.com",
          "raw.githubusercontent.com"
        ],
        "source_urls": [
          "https://drive.google.com/file/d/13BhHQhd7vcDArPmpBrl_b2FpdBqkYU4n/view",
          "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Bria_32_2026_01_12.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.769,
        "confidence": "high",
        "corroborated": true,
        "flags": [],
        "grade": "B",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 2,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://drive.google.com/file/d/13BhHQhd7vcDArPmpBrl_b2FpdBqkYU4n/view"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/GPT_Image_2_2026_08_17.pdf",
        "crawler_named": [
          "GPTBot"
        ],
        "cutoff_date": "2026-04",
        "data_sources_named": [
          "Common Crawl",
          "GPTBot (crawler)"
        ],
        "developer": "OpenAI",
        "eu_training_summary_url": "https://cdn.openai.com/pdf/chatgpt-images-2-0-eu-ai-act-public-summary-of-training-content.pdf",
        "licensed_data_declared": "other",
        "licensed_data_named": false,
        "name": "ChatGPT Images 2.0 (GPT Image 2)",
        "note": "Release date basis: EU placement date stated in the summary. Same 'Other (see below)' partnership wording as GPT-5.5, modalities Text and Image. Unlike GPT-5.5, section 2.2.2 answers No to private third-party datasets, and user data from model interactions is Yes here but No for GPT-5.5. The registry holds a Shutterstock to OpenAI image deal (2023-07) that this image model summary does not name. Model name in the summary is 'ChatGPT Images 2.0'; GPAI Ledger lists it as 'GPT Image 2.0'.",
        "orgs": [
          "OpenAI"
        ],
        "release_date": "2026-04-21",
        "synthetic_data_stated": true,
        "training_data_disclosure": "ChatGPT Images 2.0 was trained on a large-scale, multilingual mixture of publicly available data, data accessed through partnerships, synthetic data, and human-generated text",
        "url": "https://cdn.openai.com/pdf/chatgpt-images-2-0-eu-ai-act-public-summary-of-training-content.pdf",
        "uses_interaction_data": true,
        "uses_other_service_data": true,
        "weights": "closed"
      },
      "id": "model:chatgpt-images-2-0-gpt-image-2",
      "name": "ChatGPT Images 2.0 (GPT Image 2)",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "OpenAI": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "cdn.openai.com"
        ],
        "source_urls": [
          "https://cdn.openai.com/pdf/chatgpt-images-2-0-eu-ai-act-public-summary-of-training-content.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://cdn.openai.com/pdf/chatgpt-images-2-0-eu-ai-act-public-summary-of-training-content.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Claude_Fable_Mythos_5_1_2026_09_08.pdf",
        "crawler_named": [
          "ClaudeBot"
        ],
        "cutoff_date": "2026-08",
        "data_sources_named": [
          "Common Crawl",
          "GitHub (platform)",
          "HuggingFace (platform)",
          "ClaudeBot (crawler)",
          "acquired physical texts (unnamed)"
        ],
        "developer": "Anthropic",
        "eu_training_summary_url": "https://trust.anthropic.com/resources",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Claude Fable 5.1 and Claude Mythos 5.1",
        "note": "Release date basis: EU placement date stated in section 1.2. Summary: both are 'versions of the same underlying model, with differing safeguard configurations and access', Fable 5.1 generally available, Mythos 5.1 'available only through trusted access programs'. Same Anthropic template text as the Claude Opus 5 summary: 2.2.1 Yes to commercial licensing agreements for Text and Image, no licensor named; 2.2.2 Yes to private third-party datasets, list 'Not applicable'; 2.6 'A portion of the data corpus comes from acquired physical texts.' Synthetic data partly 'generated by Anthropic models not available on the market'. Cutoff wording: 'some data being acquired/collected up to August 2026'. Crawl period March 2024 to July 2026. Answers are typed words, not tick boxes, checked against a rendered page image. Trust-center URL is an index page; text read from the AI Accountability Lab mirror (third-party archive, corroborates the document text, not the claims). Nulls: weights: not established, Hugging Face API author=anthropic returns no models and the summary does not state API-only availability.",
        "orgs": [
          "Anthropic"
        ],
        "release_date": "2026-09-01",
        "synthetic_data_stated": true,
        "training_data_disclosure": "The training corpus is derived from several publicly accessible repositories, notably including Common Crawl, a repository of web crawl data, as well as specialized datasets available through platforms like GitHub and HuggingFace.",
        "url": "https://trust.anthropic.com/resources",
        "uses_interaction_data": true,
        "uses_other_service_data": true
      },
      "id": "model:claude-fable-5-1-and-claude-mythos-5-1",
      "name": "Claude Fable 5.1 and Claude Mythos 5.1",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Anthropic": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "trust.anthropic.com"
        ],
        "source_urls": [
          "https://trust.anthropic.com/resources"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://trust.anthropic.com/resources"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Claude_Mythos_5_Fable_5_2026_08_03.pdf",
        "crawler_named": [
          "ClaudeBot"
        ],
        "cutoff_date": "2026-04",
        "data_sources_named": [
          "Common Crawl",
          "GitHub (platform)",
          "HuggingFace (platform)",
          "ClaudeBot (crawler)",
          "acquired physical texts (unnamed)"
        ],
        "developer": "Anthropic",
        "eu_training_summary_url": "https://trust.anthropic.com/resources",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Claude Mythos 5 and Claude Fable 5",
        "note": "Release date basis: EU placement date stated in section 1.2. One summary covers both names. Same Anthropic template text as the Claude Opus 5 summary: 2.2.1 Yes to commercial licensing agreements for Text and Image, no licensor named; 2.2.2 Yes to private third-party datasets, list 'Not applicable'; 2.6 'A portion of the data corpus comes from acquired physical texts.' Synthetic data partly 'generated by Anthropic models not available on the market'. Cutoff wording: 'some data being acquired/collected up to April 2026'. Crawl period March 2024 to April 2026. Answers are typed words, not tick boxes, checked against a rendered page image. Trust-center URL is an index page; text read from the AI Accountability Lab mirror (third-party archive, corroborates the document text, not the claims). Nulls: weights: not established, Hugging Face API author=anthropic returns no models and the summary does not state API-only availability.",
        "orgs": [
          "Anthropic"
        ],
        "release_date": "2026-06-09",
        "synthetic_data_stated": true,
        "training_data_disclosure": "The training corpus is derived from several publicly accessible repositories, notably including Common Crawl, a repository of web crawl data, as well as specialized datasets available through platforms like GitHub and HuggingFace.",
        "url": "https://trust.anthropic.com/resources",
        "uses_interaction_data": true,
        "uses_other_service_data": true
      },
      "id": "model:claude-mythos-5-and-claude-fable-5",
      "name": "Claude Mythos 5 and Claude Fable 5",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Anthropic": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "trust.anthropic.com"
        ],
        "source_urls": [
          "https://trust.anthropic.com/resources"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://trust.anthropic.com/resources"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Claude_Mythos_Preview_2026_08_03.pdf",
        "crawler_named": [
          "ClaudeBot"
        ],
        "cutoff_date": "2026-02",
        "data_sources_named": [
          "Common Crawl",
          "GitHub (platform)",
          "HuggingFace (platform)",
          "ClaudeBot (crawler)",
          "acquired physical texts (unnamed)"
        ],
        "developer": "Anthropic",
        "eu_training_summary_url": "https://trust.anthropic.com/resources",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Claude Mythos Preview",
        "note": "Release date basis: EU placement date stated in section 1.2. Cutoff (February 2026) precedes the end of the stated crawl period (March 2026). Same Anthropic template text as the Claude Opus 5 summary: 2.2.1 Yes to commercial licensing agreements for Text and Image, no licensor named; 2.2.2 Yes to private third-party datasets, list 'Not applicable'; 2.6 'A portion of the data corpus comes from acquired physical texts.' Synthetic data partly 'generated by Anthropic models not available on the market'. Cutoff wording: 'some data being acquired/collected up to February 2026'. Crawl period March 2024 to March 2026. Answers are typed words, not tick boxes, checked against a rendered page image. Trust-center URL is an index page; text read from the AI Accountability Lab mirror (third-party archive, corroborates the document text, not the claims). Nulls: weights: not established, Hugging Face API author=anthropic returns no models and the summary does not state API-only availability.",
        "orgs": [
          "Anthropic"
        ],
        "release_date": "2026-06-02",
        "synthetic_data_stated": true,
        "training_data_disclosure": "The training corpus is derived from several publicly accessible repositories, notably including Common Crawl, a repository of web crawl data, as well as specialized datasets available through platforms like GitHub and HuggingFace.",
        "url": "https://trust.anthropic.com/resources",
        "uses_interaction_data": true,
        "uses_other_service_data": true
      },
      "id": "model:claude-mythos-preview",
      "name": "Claude Mythos Preview",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Anthropic": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "trust.anthropic.com"
        ],
        "source_urls": [
          "https://trust.anthropic.com/resources"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://trust.anthropic.com/resources"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Claude_Opus_4_7_2026_08_03.pdf",
        "crawler_named": [
          "ClaudeBot"
        ],
        "cutoff_date": "2026-04",
        "data_sources_named": [
          "Common Crawl",
          "GitHub (platform)",
          "HuggingFace (platform)",
          "ClaudeBot (crawler)",
          "acquired physical texts (unnamed)"
        ],
        "developer": "Anthropic",
        "eu_training_summary_url": "https://trust.anthropic.com/resources",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Claude Opus 4.7",
        "note": "Release date basis: EU placement date stated in section 1.2. Same Anthropic template text as the Claude Opus 5 summary: 2.2.1 Yes to commercial licensing agreements for Text and Image, no licensor named; 2.2.2 Yes to private third-party datasets, list 'Not applicable'; 2.6 'A portion of the data corpus comes from acquired physical texts.' Synthetic data partly 'generated by Anthropic models not available on the market'. Cutoff wording: 'some data being acquired/collected up to April 2026'. Crawl period March 2024 to April 2026. Answers are typed words, not tick boxes, checked against a rendered page image. Trust-center URL is an index page; text read from the AI Accountability Lab mirror (third-party archive, corroborates the document text, not the claims). Nulls: weights: not established, Hugging Face API author=anthropic returns no models and the summary does not state API-only availability.",
        "orgs": [
          "Anthropic"
        ],
        "release_date": "2026-04-16",
        "synthetic_data_stated": true,
        "training_data_disclosure": "The training corpus is derived from several publicly accessible repositories, notably including Common Crawl, a repository of web crawl data, as well as specialized datasets available through platforms like GitHub and HuggingFace.",
        "url": "https://trust.anthropic.com/resources",
        "uses_interaction_data": true,
        "uses_other_service_data": true
      },
      "id": "model:claude-opus-4-7",
      "name": "Claude Opus 4.7",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Anthropic": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "trust.anthropic.com"
        ],
        "source_urls": [
          "https://trust.anthropic.com/resources"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://trust.anthropic.com/resources"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Claude_Opus_4_8_2026_08_03.pdf",
        "crawler_named": [
          "ClaudeBot"
        ],
        "cutoff_date": "2026-05",
        "data_sources_named": [
          "Common Crawl",
          "GitHub (platform)",
          "HuggingFace (platform)",
          "ClaudeBot (crawler)",
          "acquired physical texts (unnamed)"
        ],
        "developer": "Anthropic",
        "eu_training_summary_url": "https://trust.anthropic.com/resources",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Claude Opus 4.8",
        "note": "Release date basis: EU placement date stated in section 1.2. Same Anthropic template text as the Claude Opus 5 summary: 2.2.1 Yes to commercial licensing agreements for Text and Image, no licensor named; 2.2.2 Yes to private third-party datasets, list 'Not applicable'; 2.6 'A portion of the data corpus comes from acquired physical texts.' Synthetic data partly 'generated by Anthropic models not available on the market'. Cutoff wording: 'some data being acquired/collected up to May 2026'. Crawl period March 2024 to May 2026. Answers are typed words, not tick boxes, checked against a rendered page image. Trust-center URL is an index page; text read from the AI Accountability Lab mirror (third-party archive, corroborates the document text, not the claims). Nulls: weights: not established, Hugging Face API author=anthropic returns no models and the summary does not state API-only availability.",
        "orgs": [
          "Anthropic"
        ],
        "release_date": "2026-05-28",
        "synthetic_data_stated": true,
        "training_data_disclosure": "The training corpus is derived from several publicly accessible repositories, notably including Common Crawl, a repository of web crawl data, as well as specialized datasets available through platforms like GitHub and HuggingFace.",
        "url": "https://trust.anthropic.com/resources",
        "uses_interaction_data": true,
        "uses_other_service_data": true
      },
      "id": "model:claude-opus-4-8",
      "name": "Claude Opus 4.8",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Anthropic": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "trust.anthropic.com"
        ],
        "source_urls": [
          "https://trust.anthropic.com/resources"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://trust.anthropic.com/resources"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Claude_Opus_5_2026_08_03.pdf",
        "crawler_named": [
          "ClaudeBot"
        ],
        "cutoff_date": "2026-07",
        "data_sources_named": [
          "Common Crawl",
          "GitHub (platform)",
          "HuggingFace (platform)",
          "ClaudeBot (crawler)",
          "acquired physical texts (unnamed)"
        ],
        "developer": "Anthropic",
        "eu_training_summary_url": "https://trust.anthropic.com/resources",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Claude Opus 5",
        "note": "Release date basis: EU placement date stated in the summary. Section 2.2.1 answers Yes to commercial licensing agreements for Text and Image but names no licensor. Section 2.6 adds 'A portion of the data corpus comes from acquired physical texts.' Cutoff wording: 'some data being acquired/collected up to July 2026'. weights=closed is from the provider selling API access only; not re-verified against a second party this session. Trust-center URL is an index page; the PDF was read from the AI Accountability Lab mirror (third-party archive, corroborates the document text, not the claims).",
        "orgs": [
          "Anthropic"
        ],
        "release_date": "2026-07-23",
        "synthetic_data_stated": true,
        "training_data_disclosure": "The training corpus is derived from several publicly accessible repositories, notably including Common Crawl, a repository of web crawl data, as well as specialized datasets available through platforms like GitHub and HuggingFace.",
        "url": "https://trust.anthropic.com/resources",
        "uses_interaction_data": true,
        "uses_other_service_data": true,
        "weights": "closed"
      },
      "id": "model:claude-opus-5",
      "name": "Claude Opus 5",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Anthropic": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "trust.anthropic.com"
        ],
        "source_urls": [
          "https://trust.anthropic.com/resources"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://trust.anthropic.com/resources"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Claude_Sonnet_5_2026_08_03.pdf",
        "crawler_named": [
          "ClaudeBot"
        ],
        "cutoff_date": "2026-05",
        "data_sources_named": [
          "Common Crawl",
          "GitHub (platform)",
          "HuggingFace (platform)",
          "ClaudeBot (crawler)",
          "acquired physical texts (unnamed)"
        ],
        "developer": "Anthropic",
        "eu_training_summary_url": "https://trust.anthropic.com/resources",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Claude Sonnet 5",
        "note": "Release date basis: EU placement date stated in section 1.2. Same Anthropic template text as the Claude Opus 5 summary: 2.2.1 Yes to commercial licensing agreements for Text and Image, no licensor named; 2.2.2 Yes to private third-party datasets, list 'Not applicable'; 2.6 'A portion of the data corpus comes from acquired physical texts.' Synthetic data partly 'generated by Anthropic models not available on the market'. Cutoff wording: 'some data being acquired/collected up to May 2026'. Crawl period March 2024 to May 2026. Answers are typed words, not tick boxes, checked against a rendered page image. Trust-center URL is an index page; text read from the AI Accountability Lab mirror (third-party archive, corroborates the document text, not the claims). Nulls: weights: not established, Hugging Face API author=anthropic returns no models and the summary does not state API-only availability.",
        "orgs": [
          "Anthropic"
        ],
        "release_date": "2026-06-30",
        "synthetic_data_stated": true,
        "training_data_disclosure": "The training corpus is derived from several publicly accessible repositories, notably including Common Crawl, a repository of web crawl data, as well as specialized datasets available through platforms like GitHub and HuggingFace.",
        "url": "https://trust.anthropic.com/resources",
        "uses_interaction_data": true,
        "uses_other_service_data": true
      },
      "id": "model:claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Anthropic": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "trust.anthropic.com"
        ],
        "source_urls": [
          "https://trust.anthropic.com/resources"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://trust.anthropic.com/resources"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Command_A_Plus_2026_08_03.pdf",
        "cutoff_date": "2026-04",
        "data_sources_named": [
          "Common Crawl"
        ],
        "developer": "Cohere",
        "licensed_data_declared": "other",
        "licensed_data_named": false,
        "name": "Command A+ (C4AI Command A Plus)",
        "note": "Release date basis: EU placement date stated in the summary. 2.2.1 ticks 'Other' with comment 'Some of the third parties identified in Section 2.2.2 license their datasets.' 2.2.2 says 'None of the datasets subject to this section 2.2.2 are publicly known.' Checkbox answers read from rendered page images. Registry holds Advance Local Media v. Cohere (publisher suit). weights=open verified via Hugging Face API: CohereLabs/command-a-plus-05-2026-bf16 public, license tag apache-2.0. source_url is the third-party mirror because the provider link expired; eu_training_summary_url null for that reason. Nulls: eu summary url: provider copy is served from a pre-signed S3 URL that expired (X-Amz-Expires=604800 from 2026-08-01); the stable reference is Cohere's docs page and the AIAL mirror",
        "orgs": [
          "Cohere"
        ],
        "release_date": "2026-05-20",
        "status_note": "2.3: 'Were crawlers used by the provider or on behalf of? Yes'. The field 'If yes, specify crawler name(s)/identifier(s):' is answered with a link, not a name: 'Cohere makes information about its web crawlers available at https://docs.cohere.com/docs/cohere-web-crawlers.' Purposes adds 'Prior to August 2025, Cohere collected certain web data using a crawler bot that is no longer in use', again unnamed. 2.1 names Common Crawl.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "The training data for Command A+ includes text from Common Crawl (https://commoncrawl.org/).",
        "update_note": "updated 2026-09-18 from raw.githubusercontent.com: Pointer answer. The registry carries crawler:cohere-ai and crawler:cohere-training-data-crawler, but the summary itself names neither, so crawler_named stays unset. crawler_named deliberately left unset: the summary names no crawler. names a third-party corpus but no crawler. developer URL (none carried) returned HTTP no-url on 2026-09-18, read fr",
        "url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Command_A_Plus_2026_08_03.pdf",
        "uses_interaction_data": true,
        "uses_other_service_data": true,
        "weights": "open"
      },
      "id": "model:command-a-c4ai-command-a-plus",
      "name": "Command A+ (C4AI Command A Plus)",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Cohere": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "raw.githubusercontent.com"
        ],
        "source_urls": [
          "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Command_A_Plus_2026_08_03.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.846,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Command_A_Plus_2026_08_03.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/DeepSeek_DeepSeek_V4_2026_09_08.pdf",
        "cutoff_date": "2025-05",
        "data_sources_named": [
          "Common Crawl",
          "Stack Exchange"
        ],
        "developer": "DeepSeek",
        "eu_training_summary_url": "https://cdn.deepseek.com/policies/DeepSeek_Template_for_the_Public_Summary_of_Training_Content_for_GeneralPurpose_AI_model.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "DeepSeek-V4 (Pro and Flash)",
        "note": "Release date basis: EU placement date stated in the summary. Template section 2.3 (crawled data) is missing, sections are renumbered (2.3 is user data), so crawler_named null although section 3 refers to 'the crawler'. Both user-data questions ticked No, yet the comment says 'If user input is used to construct training data, we apply ... de-identification' and offers an opt-out: internal inconsistency worth flagging. Cutoff 'approximately no later than May 2025'. weights=open verified via Hugging Face API: deepseek-ai/DeepSeek-V4-Pro and -Flash public, license tag mit, created 2026-04-22. Licensed Text Yes, no licensor named.",
        "orgs": [
          "DeepSeek"
        ],
        "release_date": "2026-04-24",
        "status_note": "The crawled-data section of the template is missing from this summary: sections run 2.1 publicly available datasets, 2.2 private datasets, 2.3 User data, 2.4 Synthetic data, 2.5 Other sources. The only crawler reference is in 3.1: 'the crawler is designed to respect robots.txt instructions and other standard web protocols', with no name. 2.1 names 'Common Crawl and Stack Exchange'.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Publicly available internet information, licensed datasets, and diverse textual content including mathematical texts, source code, multilingual materials, long-form documents and text used for agentic and domain-specific training.",
        "update_note": "updated 2026-09-18 from cdn.deepseek.com: DeepSeek deletes the crawler section and then refers to 'the crawler' in the TDM section. No name, no user agent, no crawl period. crawler_named deliberately left unset: the summary names no crawler. names a third-party corpus but no crawler.",
        "url": "https://cdn.deepseek.com/policies/DeepSeek_Template_for_the_Public_Summary_of_Training_Content_for_GeneralPurpose_AI_model.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": false,
        "weights": "open"
      },
      "id": "model:deepseek-v4-pro-and-flash",
      "name": "DeepSeek-V4 (Pro and Flash)",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "DeepSeek": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "cdn.deepseek.com"
        ],
        "source_urls": [
          "https://cdn.deepseek.com/policies/DeepSeek_Template_for_the_Public_Summary_of_Training_Content_for_GeneralPurpose_AI_model.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://cdn.deepseek.com/policies/DeepSeek_Template_for_the_Public_Summary_of_Training_Content_for_GeneralPurpose_AI_model.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Domyn_2026_07_21.pdf",
        "cutoff_date": "2024-09",
        "data_sources_named": [
          "DCML (as written, DCLM-style web corpus)",
          "The Stack v2",
          "HPLT",
          "FineWeb-2",
          "FineWeb2-HQ",
          "Dolma",
          "ProofPile 2 (arXiv)",
          "Wikipedia dumps",
          "Colosseum 355B (synthetic data generator)",
          "Phi 3 Medium (synthetic data generator)",
          "Llama-3_1-Nemotron-Ultra-253B-v1 (synthetic data generator)"
        ],
        "developer": "Domyn",
        "eu_training_summary_url": "https://www.domyn.com/summary-of-training-data/domyn-large",
        "licensed_data_declared": "no",
        "licensed_data_named": false,
        "name": "Domyn Large",
        "note": "Release date basis: EU placement date stated in the summary ('20 March 2026'). Web-page summary, no tick boxes; answers are narrative: 2.2 'We have not used private, non-publicly accessible datasets of third parties.', so licensed_data_declared=no from that sentence. No crawling, no user data. 11.1 trillion text tokens; a modification of Colosseum 355B. Cutoff 'September 2024 (based on pre-training dataset cut-off)', yet FineWeb-2 is dated September 2025. Signatory to the Code of Practice. Nulls: weights: not established, only Domyn-Small-v1.0 is on the domyn Hugging Face account, no Domyn Large repo and no second party states it; crawler_named: the summary says no crawling.",
        "orgs": [
          "Domyn"
        ],
        "release_date": "2026-03-20",
        "status_note": "2.3 'Data crawled and scraped from online sources': 'We have not crawled, scraped, or otherwise directly compiled data from online sources ourselves or through third parties on our behalf'. 3.1 says only 'All open dataset have used web crawlers that honor machine-readable opt-out signals, such as robots.txt and standard metadata', naming no crawler.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "The model was trained on text-only data drawn from publicly available datasets.",
        "update_note": "updated 2026-09-18 from domyn.com: No crawler named; web content arrives through third-party corpora (DCLM-style corpus, The Stack v2, HPLT, FineWeb-2, Dolma). crawler_named deliberately left unset: the summary names no crawler. names a third-party corpus but no crawler.",
        "url": "https://www.domyn.com/summary-of-training-data/domyn-large",
        "uses_interaction_data": false,
        "uses_other_service_data": false
      },
      "id": "model:domyn-large",
      "name": "Domyn Large",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Domyn": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "domyn.com"
        ],
        "source_urls": [
          "https://www.domyn.com/summary-of-training-data/domyn-large"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.846,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://www.domyn.com/summary-of-training-data/domyn-large"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/ElevenLabs_2026_08_03.pdf",
        "developer": "ElevenLabs",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "ElevenLabs model families (TTS, Voice, STT, Music, Non-Speech)",
        "note": "Release date basis: disclosure covers model families across versions; no release date given. Different regime, different shape: AB 2013 disclosure by model family, not per model, no cutoff date or size figures ('may be expressed as ranges or estimates'), so cutoff_date and release_date null. release_date null because the disclosure is family-level. user_data_used null: disclosure says user data is included 'where users have permitted such use' without the template's two yes/no questions. Music models declare licensed recordings but name no licensor, while the registry carries Kobalt Music Group and Merlin deals with ElevenLabs (2025-08-05) plus Amer v. Eleven Labs Inc. provider URL for the disclosure not recorded by AIAL, so source_url is the mirror. weights=closed from service-only availability, not re-verified. Nulls: eu summary url: the document AIAL archived as ElevenLabs is a California AB 2013 (Civ. Code 3110-3111) disclosure effective 2026-01-01, not the EU Article 53(1)(d) template; GPAI Ledger does not list an ElevenLabs EU summary",
        "orgs": [
          "ElevenLabs"
        ],
        "status_note": "This is a California AB 2013 'Training Data Transparency Disclosure' (effective 1 January 2026), not the EU Article 53 template, and it has no crawled-data section at all. The words crawler, crawl and scrape do not appear in the document. Sources are described only as categories: 'Licensed voice recordings', 'Publicly available data, including lawfully accessible speech and text data', 'Proprietary data', 'User-provided data'.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Licensed music recordings; Proprietary datasets, including internally created, commissioned audio, or internally labeled datasets derived from or based on licensed music recordings; and Synthetic audio, generated to supplement training or evaluation.",
        "update_note": "updated 2026-09-18 from raw.githubusercontent.com: Different regime, different shape: no crawler question exists to answer, and no dataset is named. crawler_named deliberately left unset: the summary names no crawler. developer URL (none carried) returned HTTP no-url on 2026-09-18, read from the AI Accountability Lab archive copy https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-",
        "url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/ElevenLabs_2026_08_03.pdf",
        "weights": "closed"
      },
      "id": "model:elevenlabs-model-families-tts-voice-stt-music-non-speech",
      "name": "ElevenLabs model families (TTS, Voice, STT, Music, Non-Speech)",
      "notes": {
        "_org_roles": {
          "ElevenLabs": [
            "developer"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "raw.githubusercontent.com"
        ],
        "source_urls": [
          "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/ElevenLabs_2026_08_03.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.615,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/ElevenLabs_2026_08_03.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Sintesi_2026_07_22.pdf",
        "cutoff_date": "2025-02",
        "data_sources_named": [
          "Common Crawl",
          "Red Pajama",
          "FineWeb Edu",
          "Wikipedia (via Hugging Face)",
          "Mondadori (licensor)",
          "Bignami (licensor)",
          "Istat (licensor)",
          "Phi 3.5 (synthetic data generator)"
        ],
        "developer": "Fastweb",
        "eu_training_summary_url": "https://www.fastweb.it/grandi-aziende/artificial-intelligence/fastweb-miia/documentazione-trasparenza-ai/sintesi%20contenuti%20training.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": true,
        "name": "FastwebMIIA (FastwebMIIA-7B, FastwebMIIA-7B-2603)",
        "note": "Release date basis: 'Data di rilascio' stated in the summary, 29/05/2025 for FastwebMIIA-7B and 23/03/2026 for -2603; no EU placement field. Italian narrative adaptation (dated 23.03.2026) with no tick boxes; licensed_data_declared=yes from the text: 'Fastweb ha sottoscritto accordi di licenza con partner quali Mondadori, Bignami, Istat.' Names licensors, with 'ad esempio' implying the list is partial. Text only, about 85% Common Crawl, synthetic under 1% via Phi 3.5, no user data. Cutoff basis: stated crawl period 'da marzo 2024 a febbraio 2025'. weights=open verified via Hugging Face API: Fastweb/FastwebMIIA-7B public, gated=auto, license tag 'other'; the -2603 version is not a separate repo (repo last modified 2026-04-10). Nulls: crawler_named: section U lists Common Crawl, Red Pajama and FineWeb Edu, which are third-party datasets, not a provider crawler.",
        "orgs": [
          "Fastweb"
        ],
        "release_date": "2025-05-29",
        "status_note": "Row U of the Italian narrative adaptation, 'Identificazione dei crawler, loro scopo e comportamento', is answered with three dataset names, not crawlers: 'I principali Crawler utilizzati sono: 1) Common Crawl 2) Red Pajama 3) FineWeb Edu'. Row W repeats the same three names as the most relevant internet domains. Row T states 'Circa l'85% del dataset e tratto da Common Crawl' with a collection period 'da marzo 2024 a febbraio 2025'.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Circa l’85% del dataset è tratto da Common Crawl (CC), un archivio realizzato attraverso scraping non indiscriminato sul web",
        "update_note": "updated 2026-09-18 from fastweb.it: The crawler-identification question is answered with corpora. Common Crawl's user agent is CCBot but the summary never says CCBot, so crawler_named stays unset. crawler_named deliberately left unset: the summary names no crawler. names a third-party corpus but no crawler.",
        "url": "https://www.fastweb.it/grandi-aziende/artificial-intelligence/fastweb-miia/documentazione-trasparenza-ai/sintesi%20contenuti%20training.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": false,
        "weights": "open"
      },
      "id": "model:fastwebmiia-fastwebmiia-7b-fastwebmiia-7b-2603",
      "name": "FastwebMIIA (FastwebMIIA-7B, FastwebMIIA-7B-2603)",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Fastweb": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "fastweb.it"
        ],
        "source_urls": [
          "https://www.fastweb.it/grandi-aziende/artificial-intelligence/fastweb-miia/documentazione-trasparenza-ai/sintesi%20contenuti%20training.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://www.fastweb.it/grandi-aziende/artificial-intelligence/fastweb-miia/documentazione-trasparenza-ai/sintesi%20contenuti%20training.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/FIBO_2026_08_03.pdf",
        "cutoff_date": "2025-06",
        "developer": "Bria AI",
        "eu_training_summary_url": "https://drive.google.com/file/d/1z3JFPBQCRKdj5F-Qfgp4TaNITWcMZtOi/view",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "FIBO (FIBO, FIBO Lite, FIBO Edit)",
        "note": "Release date basis: EU placement date stated in the summary ('FIBO on or about 29 October 2025'; Lite about 2025-12-01, Edit about 2026-01-16). Narrative Yes/No answers, no tick boxes. 479 million licensed images, 2.1 publicly available datasets No, 2.2.1 Yes (Image), 2.2.2 No, crawlers No, user data No/No, synthetic No (third-party AI captions disclosed as enrichment). No data partner named. Discloses frozen third-party components Wan2.2-VAE (Alibaba) and SmolLM3-3B text encoder whose training data 'lies outside Bria's content licensing programme'. The archived PDF also carries a Bria 3.2 summary v2.0 (last update 02/08/2026) that supersedes the version behind the carried Bria 3.2 row and links huggingface.co/briaai/BRIA-3.2. weights=open verified via Hugging Face API: briaai/FIBO public, gated=auto (click-through), license tag 'other'. Nulls: crawler_named: Bria states it does not crawl.",
        "orgs": [
          "Bria AI"
        ],
        "release_date": "2025-10-29",
        "status_note": "2.3 'Data crawled and scraped from online sources': 'Were crawlers used by the provider or on their behalf? No. Bria does not engage in web crawling or scraping. Accordingly no crawler identifiers, collection period or domain name list falls to be disclosed.' The annex 'No Web-Crawling Policy' repeats it. The only corpus in the document is disclosed for an embedded third-party component, not for FIBO: 'T5 version 1.1 was pretrained on the Colossal Clean Crawled Corpus (C4), a dataset derived from Common Crawl web data'.",
        "synthetic_data_stated": false,
        "training_data_disclosure": "Fully licensed images provided by Bria data partners under transactional commercial licensing agreements with rightsholders globally.",
        "update_note": "updated 2026-09-18 from drive.google.com: Notable: the C4 / Common Crawl mention belongs to the frozen T5 text encoder Bria embeds, whose provider 'has not published a summary of training content'. It is not FIBO's own training data. crawler_named deliberately left unset: the summary names no crawler.",
        "url": "https://drive.google.com/file/d/1z3JFPBQCRKdj5F-Qfgp4TaNITWcMZtOi/view",
        "uses_interaction_data": false,
        "uses_other_service_data": false,
        "weights": "open"
      },
      "id": "model:fibo-fibo-fibo-lite-fibo-edit",
      "name": "FIBO (FIBO, FIBO Lite, FIBO Edit)",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Bria AI": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "drive.google.com"
        ],
        "source_urls": [
          "https://drive.google.com/file/d/1z3JFPBQCRKdj5F-Qfgp4TaNITWcMZtOi/view"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.846,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://drive.google.com/file/d/1z3JFPBQCRKdj5F-Qfgp4TaNITWcMZtOi/view"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/FLUX_3_2026_08_03.pdf",
        "cutoff_date": "2026-06",
        "data_sources_named": [
          "Egocentric-100K (Build AI, Hugging Face, Apache 2.0)",
          "FLUX.1 and FLUX.2 (synthetic data generators)"
        ],
        "developer": "Black Forest Labs",
        "eu_training_summary_url": "https://bfl.ai/transparency?tab=training-data-summaries",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "FLUX 3",
        "note": "Release date basis: EU placement date stated in the summary. Licensed: 'Yes , BFL has entered into data access agreements' for Image, Video, Audio, unnamed. No crawlers. User data from API customers used 'Subject to user opt-out'. Of direct Blomega relevance: an image/video generator names an egocentric human-activity dataset (Egocentric-100K, action prediction) as training data, a signal that first-person capture data is feeding generative models, not only robotics. weights null: FLUX 3 open-weight status not checked this session (earlier FLUX [dev] variants were open, not assumed here). Nulls: weights: not established, the model is not published on Hugging Face under the developer's account and no second party states it",
        "orgs": [
          "Black Forest Labs"
        ],
        "release_date": "2026-07-16",
        "status_note": "2.3 'Data crawled and scraped from online sources': 'Were crawlers used by the provider or on behalf of? [x] No'. No crawler sub-fields are printed. The only 'common crawl' strings in the document are the template's own boilerplate ('platforms such as common crawl that are covered under Section 2.1') and a generic reference to 'snapshots of common crawl' in the publicly-available-datasets prose.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Training data includes captioning, images, videos, action prediction, text-image and text-video pairs from open source and publicly available scientific research, technical and educational repositories, and specialised collections, for example, Egocentric-100K made available by Build AI",
        "update_note": "updated 2026-09-18 from bfl.ai: No crawler named and no crawling declared; the one dataset named is Egocentric-100K (Build AI, Apache 2.0). crawler_named deliberately left unset: the summary names no crawler. names a third-party corpus but no crawler.",
        "url": "https://bfl.ai/transparency?tab=training-data-summaries",
        "uses_interaction_data": true,
        "uses_other_service_data": true
      },
      "id": "model:flux-3",
      "name": "FLUX 3",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Black Forest Labs": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "bfl.ai"
        ],
        "source_urls": [
          "https://bfl.ai/transparency?tab=training-data-summaries"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.846,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://bfl.ai/transparency?tab=training-data-summaries"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Gemini_2026_07_21.pdf",
        "crawler_named": [
          "Google-Extended"
        ],
        "cutoff_date": "2025-01",
        "developer": "Google",
        "eu_training_summary_url": "https://storage.googleapis.com/transparencyreport/report-downloads/pdf-report-ii_2026-7-2_2026-7-2_en_v1.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Gemini 3 Pro (family)",
        "note": "Release date basis: EU placement month stated in the summary ('November 2025'). Checkbox state is not in the PDF text layer; Yes/No answers were read from rendered page images (2.2.1 Yes for Text, Image, Video, Audio; 2.2.2 Yes; user data Yes/Yes; synthetic Yes). No dataset, licensor or crawler is named: crawler field says 'See list of crawlers here.' so crawler_named is null. Covers the family incl. Gemini 3 Flash, 3.1 Pro, 3.5 Flash. Registry holds Reddit, Associated Press, Stack Overflow and Shutterstock deals with Google, none named. Cutoff stated as 'knowledge cut-off date ... is 01 / 2025'. weights=closed from API-only availability, not re-verified this session.",
        "orgs": [
          "Google"
        ],
        "release_date": "2025-11",
        "status_note": "2.3: 'Were crawlers used by the provider or on behalf of? Yes'. The field 'If yes, specify crawler name(s)/identifier(s)' is answered with a bare pointer, 'See list of crawlers here.', and the domains field repeats 'See list of our common crawlers here. Google's common crawlers obey robots.txt rules'. The only crawler string actually printed in the document is in 3.1: 'our Google-Extended control lets web publishers manage whether content Google crawls from their sites may be used for training Gemini models that power Gemini Apps and Gemini Enterprise Agent Platform'.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "The pre-training dataset was a large-scale, diverse collection of data encompassing a wide range of domains and modalities, which included publicly-available web-documents, code, images, audio (including speech and other audio types), and video.",
        "update_note": "updated 2026-09-18 from storage.googleapis.com: Judgement call recorded openly: 2.3 names nothing, so the value comes from 3.1, where Google names its own crawl-control token for Gemini training data. Matches crawler:google-extended. Google-Extended matches crawler:google-extended.",
        "url": "https://storage.googleapis.com/transparencyreport/report-downloads/pdf-report-ii_2026-7-2_2026-7-2_en_v1.pdf",
        "uses_interaction_data": true,
        "uses_other_service_data": true,
        "weights": "closed"
      },
      "id": "model:gemini-3-pro-family",
      "name": "Gemini 3 Pro (family)",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Google": [
            "developer"
          ]
        },
        "_release_date_precision": "month"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "storage.googleapis.com"
        ],
        "source_urls": [
          "https://storage.googleapis.com/transparencyreport/report-downloads/pdf-report-ii_2026-7-2_2026-7-2_en_v1.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://storage.googleapis.com/transparencyreport/report-downloads/pdf-report-ii_2026-7-2_2026-7-2_en_v1.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Gemma_4_2028_08_03.pdf",
        "crawler_named": [
          "Google-Extended"
        ],
        "cutoff_date": "2025-01",
        "developer": "Google",
        "eu_training_summary_url": "https://storage.googleapis.com/transparencyreport/report-downloads/pdf-report-nn_2026-7-31_2026-7-31_en_v1.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Gemma 4 (family)",
        "note": "Release date basis: EU placement month stated in the summary ('April 2026'). Open-weight model whose summary is near-identical boilerplate to the closed Gemini 3 Pro summary, including Yes to licensed data and Yes to user data from Google products; no dataset named. weights=open verified via Hugging Face API: google/gemma-4-31B public, license tag apache-2.0. Checkbox answers read from rendered page images. AIAL archive file is misdated '2028'.",
        "orgs": [
          "Google"
        ],
        "release_date": "2026-04",
        "status_note": "2.3: crawlers Yes; 'If yes, specify crawler name(s)/identifier(s): See list of crawlers here.' The Gemma 4 text is word-for-word the Gemini 3 Pro text. The only crawler string printed is in 3.1: 'our Google-Extended control lets web publishers manage whether content Google crawls from their sites may be used for training Gemini models'.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Our publicly available datasets include data across various sectors such as educational, government, legal, and research sectors comprising a wide variety of media types and languages.",
        "update_note": "updated 2026-09-18 from storage.googleapis.com: Same judgement as Gemini 3 Pro. Note the open-weight Gemma summary reuses the Gemini boilerplate verbatim, including the Gemini-specific Google-Extended sentence. Matches crawler:google-extended. Google-Extended matches crawler:google-extended.",
        "url": "https://storage.googleapis.com/transparencyreport/report-downloads/pdf-report-nn_2026-7-31_2026-7-31_en_v1.pdf",
        "uses_interaction_data": true,
        "uses_other_service_data": true,
        "weights": "open"
      },
      "id": "model:gemma-4-family",
      "name": "Gemma 4 (family)",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Google": [
            "developer"
          ]
        },
        "_release_date_precision": "month"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "storage.googleapis.com"
        ],
        "source_urls": [
          "https://storage.googleapis.com/transparencyreport/report-downloads/pdf-report-nn_2026-7-31_2026-7-31_en_v1.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://storage.googleapis.com/transparencyreport/report-downloads/pdf-report-nn_2026-7-31_2026-7-31_en_v1.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/GPT_5_2_2026_08_17.pdf",
        "crawler_named": [
          "GPTBot"
        ],
        "cutoff_date": "2025-08",
        "data_sources_named": [
          "Common Crawl",
          "GPTBot (crawler)",
          "GPT-5 (synthetic data generator)",
          "o-series models (synthetic data generator)",
          "GPT-4 family models (synthetic data generator)"
        ],
        "developer": "OpenAI",
        "eu_training_summary_url": "https://cdn.openai.com/pdf/gpt-5-2-eu-ai-act-public-summary-of-training-content.pdf",
        "licensed_data_declared": "other",
        "licensed_data_named": false,
        "name": "GPT-5.2",
        "note": "Release date basis: EU placement date stated in section 1.2 ('11 December 2025'). Summary v1 last updated 30 July 2026. Modalities Text, Image, Audio, Video. 2.2.1 ticks neither Yes nor No but 'Other (see below)': 'OpenAI enters into broad partnerships with third parties that may include ... access to non-publicly available content, such as archives and metadata.' No partner named. 2.2.2 Yes to private third-party datasets, list N/A. 2.4: user interactions with the model No, other services (ChatGPT, Codex) Yes. Cutoff wording: 'some data collected no later than August 2025'. Crawl period 'Approximately 2018 - August 2025'. Tick boxes are glyphs in the text layer (confirmed against rendered pages in this template). Nulls: weights: not established, Hugging Face API author=openai lists no GPT-5.2 repo and the summary does not state API-only availability.",
        "orgs": [
          "OpenAI"
        ],
        "release_date": "2025-12-11",
        "synthetic_data_stated": true,
        "training_data_disclosure": "GPT-5.2 was trained on a large-scale, multilingual mixture of publicly available data, data accessed through partnerships, synthetic data, and human-generated text",
        "url": "https://cdn.openai.com/pdf/gpt-5-2-eu-ai-act-public-summary-of-training-content.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": true
      },
      "id": "model:gpt-5-2",
      "name": "GPT-5.2",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "OpenAI": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "cdn.openai.com"
        ],
        "source_urls": [
          "https://cdn.openai.com/pdf/gpt-5-2-eu-ai-act-public-summary-of-training-content.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://cdn.openai.com/pdf/gpt-5-2-eu-ai-act-public-summary-of-training-content.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/GPT_5_4_Nano_2026_08_17.pdf",
        "crawler_named": [
          "GPTBot"
        ],
        "cutoff_date": "2025-08",
        "data_sources_named": [
          "Common Crawl",
          "GPTBot (crawler)",
          "GPT-5.2 (synthetic data generator)"
        ],
        "developer": "OpenAI",
        "eu_training_summary_url": "https://cdn.openai.com/pdf/gpt-5-4-nano-eu-ai-act-public-summary-of-training-content.pdf",
        "licensed_data_declared": "other",
        "licensed_data_named": false,
        "name": "GPT-5.4 nano",
        "note": "Release date basis: EU placement date stated in section 1.2 ('12 March 2026'). Summary last updated 31 July 2026. Audio and video each ticked 'Less than 10 000 hours'; public datasets and 2.2.1 modalities Text and Image only. 2.2.1 ticks neither Yes nor No but 'Other (see below)': 'OpenAI enters into broad partnerships with third parties that may include ... access to non-publicly available content, such as archives and metadata.' No partner named. 2.2.2 Yes to private third-party datasets, list N/A. 2.4: user interactions with the model No, other services (ChatGPT, Codex) Yes. Cutoff wording: 'some data collected no later than August 2025'. Crawl period 'Approximately 2018 - August 2025'. Tick boxes are glyphs in the text layer (confirmed against rendered pages in this template). Nulls: weights: not established, Hugging Face API author=openai lists no GPT-5.4 nano repo and the summary does not state API-only availability.",
        "orgs": [
          "OpenAI"
        ],
        "release_date": "2026-03-12",
        "synthetic_data_stated": true,
        "training_data_disclosure": "GPT-5.4 nano was trained on a large-scale, multilingual mixture of publicly available data, data accessed through partnerships, synthetic data, and human-generated text",
        "url": "https://cdn.openai.com/pdf/gpt-5-4-nano-eu-ai-act-public-summary-of-training-content.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": true
      },
      "id": "model:gpt-5-4-nano",
      "name": "GPT-5.4 nano",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "OpenAI": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "cdn.openai.com"
        ],
        "source_urls": [
          "https://cdn.openai.com/pdf/gpt-5-4-nano-eu-ai-act-public-summary-of-training-content.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://cdn.openai.com/pdf/gpt-5-4-nano-eu-ai-act-public-summary-of-training-content.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/GPT_5_5_2026_07_14.pdf",
        "crawler_named": [
          "GPTBot"
        ],
        "cutoff_date": "2026-02",
        "data_sources_named": [
          "Common Crawl",
          "GPTBot (crawler)",
          "GPT-5.4 (synthetic data generator)"
        ],
        "developer": "OpenAI",
        "eu_training_summary_url": "https://cdn.openai.com/pdf/eb1f5ac3-009e-4d51-b2da-f5f327913115/gpt-5-5-eu-ai-act-public-summary-of-training-content.pdf",
        "licensed_data_declared": "other",
        "licensed_data_named": false,
        "name": "GPT-5.5",
        "note": "Release date basis: EU placement date stated in the summary ('23 April 2026'); AIAL eval metadata lists model_publication_date 2025-08-07, which conflicts and looks copied from GPT-5. Section 2.2.1 ticks neither Yes nor No but 'Other (see below)': 'OpenAI enters into broad partnerships with third parties that may include ... access to non-publicly available content, such as archives and metadata. OpenAI does not pursue partnerships solely for access to publicly available data.' No partner is named, although the registry carries many announced OpenAI content deals (see registry_links). Crawl period stated 'Approximately 2018 - December 2025'. weights=closed from API-only availability, not re-verified this session.",
        "orgs": [
          "OpenAI"
        ],
        "release_date": "2026-04-23",
        "synthetic_data_stated": true,
        "training_data_disclosure": "GPT-5.5 was trained on a large-scale, multilingual mixture of publicly available data, data accessed through partnerships, synthetic data, and human-generated text",
        "url": "https://cdn.openai.com/pdf/eb1f5ac3-009e-4d51-b2da-f5f327913115/gpt-5-5-eu-ai-act-public-summary-of-training-content.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": true,
        "weights": "closed"
      },
      "id": "model:gpt-5-5",
      "name": "GPT-5.5",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "OpenAI": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "cdn.openai.com"
        ],
        "source_urls": [
          "https://cdn.openai.com/pdf/eb1f5ac3-009e-4d51-b2da-f5f327913115/gpt-5-5-eu-ai-act-public-summary-of-training-content.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://cdn.openai.com/pdf/eb1f5ac3-009e-4d51-b2da-f5f327913115/gpt-5-5-eu-ai-act-public-summary-of-training-content.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/GPT_5_6_Luna_2026_08_03.pdf",
        "crawler_named": [
          "GPTBot"
        ],
        "cutoff_date": "2026-06",
        "data_sources_named": [
          "Common Crawl",
          "GPTBot (crawler)",
          "GPT-5.4 (synthetic data generator)",
          "GPT-5.5 (synthetic data generator)"
        ],
        "developer": "OpenAI",
        "eu_training_summary_url": "https://cdn.openai.com/pdf/gpt-5-6-luna-eu-ai-act-public-summary-of-training-content.pdf",
        "licensed_data_declared": "other",
        "licensed_data_named": false,
        "name": "GPT-5.6 Luna",
        "note": "Release date basis: EU placement date stated in section 1.2 ('9 July 2026'). Summary v1 last updated 23 July 2026. 2.6 Yes: 'we worked with experienced professionals to create data representing real-world knowledge work'. 2.2.1 ticks neither Yes nor No but 'Other (see below)': 'OpenAI enters into broad partnerships with third parties that may include ... access to non-publicly available content, such as archives and metadata.' No partner named. 2.2.2 Yes to private third-party datasets, list N/A. 2.4: user interactions with the model No, other services (ChatGPT, Codex) Yes. Cutoff wording: 'some data collected no later than June 2026'. Crawl period 'Approximately 2018 - February 2026'. Tick boxes are glyphs in the text layer (confirmed against rendered pages in this template). Nulls: weights: not established, Hugging Face API author=openai lists no GPT-5.6 Luna repo and the summary does not state API-only availability.",
        "orgs": [
          "OpenAI"
        ],
        "release_date": "2026-07-09",
        "synthetic_data_stated": true,
        "training_data_disclosure": "GPT-5.6 Luna was trained on a large-scale, multilingual mixture of publicly available data, data accessed through partnerships, synthetic data, and human-generated text",
        "url": "https://cdn.openai.com/pdf/gpt-5-6-luna-eu-ai-act-public-summary-of-training-content.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": true
      },
      "id": "model:gpt-5-6-luna",
      "name": "GPT-5.6 Luna",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "OpenAI": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "cdn.openai.com"
        ],
        "source_urls": [
          "https://cdn.openai.com/pdf/gpt-5-6-luna-eu-ai-act-public-summary-of-training-content.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://cdn.openai.com/pdf/gpt-5-6-luna-eu-ai-act-public-summary-of-training-content.pdf"
    },
    {
      "fields": {
        "crawler_named": [
          "GPTBot"
        ],
        "cutoff_date": "2026-08-06",
        "data_sources_named": [
          "Common Crawl",
          "GPTBot (crawler)",
          "GPT-5.6 (synthetic data generator)",
          "GPT-5.5 (synthetic data generator)",
          "GPT-5.4 (synthetic data generator)"
        ],
        "developer": "OpenAI",
        "eu_training_summary_url": "https://cdn.openai.com/pdf/gpt-6-astra-eu-ai-act-public-summary-of-training-content.pdf",
        "licensed_data_declared": "other",
        "licensed_data_named": false,
        "name": "GPT-6 Astra",
        "note": "Release date basis: EU placement date stated in section 1.2 ('3 September 2026'). Read directly from the provider PDF (HTTP 200, 7 pages) because the AIAL mirror has no archive copy; AIAL eval metadata lists model_publication_date 2025-09-03, which conflicts with the summary. Summary last updated 2 September 2026. Cutoff stated to the day, 6 August 2026. 2.2.1 ticks neither Yes nor No but 'Other (see below)': 'OpenAI enters into broad partnerships with third parties that may include ... access to non-publicly available content, such as archives and metadata.' No partner named. 2.2.2 Yes to private third-party datasets, list N/A. 2.4: user interactions with the model No, other services (ChatGPT, Codex) Yes. Cutoff wording: 'some data collected no later than 6 August 2026'. Crawl period 'Approximately March 2013 - May 2026'. Tick boxes are glyphs in the text layer (confirmed against rendered pages in this template). Nulls: archive_url: no AIAL mirror copy exists for this summary as of 2026-09-17; weights: not established, Hugging Face API author=openai lists no GPT-6 Astra repo and the summary does not state API-only availability.",
        "orgs": [
          "OpenAI"
        ],
        "release_date": "2026-09-03",
        "synthetic_data_stated": true,
        "training_data_disclosure": "GPT-6 Astra was trained on a large-scale, multilingual mixture of publicly available data, data accessed through partnerships, synthetic data, and human-generated text",
        "url": "https://cdn.openai.com/pdf/gpt-6-astra-eu-ai-act-public-summary-of-training-content.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": true
      },
      "id": "model:gpt-6-astra",
      "name": "GPT-6 Astra",
      "notes": {
        "_cutoff_date_precision": "day",
        "_org_roles": {
          "OpenAI": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "cdn.openai.com"
        ],
        "source_urls": [
          "https://cdn.openai.com/pdf/gpt-6-astra-eu-ai-act-public-summary-of-training-content.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://cdn.openai.com/pdf/gpt-6-astra-eu-ai-act-public-summary-of-training-content.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Phi_MoE_2026_09_11.pdf",
        "cutoff_date": "2024-06",
        "developer": "Microsoft",
        "eu_training_summary_url": "https://huggingface.co/microsoft/Phi-tiny-MoE-instruct/blob/main/data_summary_card.md",
        "licensed_data_declared": "no",
        "licensed_data_named": false,
        "name": "GRIN-MoE and Phi MoE (tiny, mini) instruct",
        "note": "Release date basis: summary states '18-Sept-2024' for GRIN MoE 16x3.8B (HF microsoft/GRIN-MoE created 2024-09-10); the card title also covers phi-tiny-MoE-instruct and phi-mini-MoE-instruct, released later (AIAL lists 2025-06-23). No dataset named. Retroactive summary, last update 10-Dec-2025. Format: a 3-page Microsoft 'Data Summary' card that answers template questions in plain text (no tick boxes), so Yes/No values are the card's written answers. The card has no crawler, user-data or other-service questions, so crawler_named, uses_interaction_data and uses_other_service_data are null: not asked in this format. 2.2.1.A No, 2.2.2.A No, synthetic Yes. Latest acquisition 03-Jun-2024. weights=open verified via Hugging Face API: microsoft/GRIN-MoE, Phi-tiny-MoE-instruct, Phi-mini-MoE-instruct public, ungated, license tag mit.",
        "orgs": [
          "Microsoft"
        ],
        "release_date": "2024-09-18",
        "status_note": "Microsoft 'Data Summary' card, three pages. Its section 2.3 is 'Personal Information', not crawled data: the card carries no crawled-data question at all, and the words crawler, crawl and scrape do not appear. Sources are given only as '1.3.1.B Text training data content: Our training data includes a wide variety of sources and is a combination of publicly available documents selected for quality, educational data, and code'.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Our training data includes a wide variety of sources and is a combination of publicly available documents selected for quality, educational data, and code",
        "update_note": "updated 2026-09-18 from huggingface.co: Microsoft's card format simply omits the EU template's crawled-data section, so there is nothing to name. crawler_named deliberately left unset: the summary names no crawler.",
        "url": "https://huggingface.co/microsoft/Phi-tiny-MoE-instruct/blob/main/data_summary_card.md",
        "weights": "open"
      },
      "id": "model:grin-moe-and-phi-moe-tiny-mini-instruct",
      "name": "GRIN-MoE and Phi MoE (tiny, mini) instruct",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Microsoft": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "huggingface.co"
        ],
        "source_urls": [
          "https://huggingface.co/microsoft/Phi-tiny-MoE-instruct/blob/main/data_summary_card.md"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.846,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://huggingface.co/microsoft/Phi-tiny-MoE-instruct/blob/main/data_summary_card.md"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Grok_4_5_2026_07_17.pdf",
        "crawler_named": [
          "xAI Web Crawler"
        ],
        "cutoff_date": "2026-06",
        "data_sources_named": [
          "xAI Web Crawler (crawler)"
        ],
        "developer": "xAI",
        "eu_training_summary_url": "https://media.x.ai/v1/website/public-summary-of-training-content-for-grok-4.5_8jul2026.docx-fc25014a.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Grok 4.5",
        "note": "Release date basis: EU placement date stated in the summary. weights null: Hugging Face API for xai-org/grok-4.5 returned 401 (no public repo found) but absence of that guess is not proof of closed weights. Summary says Yes to licensed Text and Image with no licensor named, and 'The model is also continuously refined on new data after this date.' Crawl period 01/2024 to 06/2026. 'social media posts' named as a content type but X is not named as a source. No xAI deals or suits in the registry. Nulls: weights: not established, the model is not published on Hugging Face under the developer's account and no second party states it",
        "orgs": [
          "xAI"
        ],
        "release_date": "2026-07-14",
        "synthetic_data_stated": true,
        "training_data_disclosure": "The primary source of Grok 4.5's training is data consisting of text, images, audio content and audiovisual content from publicly available sources from the Internet.",
        "url": "https://media.x.ai/v1/website/public-summary-of-training-content-for-grok-4.5_8jul2026.docx-fc25014a.pdf",
        "uses_interaction_data": true,
        "uses_other_service_data": true
      },
      "id": "model:grok-4-5",
      "name": "Grok 4.5",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "xAI": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "media.x.ai"
        ],
        "source_urls": [
          "https://media.x.ai/v1/website/public-summary-of-training-content-for-grok-4.5_8jul2026.docx-fc25014a.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://media.x.ai/v1/website/public-summary-of-training-content-for-grok-4.5_8jul2026.docx-fc25014a.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Grok_Voice_Think_Fast_2_0_2026_07_17.pdf",
        "crawler_named": [
          "xAI Web Crawler"
        ],
        "cutoff_date": "2026-06",
        "data_sources_named": [
          "Common Voice (Mozilla)",
          "LibriSpeech",
          "VoxPopuli",
          "FLEURS",
          "X social media posts and videos",
          "xAI Web Crawler (crawler)",
          "Grok 4.3 (synthetic data generator)",
          "Grok 4.5 (synthetic data generator)"
        ],
        "developer": "xAI",
        "eu_training_summary_url": "https://media.x.ai/v1/website/public-summary-of-training-content-grok-voice-think-fast-2.0_29jul2026-0fb8805e.pdf",
        "licensed_data_declared": "no",
        "licensed_data_named": false,
        "name": "Grok Voice Think Fast 2.0",
        "note": "Release date basis: EU placement date stated in section 1.2 (29 July 2026). Unlike Grok 4.5 (Yes), 2.2.1 ticks No to commercial licensing agreements; 2.2.2 ticks Yes for Audio only: 'conversations on various different topics, recorded in clean and high quality environments', none named. Names four public speech datasets (Common Voice, LibriSpeech, VoxPopuli from 2009-2020 European Parliament recordings, FLEURS). 'X social media posts' and 'X social media videos' are described as publicly available content, while 2.4 other-services user data is ticked No. Crawl period January 2024 to June 2026. 3.1 ticks No to Code of Practice signatory. Tick boxes are glyphs in the text layer (confirmed on a rendered page). AIAL eval YAML for this file carries Muse Glimmer metadata by copy error. Nulls: weights: not established, Hugging Face API author=xai-org lists only grok-1 and grok-2 and the summary does not state API-only availability.",
        "orgs": [
          "xAI"
        ],
        "release_date": "2026-07-29",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Grok Voice Think Fast 2.0 was trained using a carefully designated data recipe that incorporates a diverse corpus of publicly available audio data, including, for example, public speech, media, voice application recordings, noise, audio communications, certain clips and audio from social media videos.",
        "url": "https://media.x.ai/v1/website/public-summary-of-training-content-grok-voice-think-fast-2.0_29jul2026-0fb8805e.pdf",
        "uses_interaction_data": true,
        "uses_other_service_data": false
      },
      "id": "model:grok-voice-think-fast-2-0",
      "name": "Grok Voice Think Fast 2.0",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "xAI": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "media.x.ai"
        ],
        "source_urls": [
          "https://media.x.ai/v1/website/public-summary-of-training-content-grok-voice-think-fast-2.0_29jul2026-0fb8805e.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://media.x.ai/v1/website/public-summary-of-training-content-grok-voice-think-fast-2.0_29jul2026-0fb8805e.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Tencent_Hy3_2026_09_08.pdf",
        "crawler_named": [
          "Sogou Web Spider",
          "Sogou News Spider"
        ],
        "cutoff_date": "2026-02",
        "data_sources_named": [
          "Common Crawl",
          "Sogou Web Spider (crawler)",
          "Sogou News Spider (crawler)",
          "Hy2 (synthetic data generator)"
        ],
        "developer": "Tencent",
        "eu_training_summary_url": "https://hy.tencent.ai/legal/Hy3-Training-Data-Summary-20260820.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Hy3",
        "note": "Release date basis: EU placement date stated in the summary ('July 6, 2026'); summary V2, last update August 20, 2026. Provider entity OriGen Tech Pte. Ltd. (Singapore), EU representative Tencent International Service Europe B.V. 2.2.1 Yes for Text, no licensor named. 2.2.2 Yes: third-party 'textbooks, problem sets', predominantly English and Chinese, unnamed. The training crawlers are the Sogou search-index spiders, i.e. one crawler for search and training. Knowledge cutoff 'is 28 February 2026'; latest collection 'no later than March 2026'. Both user-data answers No. Not a Code of Practice signatory. Checkbox answers read from text-layer glyphs. weights=open verified via Hugging Face API: tencent/Hy3 public, license tag apache-2.0. No Tencent deals or suits in the registry.",
        "orgs": [
          "Tencent"
        ],
        "release_date": "2026-07-06",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Hy3 was trained on a large-scale, multilingual mixture of publicly available data, data accessed through partnerships, synthetic data, and human-generated text",
        "url": "https://hy.tencent.ai/legal/Hy3-Training-Data-Summary-20260820.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": false,
        "weights": "open"
      },
      "id": "model:hy3",
      "name": "Hy3",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Tencent": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "hy.tencent.ai"
        ],
        "source_urls": [
          "https://hy.tencent.ai/legal/Hy3-Training-Data-Summary-20260820.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://hy.tencent.ai/legal/Hy3-Training-Data-Summary-20260820.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Inkling_2026_07_17.pdf",
        "cutoff_date": "2026-07",
        "data_sources_named": [
          "Common Crawl"
        ],
        "developer": "Thinking Machines Lab",
        "eu_training_summary_url": "https://thinkingmachines.ai/documents/inkling-public-summary-training-content.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Inkling",
        "note": "Release date basis: EU placement date stated in section 1.2. 2.2.1 Yes to commercial licensing agreements for Text and Image, no licensor named; 2.2.2 Yes for all four modalities: 'acquired text, image, audio, and video content from various third parties ... includes items in the public domain as well as content that may be subject to intellectual property protection', list N/A. Both user-data questions No. Crawl period 'From 2025 to 2026'. 3.1 No to Code of Practice signatory. Cutoff: 'latest date of data collection ... is July 2026', collected 'on an ongoing basis'. Tick boxes read from text layer and confirmed on rendered page 3. weights=open verified via Hugging Face API: thinkingmachines/Inkling public, not gated, license tag apache-2.0. Nulls: crawler_named: crawlers ticked Yes but the name field reads 'N/A'.",
        "orgs": [
          "Thinking Machines Lab"
        ],
        "release_date": "2026-07-15",
        "status_note": "2.3: 'Were crawlers used by the provider or on behalf of? [x] Yes'. 'If yes, specify crawler name(s)/identifier(s): N/A'. Purposes: 'Crawlers were used to download content from publicly available sources from the internet for the purpose of model training'. Behaviour: 'Thinking Machines Lab's policy is that crawlers should not circumvent captchas, password-protections, or other access controls, and respect robots.txt'. Period 'From 2025 to 2026'. 2.1 names 'Common Crawl (https://commoncrawl.org/)'.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Inkling was trained on a mixture of publicly available content, content acquired through partnerships, synthetic content, and generated content, including general web content, reference materials, technical documentation, source code, and other text, curated and filtered.",
        "update_note": "updated 2026-09-18 from thinkingmachines.ai: Crawlers admitted, name declined with 'N/A'. Only the corpus is named. crawler_named deliberately left unset: the summary names no crawler. names a third-party corpus but no crawler.",
        "url": "https://thinkingmachines.ai/documents/inkling-public-summary-training-content.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": false,
        "weights": "open"
      },
      "id": "model:inkling",
      "name": "Inkling",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Thinking Machines Lab": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "thinkingmachines.ai"
        ],
        "source_urls": [
          "https://thinkingmachines.ai/documents/inkling-public-summary-training-content.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://thinkingmachines.ai/documents/inkling-public-summary-training-content.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Inkling_Small_2026_08_03.pdf",
        "cutoff_date": "2026-07",
        "data_sources_named": [
          "Common Crawl"
        ],
        "developer": "Thinking Machines Lab",
        "eu_training_summary_url": "https://thinkingmachines.ai/documents/inkling-small-public-summary-training-content.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Inkling-Small",
        "note": "Release date basis: EU placement date stated in section 1.2. Text identical to the Inkling summary apart from the model name. 2.2.1 Yes to commercial licensing agreements for Text and Image, no licensor named; 2.2.2 Yes for all four modalities: 'acquired text, image, audio, and video content from various third parties ... includes items in the public domain as well as content that may be subject to intellectual property protection', list N/A. Both user-data questions No. Crawl period 'From 2025 to 2026'. 3.1 No to Code of Practice signatory. Cutoff: 'latest date of data collection ... is July 2026', collected 'on an ongoing basis'. Tick boxes read from text layer and confirmed on rendered page 3. weights=open verified via Hugging Face API: thinkingmachines/Inkling-Small public, not gated, license tag apache-2.0. Nulls: crawler_named: crawlers ticked Yes but the name field reads 'N/A'.",
        "orgs": [
          "Thinking Machines Lab"
        ],
        "release_date": "2026-07-30",
        "status_note": "2.3: 'Were crawlers used by the provider or on behalf of? [x] Yes'. 'If yes, specify crawler name(s)/identifier(s): N/A'. The Inkling-Small text is identical to the Inkling summary apart from the model name, including the crawl period 'From 2025 to 2026'. 2.1 names 'Common Crawl (https://commoncrawl.org/)'.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Inkling-Small was trained on a mixture of publicly available content, content acquired through partnerships, synthetic content, and generated content, including general web content, reference materials, technical documentation, source code, and other text, curated and filtered.",
        "update_note": "updated 2026-09-18 from thinkingmachines.ai: Same 'N/A' non-answer as Inkling. crawler_named deliberately left unset: the summary names no crawler. names a third-party corpus but no crawler.",
        "url": "https://thinkingmachines.ai/documents/inkling-small-public-summary-training-content.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": false,
        "weights": "open"
      },
      "id": "model:inkling-small",
      "name": "Inkling-Small",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Thinking Machines Lab": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "thinkingmachines.ai"
        ],
        "source_urls": [
          "https://thinkingmachines.ai/documents/inkling-small-public-summary-training-content.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://thinkingmachines.ai/documents/inkling-small-public-summary-training-content.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/OpenLLM_France_Luciole_2026_09_08.pdf",
        "crawler_named": [
          "CCBot"
        ],
        "cutoff_date": "2025-12",
        "data_sources_named": [
          "FineWeb 2",
          "FineWeb2-HQ",
          "FineWeb-Edu",
          "DCLM Dolmino",
          "CulturaX",
          "HPLT 2",
          "Wikipedia and Wikimedia projects",
          "Common Corpus (EUR-Lex, OECD, WTO)",
          "data.gouv.fr",
          "INSEE",
          "French Parliament",
          "Europarl",
          "Common Pile",
          "PubMed",
          "arXiv",
          "HAL",
          "Project Gutenberg",
          "Gallica",
          "StarCoder Data",
          "Stack-Edu",
          "FineMath",
          "MegaMath",
          "Nemotron Post-Training v2",
          "PleiasSynth",
          "Qwen 3 8B (synthetic data generator)"
        ],
        "developer": "LINAGORA",
        "eu_training_summary_url": "https://dl.labs.linagora.com/files/models/OpenLLM-France/AI_Act_summaries/Luciole_Base_Training_Content_Summary.pdf",
        "licensed_data_declared": "no",
        "licensed_data_named": false,
        "name": "Luciole Base (1B, 8B, 23B)",
        "note": "Release date basis: EU placement date stated in the summary ('02 June 2026'). Summary V1.1, last update 31/07/2026. Only the selected answer is printed with a '☒' glyph: 2.2.1 No, 2.2.2 No, crawlers No, user data No/No, synthetic Yes, other No, Code of Practice signatory Yes. About 4.65 trillion text tokens from 87 openly licensed source configurations, full list with per-source licences at huggingface.co/datasets/OpenLLM-France/Luciole-Training-Dataset (data_sources_named is the principal subset). Cutoff: principal phases June 2025 (CC-MAIN-2025-26), annealing and context extension December 2025. Retrospective robots.txt filter on CCBot only, limits stated by the provider. weights=open verified via Hugging Face API: OpenLLM-France/Luciole-23B-Base public, not gated, apache-2.0. Nulls: crawler_named: crawlers answered No.",
        "orgs": [
          "LINAGORA"
        ],
        "release_date": "2026-06-02",
        "status_note": "2.3 'Data crawled and scraped from online sources': 'Were crawlers used by the provider or on behalf of? [x] No'. LINAGORA operated no crawler. The crawler name appears in 3.1, describing the retroactive opt-out filter applied to every web-derived dataset: 'Robots.txt files were retrieved from the CommonCrawl dump CC-MAIN-2025-26 ... and a document was kept only where the robots.txt explicitly permitted crawling by CCBot or where the file was malformed', and in the stated limits, 'It evaluates the CCBot user-agent only'.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Substantially all of the training content consists of pre-packaged, openly licensed datasets compiled by third parties.",
        "update_note": "updated 2026-09-18 from dl.labs.linagora.com: Recorded because the summary names the user agent that actually collected the web content the model trained on, Common Crawl's CCBot, even though LINAGORA crawled nothing itself. Matches crawler:ccbot. The string is the only named crawler in the document. CCBot matches crawler:ccbot.",
        "url": "https://dl.labs.linagora.com/files/models/OpenLLM-France/AI_Act_summaries/Luciole_Base_Training_Content_Summary.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": false,
        "weights": "open"
      },
      "id": "model:luciole-base-1b-8b-23b",
      "name": "Luciole Base (1B, 8B, 23B)",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "LINAGORA": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "dl.labs.linagora.com"
        ],
        "source_urls": [
          "https://dl.labs.linagora.com/files/models/OpenLLM-France/AI_Act_summaries/Luciole_Base_Training_Content_Summary.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://dl.labs.linagora.com/files/models/OpenLLM-France/AI_Act_summaries/Luciole_Base_Training_Content_Summary.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/OpenLLM_France_Luciole_1_1_2026_09_08.pdf",
        "cutoff_date": "2026-06",
        "data_sources_named": [
          "DOLCI",
          "Nemotron Posttraining v3",
          "Nemotron Posttraining v2",
          "Open Code Reasoning",
          "Nemotron agentic SFT v2",
          "smolagent tool calling",
          "Pleias RAG",
          "XLAM",
          "Hermes",
          "When2call",
          "ParaDocs",
          "CroissantAligned",
          "Smol (instruct, rewrite, summarize)",
          "OpenMathInstruct",
          "Nemotron Safety",
          "Qwen3-32B (synthetic data generator)",
          "Qwen3-0.6B (synthetic data generator)",
          "Qwen3-14B (synthetic data generator)",
          "Ministral-3-14B-Instruct (synthetic data generator)"
        ],
        "developer": "LINAGORA",
        "eu_training_summary_url": "https://dl.labs.linagora.com/files/models/OpenLLM-France/AI_Act_summaries/Luciole_Instruct_1.1_Training_Content_Summary.pdf",
        "licensed_data_declared": "no",
        "licensed_data_named": false,
        "name": "Luciole Instruct 1.1 (1B, 8B, 23B)",
        "note": "Release date basis: EU placement date stated in the summary ('09 July 2026'). Summary V1.1, last update 31/07/2026. Post-trained from Luciole Base (SFT with and without thinking traces, then DPO). Selected answers printed with '☒': 2.2.1 No, 2.2.2 No, crawlers No, user data No/No ('no logs from public Luciole demonstrators'), synthetic Yes, Code of Practice signatory Yes. Less than 1 billion tokens per phase (about 4B SFT Thinking, 2.3B SFT, 0.75B DPO stated). DPO pairs generated with Qwen3-32B (chosen) and Qwen3-0.6B (rejected); safety data also used an interim Luciole-8B-Instruct checkpoint. Cutoff 'June 2026'. weights=open verified via Hugging Face API: OpenLLM-France/Luciole-23B-Instruct-1.1 public, not gated, apache-2.0. Nulls: crawler_named: crawlers answered No.",
        "orgs": [
          "LINAGORA"
        ],
        "release_date": "2026-07-09",
        "status_note": "2.3 'Data crawled and scraped from online sources': 'Were crawlers used by the provider or on behalf of? [x] No'. Unlike the Luciole Base summary this one contains no crawler name anywhere: the strings CCBot, Common Crawl and robots.txt do not appear. Post-training content is described as 'pre-packaged, openly licensed datasets compiled by third parties or by OpenLLM partners'.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Much of the training content for the supervised fine-tuning (SFT) phases of Instruct 1.1 models consists of pre-packaged, openly licensed datasets compiled by third parties or by OpenLLM partners.",
        "update_note": "updated 2026-09-18 from dl.labs.linagora.com: Post-training only summary. No crawler, and the CCBot filter described in the Base summary is not repeated here. crawler_named deliberately left unset: the summary names no crawler. names a third-party corpus but no crawler.",
        "url": "https://dl.labs.linagora.com/files/models/OpenLLM-France/AI_Act_summaries/Luciole_Instruct_1.1_Training_Content_Summary.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": false,
        "weights": "open"
      },
      "id": "model:luciole-instruct-1-1-1b-8b-23b",
      "name": "Luciole Instruct 1.1 (1B, 8B, 23B)",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "LINAGORA": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "dl.labs.linagora.com"
        ],
        "source_urls": [
          "https://dl.labs.linagora.com/files/models/OpenLLM-France/AI_Act_summaries/Luciole_Instruct_1.1_Training_Content_Summary.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://dl.labs.linagora.com/files/models/OpenLLM-France/AI_Act_summaries/Luciole_Instruct_1.1_Training_Content_Summary.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Magma_8B_2026_09_11.pdf",
        "cutoff_date": "2024-01",
        "data_sources_named": [
          "ShareGPT4V",
          "LLaVA-1.5 instruction data",
          "InfoGraphicVQA",
          "ChartQA",
          "FigureQA",
          "TQA",
          "ScienceQA",
          "SeeClick",
          "Vision2UI",
          "Epic-Kitchens",
          "Ego4D",
          "Something-Something v2",
          "Open-X-Embodiment"
        ],
        "developer": "Microsoft",
        "eu_training_summary_url": "https://huggingface.co/microsoft/Magma-8B/blob/main/data_summary_card.md",
        "licensed_data_declared": "no",
        "licensed_data_named": false,
        "name": "Magma-8B",
        "note": "Release date basis: summary states '19-Feb-2025'. Robotics-relevant: ~9.4 million image-language-action triplets from ~326,000 Open-X-Embodiment trajectories, plus Ego4D and Epic-Kitchens egocentric video. Format: a 3-page Microsoft 'Data Summary' card that answers template questions in plain text (no tick boxes), so Yes/No values are the card's written answers. The card has no crawler, user-data or other-service questions, so crawler_named, uses_interaction_data and uses_other_service_data are null: not asked in this format. 2.2.1.A No, 2.2.2.A No, synthetic Yes. Latest data acquisition '11-Jan-2024'. weights=open verified via Hugging Face API: microsoft/Magma-8B public, ungated, license tag mit.",
        "orgs": [
          "Microsoft"
        ],
        "release_date": "2025-02-19",
        "status_note": "Microsoft 'Data Summary' card, three pages. No crawled-data section exists in the card (its 2.3 is 'Personal Information') and the words crawler, crawl and scrape do not appear. Sources are named as datasets in 1.3.1: Open-X-Embodiment, Ego4D, Epic-Kitchens, Something-Something v2, ShareGPT4V, SeeClick and others.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Robotics manipulation datasets from Open-X-Embodiment used for vision-language-action learning, including 7-DoF gripper states and visual traces to support action prediction",
        "update_note": "updated 2026-09-18 from huggingface.co: Names many datasets, no crawler, because the card format has no crawler question. crawler_named deliberately left unset: the summary names no crawler. names a third-party corpus but no crawler.",
        "url": "https://huggingface.co/microsoft/Magma-8B/blob/main/data_summary_card.md",
        "weights": "open"
      },
      "id": "model:magma-8b",
      "name": "Magma-8B",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Microsoft": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "huggingface.co"
        ],
        "source_urls": [
          "https://huggingface.co/microsoft/Magma-8B/blob/main/data_summary_card.md"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://huggingface.co/microsoft/Magma-8B/blob/main/data_summary_card.md"
    },
    {
      "fields": {
        "crawler_named": [
          "Bingbot"
        ],
        "cutoff_date": "2026-07",
        "data_sources_named": [
          "GitHub public repositories",
          "Wikipedia",
          "Common Crawl",
          "GitHub Copilot Free, Pro and Pro+ conversation contexts",
          "Bingbot (crawler)",
          "MAI-Thinking-1 (base model)",
          "MAI-Code-1-Flash (dependency)"
        ],
        "developer": "Microsoft",
        "eu_training_summary_url": "https://microsoft.ai/pdf/MAI-Code-1.1-Flash-Data-Card.PDF",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "MAI-Code-1.1-Flash",
        "note": "Release date basis: release and EU placement date both stated in the summary (11 August 2026). Read directly from the provider PDF (HTTP 200, 7 pages). Unlike MAI-Code-1-Flash, 2.2.1.B ticks Text and Image for licensed data; 2.2.1.A 'Yes, we leveraged data acquisition agreements'. Licensors unnamed: 2.2.2.C says 'Relevant data acquisition deals are bound by confidentiality terms and conditions.' 2.4.2 Yes (GitHub Copilot Free, Pro, Pro+ prompts), 2.4.1 No. Synthetic data from internal MAI-Thinking-1 checkpoints (MAI-Base-1), stated as not from public models. Cutoff from 'datasets collected as late as July 2026'; crawl period February 2024 to December 2025. Nulls: archive_url: the AIAL mirror holds no PDF for this model (its eval lists archive_file_name None); weights null: no microsoft/MAI-* repo for this model on Hugging Face (API search author=microsoft returned none) and the summary does not state API-only availability.",
        "orgs": [
          "Microsoft"
        ],
        "release_date": "2026-08-11",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Training data combines publicly available, commercially acquired, and crawled web data (inherited from the MAI-Thinking-1 base) with code-specific public datasets.",
        "url": "https://microsoft.ai/pdf/MAI-Code-1.1-Flash-Data-Card.PDF",
        "uses_interaction_data": false,
        "uses_other_service_data": true
      },
      "id": "model:mai-code-1-1-flash",
      "name": "MAI-Code-1.1-Flash",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Microsoft": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "microsoft.ai"
        ],
        "source_urls": [
          "https://microsoft.ai/pdf/MAI-Code-1.1-Flash-Data-Card.PDF"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://microsoft.ai/pdf/MAI-Code-1.1-Flash-Data-Card.PDF"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/MAI_Code_1_Flash_2026_08_03.pdf",
        "crawler_named": [
          "Bingbot"
        ],
        "cutoff_date": "2026-05",
        "data_sources_named": [
          "GitHub public repositories",
          "Wikipedia",
          "Common Crawl",
          "GitHub Copilot Free, Pro and Pro+ conversation contexts",
          "Bingbot (crawler)",
          "GPT-4o and GPT-5 (data preparation, OpenAI)",
          "MAI-Thinking-1 (base model)"
        ],
        "developer": "Microsoft",
        "eu_training_summary_url": "https://microsoft.ai/pdf/MAI-Code-1-Flash-Data-Card.PDF",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "MAI-Code-1-Flash",
        "note": "Release date basis: release and EU placement date both stated in the summary (2 June 2026). 2.2.1.A ticks 'Yes, we leveraged data acquisition agreements' for Text; 2.2.2.A Yes. Licensors unnamed: 2.2.2.C says 'Relevant data acquisition deals are bound by confidentiality terms and conditions.' 2.4.2 Yes: GitHub Copilot Free, Pro and Pro+ prompts from users who did not opt out were used as RL rollout inputs and for a reward model; 2.4.1 (the model's own users) No. Crawl period February 2024 to December 2025; Bingbot purpose 'Index web content for both search and model training'. Cutoff from 'datasets collected as late as May 2026'. Tick glyphs checked against a rendered page (the 'Yes' box of 2.4.1 is an empty box drawn without a text glyph). Nulls: weights null: no microsoft/MAI-* repo for this model on Hugging Face (API search author=microsoft returned none) and the summary does not state API-only availability.",
        "orgs": [
          "Microsoft"
        ],
        "release_date": "2026-06-02",
        "synthetic_data_stated": true,
        "training_data_disclosure": "We used a variety of large, publicly available datasets including GitHub public repositories, Wikipedia, and CommonCrawl (English and multilingual).",
        "url": "https://microsoft.ai/pdf/MAI-Code-1-Flash-Data-Card.PDF",
        "uses_interaction_data": false,
        "uses_other_service_data": true
      },
      "id": "model:mai-code-1-flash",
      "name": "MAI-Code-1-Flash",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Microsoft": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "microsoft.ai"
        ],
        "source_urls": [
          "https://microsoft.ai/pdf/MAI-Code-1-Flash-Data-Card.PDF"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://microsoft.ai/pdf/MAI-Code-1-Flash-Data-Card.PDF"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/MAI_Cyber_1_Flash_2026_08_03.pdf",
        "crawler_named": [
          "Bingbot"
        ],
        "cutoff_date": "2026-05",
        "data_sources_named": [
          "GitHub public repositories",
          "Wikipedia",
          "Common Crawl",
          "GitHub Copilot Free, Pro and Pro+ conversation contexts",
          "Bingbot (crawler)",
          "GPT-4o and GPT-5 (data preparation, OpenAI)",
          "MAI-Code-1-Flash (base model)"
        ],
        "developer": "Microsoft",
        "eu_training_summary_url": "https://microsoft.ai/pdf/MAI-Cyber-1-Flash-Data-Card.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "MAI-Cyber-1-Flash",
        "note": "Release date basis: release and EU placement date both stated in the summary (27 July 2026). Dependency stated as 'MAI-Code-1-Flash, which is based on MAI-Thinking-1'. 2.2.1.A Yes (data acquisition agreements, Text), 2.2.2.A Yes. Licensors unnamed: 2.2.2.C says 'Relevant data acquisition deals are bound by confidentiality terms and conditions.' 2.4.2 Yes (GitHub Copilot prompts), 2.4.1 No. Cutoff from 'datasets collected as late as May 2026'; crawl period February 2024 to December 2025. AIAL eval lists no archive file name but the mirror holds MAI_Cyber_1_Flash_2026_08_03.pdf, which is what was read. Nulls: weights null: no microsoft/MAI-* repo for this model on Hugging Face (API search author=microsoft returned none) and the summary does not state API-only availability.",
        "orgs": [
          "Microsoft"
        ],
        "release_date": "2026-07-27",
        "synthetic_data_stated": true,
        "training_data_disclosure": "The inherited MAI-Code-1-Flash corpus includes source code and software-engineering data, including source files, pull requests, and related code artifacts, drawn from publicly available, commercially licensed and crawled web data",
        "url": "https://microsoft.ai/pdf/MAI-Cyber-1-Flash-Data-Card.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": true
      },
      "id": "model:mai-cyber-1-flash",
      "name": "MAI-Cyber-1-Flash",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Microsoft": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "microsoft.ai"
        ],
        "source_urls": [
          "https://microsoft.ai/pdf/MAI-Cyber-1-Flash-Data-Card.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://microsoft.ai/pdf/MAI-Cyber-1-Flash-Data-Card.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/MAI_DS_R1_2026_09_11.pdf",
        "cutoff_date": "2025-03",
        "data_sources_named": [
          "Tulu 3 SFT (Safety and Non-Compliance subset)"
        ],
        "developer": "Microsoft",
        "eu_training_summary_url": "https://huggingface.co/microsoft/MAI-DS-R1/blob/main/data_summary_card.md",
        "licensed_data_named": false,
        "name": "MAI-DS-R1",
        "note": "Release date basis: summary states 'April 2025' (HF repo created 2025-04-16). Card covers only post-training data; base model is not named in the card. Format: a 3-page Microsoft 'Data Summary' card that answers template questions in plain text (no tick boxes), so Yes/No values are the card's written answers. The card has no crawler, user-data or other-service questions, so crawler_named, uses_interaction_data and uses_other_service_data are null: not asked in this format. 2.2.2.A No; synthetic Yes. Nulls: licensed_data_declared null: 2.2.1.A is answered 'Not applicable', which is not a Yes/No/Other box value. weights=open verified via Hugging Face API: microsoft/MAI-DS-R1 public, ungated, license tag mit.",
        "orgs": [
          "Microsoft"
        ],
        "release_date": "2025-04",
        "status_note": "Microsoft 'Data Summary' card, three pages. No crawled-data section (its 2.3 is 'Personal Information'); crawler, crawl and scrape do not appear. The card covers post-training only: '110k Safety and Non-Compliance examples from the Tulu 3 SFT dataset and ~350k multilingual examples internally developed'.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "The model was post-trained using 110k Safety and Non-Compliance examples from the Tulu 3 SFT dataset and ~350k multilingual examples internally developed capturing various topics with reported biases",
        "update_note": "updated 2026-09-18 from huggingface.co: No crawler question in the format; the one named source is the Tulu 3 SFT subset. crawler_named deliberately left unset: the summary names no crawler. names a third-party corpus but no crawler.",
        "url": "https://huggingface.co/microsoft/MAI-DS-R1/blob/main/data_summary_card.md",
        "weights": "open"
      },
      "id": "model:mai-ds-r1",
      "name": "MAI-DS-R1",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Microsoft": [
            "developer"
          ]
        },
        "_release_date_precision": "month"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "huggingface.co"
        ],
        "source_urls": [
          "https://huggingface.co/microsoft/MAI-DS-R1/blob/main/data_summary_card.md"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.846,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 0,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://huggingface.co/microsoft/MAI-DS-R1/blob/main/data_summary_card.md"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/MAI_Image_2_2026_07_21.pdf",
        "crawler_named": [
          "Bingbot"
        ],
        "cutoff_date": "2026-02",
        "data_sources_named": [
          "Wikipedia",
          "Bingbot (crawler)"
        ],
        "developer": "Microsoft",
        "eu_training_summary_url": "https://microsoft.ai/pdf/MAI-Image-2-Data-Summary.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "MAI-Image-2",
        "note": "Release date basis: model release date stated in the summary; EU placement listed as 'Coming soon.'. Most explicit about why licensors go unnamed: 'Relevant data acquisition deals are bound by confidentiality terms and conditions. If the parties mutually agree to publicize the partnership in the future, we will update this data summary'. Bingbot stated purpose: 'Index web content for both search and model training', i.e. one crawler for search and training. weights null, not checked. Registry Microsoft deals (AP, HarperCollins, Taylor & Francis, Copilot Daily publishers) are text, not image, so non-naming is not a contradiction for this image model. Nulls: weights: not established, the model is not published on Hugging Face under the developer's account and no second party states it",
        "orgs": [
          "Microsoft"
        ],
        "release_date": "2026-03-19",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Training data includes text and images from Wikipedia, which is a large scale, multi-domain, open source, and publicly available collection from a reputable online source.",
        "url": "https://microsoft.ai/pdf/MAI-Image-2-Data-Summary.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": false
      },
      "id": "model:mai-image-2",
      "name": "MAI-Image-2",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Microsoft": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "microsoft.ai"
        ],
        "source_urls": [
          "https://microsoft.ai/pdf/MAI-Image-2-Data-Summary.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://microsoft.ai/pdf/MAI-Image-2-Data-Summary.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/MAI_Image_2_5_2026_08_03.pdf",
        "crawler_named": [
          "Bingbot"
        ],
        "cutoff_date": "2026-05",
        "data_sources_named": [
          "Wikipedia",
          "Bingbot (crawler)",
          "GPT-4o (image captioning)"
        ],
        "developer": "Microsoft",
        "eu_training_summary_url": "https://microsoft.ai/pdf/MAI-Image-2.5-Data-Card.PDF",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "MAI-Image-2.5",
        "note": "Release date basis: summary states '6/2/2026', read as 2 June 2026 (US order), matching AIAL metadata 2026-06-02; EU placement listed as 'Coming soon.'. Last update 31 May 2026. 2.2.1.A 'Yes, we leveraged data acquisition agreements' for Image only; 2.2.2.A Yes, described as 'synthetic image-based text captions and images'. Licensors unnamed: 2.2.2.C says 'Relevant data acquisition deals are bound by confidentiality terms and conditions.' Both user-data questions ticked No (checked on a rendered page). Synthetic: VLM captions, 'GPT-4o was used to caption approximately 1000 images'. Crawl period February 2024 to May 2026 (image alt text); cutoff from 'datasets collected as late as May 2026'. Same boilerplate as the carried MAI-Image-2 row. Nulls: weights null: no microsoft/MAI-* repo for this model on Hugging Face (API search author=microsoft returned none) and the summary does not state API-only availability.",
        "orgs": [
          "Microsoft"
        ],
        "release_date": "2026-06-02",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Training data includes text and images from Wikipedia, which is a large scale, multi-domain, open source, and publicly available collection from a reputable online source.",
        "url": "https://microsoft.ai/pdf/MAI-Image-2.5-Data-Card.PDF",
        "uses_interaction_data": false,
        "uses_other_service_data": false
      },
      "id": "model:mai-image-2-5",
      "name": "MAI-Image-2.5",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Microsoft": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "microsoft.ai"
        ],
        "source_urls": [
          "https://microsoft.ai/pdf/MAI-Image-2.5-Data-Card.PDF"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://microsoft.ai/pdf/MAI-Image-2.5-Data-Card.PDF"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/MAI_Image_2_6_2026_09_11.pdf",
        "crawler_named": [
          "Bingbot"
        ],
        "cutoff_date": "2026-05",
        "data_sources_named": [
          "Wikipedia",
          "Bingbot (crawler)",
          "GPT-4o (image captioning)"
        ],
        "developer": "Microsoft",
        "eu_training_summary_url": "https://microsoft.ai/pdf/MAI-Image-2.6-Data-Summary.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "MAI-Image-2.6 (and 2.6-Flash)",
        "note": "Release date basis: release date stated in the summary (14 August 2026); EU placement stated as 4 September 2026. Covers MAI-Image-2.6 and MAI-Image-2.6-Flash; last update 4 September 2026. 2.2.1.A 'Yes, we leveraged data acquisition agreements' for Image only; 2.2.2.A Yes, described as 'synthetic image-based text captions and images'. Licensors unnamed: 2.2.2.C says 'Relevant data acquisition deals are bound by confidentiality terms and conditions.' Both user-data questions ticked No (checked on a rendered page). Synthetic: VLM captions, 'GPT-4o was used to caption approximately 1000 images'. Crawl period February 2024 to May 2026 (image alt text); cutoff from 'datasets collected as late as May 2026'. Same boilerplate as the carried MAI-Image-2 row. Nulls: weights null: no microsoft/MAI-* repo for this model on Hugging Face (API search author=microsoft returned none) and the summary does not state API-only availability.",
        "orgs": [
          "Microsoft"
        ],
        "release_date": "2026-08-14",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Training data includes text and images from Wikipedia, which is a large scale, multi-domain, open source, and publicly available collection from a reputable online source.",
        "url": "https://microsoft.ai/pdf/MAI-Image-2.6-Data-Summary.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": false
      },
      "id": "model:mai-image-2-6-and-2-6-flash",
      "name": "MAI-Image-2.6 (and 2.6-Flash)",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Microsoft": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "microsoft.ai"
        ],
        "source_urls": [
          "https://microsoft.ai/pdf/MAI-Image-2.6-Data-Summary.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://microsoft.ai/pdf/MAI-Image-2.6-Data-Summary.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/MAI_Thinking_1_2026_09_11.pdf",
        "crawler_named": [
          "Bingbot"
        ],
        "cutoff_date": "2026-07",
        "data_sources_named": [
          "GitHub public repositories",
          "Wikipedia",
          "Common Crawl",
          "Microsoft Consumer Copilot conversation logs",
          "Bingbot (crawler)"
        ],
        "developer": "Microsoft",
        "eu_training_summary_url": "https://microsoft.ai/pdf/MAI-Thinking-1-Data-Summary.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "MAI-Thinking-1",
        "note": "Release date basis: model release date stated in the summary (2 June 2026); EU placement stated separately as 12 August 2026. Base model for MAI-Code-1-Flash and MAI-Cyber-1-Flash. 2.2.1.A Yes (data acquisition agreements, Text), 2.2.2.A Yes. Licensors unnamed: 2.2.2.C says 'Relevant data acquisition deals are bound by confidentiality terms and conditions.' 2.4.2 Yes: 'Conversation logs from Microsoft Consumer Copilot users who have not opted out' used as RL inputs; 2.4.1 No ('User interaction data from the model itself is not used'). 2.5.3: 'Microsoft did not use publicly available general-purpose AI models to create synthetic outputs for direct training'; internal MAI-Base-1 checkpoints were. Cutoff from 'datasets collected as late as July 2026'; crawl period February 2024 to December 2025. Nulls: weights null: no microsoft/MAI-* repo for this model on Hugging Face (API search author=microsoft returned none) and the summary does not state API-only availability.",
        "orgs": [
          "Microsoft"
        ],
        "release_date": "2026-06-02",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Training data combines publicly available, commercially acquired, and crawled web data.",
        "url": "https://microsoft.ai/pdf/MAI-Thinking-1-Data-Summary.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": true
      },
      "id": "model:mai-thinking-1",
      "name": "MAI-Thinking-1",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Microsoft": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "microsoft.ai"
        ],
        "source_urls": [
          "https://microsoft.ai/pdf/MAI-Thinking-1-Data-Summary.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://microsoft.ai/pdf/MAI-Thinking-1-Data-Summary.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Midjourney_Midjourney_2026_09_08.pdf",
        "cutoff_date": "2026-03",
        "developer": "Midjourney",
        "eu_training_summary_url": "https://docs.midjourney.com/hc/en-us/articles/48067080311309-Public-Summary-of-Training-Content",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "Midjourney Image and Video family (V8, V8.1, V8.2)",
        "note": "Release date basis: EU placement date stated in the summary. The crawler name field reads 'Deals are bound by confidentiality obligations', a non-answer, so crawler_named null. 'Some Midjourney datasets are purchased or licensed from third parties. These deals are bound by confidentiality obligations.' Crawl period '2023 to present', 'continuously trained'. Registry holds Andersen v. Stability AI (Midjourney co-defendant) and Disney v. Midjourney. weights=closed from service-only availability, not re-verified.",
        "orgs": [
          "Midjourney"
        ],
        "release_date": "2026-03-17",
        "status_note": "2.3: 'Were crawlers used by the provider or on behalf of? [x] Yes'. The field 'If yes, specify crawler name(s)/identifier(s):' is answered 'Deals are bound by confidentiality obligations', a sentence about licensing deals placed in the crawler-name box. Purposes: 'Crawlers are used to obtain publicly available content for training purposes.' Period of data collection: '2023 to present'. Domains: 'Toplevel domain names crawled include: .com, .org, and .net'.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Midjourney datasets are composed of a wide variety of data publicly accessible online.",
        "update_note": "updated 2026-09-18 from raw.githubusercontent.com: The crawler-name answer is a non-answer imported from the licensing section. No crawler named, no dataset named, only top-level domains. crawler_named deliberately left unset: the summary names no crawler. developer URL https://docs.midjourney.com/hc/en-us/articles/48067080311309-Public-Summary-of-Training-Content returned HTTP 403 on 2026-09-18,",
        "url": "https://docs.midjourney.com/hc/en-us/articles/48067080311309-Public-Summary-of-Training-Content",
        "uses_interaction_data": true,
        "uses_other_service_data": false,
        "weights": "closed"
      },
      "id": "model:midjourney-image-and-video-family-v8-v8-1-v8-2",
      "name": "Midjourney Image and Video family (V8, V8.1, V8.2)",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Midjourney": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "docs.midjourney.com",
          "raw.githubusercontent.com"
        ],
        "source_urls": [
          "https://docs.midjourney.com/hc/en-us/articles/48067080311309-Public-Summary-of-Training-Content",
          "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Midjourney_Midjourney_2026_09_08.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.846,
        "confidence": "high",
        "corroborated": true,
        "flags": [],
        "grade": "A",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 2,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://docs.midjourney.com/hc/en-us/articles/48067080311309-Public-Summary-of-Training-Content"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Minimax_H3_2026_09_11.pdf",
        "crawler_named": [
          "MINIMAX_UA"
        ],
        "cutoff_date": "2026-07",
        "data_sources_named": [
          "MINIMAX_UA (crawler)"
        ],
        "developer": "MiniMax",
        "eu_training_summary_url": "https://file.cdn.minimax.io/public/8b640d09-235b-4351-a096-2f7bea762e75.pdf",
        "licensed_data_declared": "yes",
        "licensed_data_named": false,
        "name": "MiniMax H3",
        "note": "Release date basis: none adopted. The summary gives placement '08/02/2026' and last update '08/02/2026' but a latest data collection of 'July 2026' and crawl 'Up to July 2026'; read dd/mm (8 Feb, as AIAL does) it precedes its own cutoff, read mm/dd (2 Aug) it fits and matches the Hugging Face repo creation 2026-07-28. Text, image, audio (>1M h) and video (>1M h) model. 2.2.1 Yes for Image, Video, Audio, no licensor named; 2.2.2 Yes, 'obtained from third parties on a licensed basis'. No large public dataset named. Not a Code of Practice signatory. Both user-data answers No. weights=open verified via Hugging Face API: MiniMaxAI/MiniMax-H3 public, license tag other. Nulls: release_date: placement date format is ambiguous and one reading contradicts the stated cutoff, so no date is asserted.",
        "orgs": [
          "MiniMax"
        ],
        "synthetic_data_stated": true,
        "training_data_disclosure": "The training data comprises a mix of publicly available and licensed data. This may include image and video datasets made available by third parties through public repositories and online platforms.",
        "url": "https://file.cdn.minimax.io/public/8b640d09-235b-4351-a096-2f7bea762e75.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": false,
        "weights": "open"
      },
      "id": "model:minimax-h3",
      "name": "MiniMax H3",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "MiniMax": [
            "developer"
          ]
        }
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "file.cdn.minimax.io"
        ],
        "source_urls": [
          "https://file.cdn.minimax.io/public/8b640d09-235b-4351-a096-2f7bea762e75.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.923,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://file.cdn.minimax.io/public/8b640d09-235b-4351-a096-2f7bea762e75.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Minimax_M3_2026_08_03.pdf",
        "crawler_named": [
          "MINIMAX_UA"
        ],
        "cutoff_date": "2026-01",
        "data_sources_named": [
          "Common Crawl",
          "MINIMAX_UA (crawler)",
          "MiniMax-M2.7 (synthetic data generator)"
        ],
        "developer": "MiniMax",
        "eu_training_summary_url": "https://file.cdn.minimax.io/public/0674772e-317a-4928-8af9-2489903d1c09.pdf",
        "licensed_data_declared": "other",
        "licensed_data_named": false,
        "name": "MiniMax-M3",
        "note": "Release date basis: EU placement date stated in the summary ('01/06/2026', dd/mm reading consistent with 'Last update: 22/07/2026'). Provider entity is Nanonoble Pte. Ltd., EU representative Prighter GmbH. 2.2.1 ticks 'Other (see below)': 'MiniMax conducts commercial procurement from third party licensors for datasets. Such collaborations may include rights to access non-public materials including archives and metadata.' (Text, Image, Video, Other); no licensor named. 2.2.2 Yes, unnamed. Crawl period April 2025 to December 2025; cutoff 'no later than January 2026'. User data: model interactions No, other AI products Yes. 2.6 Yes: data produced with industry professionals and suppliers. No MiniMax deals or suits in the registry. weights=open verified via Hugging Face API: MiniMaxAI/MiniMax-M3 public, license tag other (custom licence, not re-read).",
        "orgs": [
          "MiniMax"
        ],
        "release_date": "2026-06-01",
        "synthetic_data_stated": true,
        "training_data_disclosure": "Trained on a large-scale, multilingual mixture of publicly available data, data from commercial procurement/licensed data, synthetic data, and human-generated text",
        "url": "https://file.cdn.minimax.io/public/0674772e-317a-4928-8af9-2489903d1c09.pdf",
        "uses_interaction_data": false,
        "uses_other_service_data": true,
        "weights": "open"
      },
      "id": "model:minimax-m3",
      "name": "MiniMax-M3",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "MiniMax": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-explore",
        "source_hosts": [
          "file.cdn.minimax.io"
        ],
        "source_urls": [
          "https://file.cdn.minimax.io/public/0674772e-317a-4928-8af9-2489903d1c09.pdf"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 1,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://file.cdn.minimax.io/public/0674772e-317a-4928-8af9-2489903d1c09.pdf"
    },
    {
      "fields": {
        "archive_url": "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Ministral_3_14B_2026_08_03.pdf",
        "cutoff_date": "2025-07",
        "data_sources_named": [
          "Common Crawl",
          "Mistral Small 3.1 (synthetic data generator)",
          "Mistral Medium 3 (synthetic data generator)"
        ],
        "developer": "Mistral AI",
        "eu_training_summary_url": "https://legal.cms.mistral.ai/assets/36afc281-be9c-4cd0-9763-81cc19540895",
        "licensed_data_declared": "other",
        "licensed_data_named": false,
        "name": "Ministral 3 14B (Base, Instruct, Reasoning)",
        "note": "Release date basis: EU placement date stated in the summary ('December 2, 2025', Base, Instruct and Reasoning). Model dependency: Mistral Small 3.1. 2.2.1 ticks 'Other': 'Mistral AI concludes data access agreements with rights holders or their representatives for access to non-publicly available datasets.' (Text, Image). No licensor named, though the registry holds an Agence France-Presse to Mistral AI deal. 2.2.2 Yes, 'a variety of third-party providers for access to synthetically generated and human-curated datasets', none named. Crawlers used (Yes) but the name field reads 'NA.', so crawler_named is null. User data: model interactions No, other services Yes (opt-out). Checkbox answers read from the text layer glyphs (☒/☐), same template as Mistral Large 3. Header 'Last update: 31/07/2025' predates the stated placement date, likely a year typo for 2026. Cutoff: 'The latest date of data collection was July 2025.' weights=open verified via Hugging Face API: mistralai/Ministral-3-14B-Base-2512 public, license tag apache-2.0.",
        "orgs": [
          "Mistral AI"
        ],
        "release_date": "2025-12-02",
        "status_note": "2.3 'Data crawled and scraped from online sources': 'Were crawlers used by the provider or on behalf of? [x] Yes'. 'If yes, specify crawler name(s)/identifier(s): NA.' Purposes: 'Crawlers were used to collect publicly available sources on the internet.' Behaviour: 'Our crawlers are designed to respect robots.txt, extract [content] ...'. 2.1 names one dataset: 'The datasets used to train the model include Common Crawl.' The crawler-name answer is literally the two characters NA followed by a full stop.",
        "synthetic_data_stated": true,
        "training_data_disclosure": "The text dataset is a large-scale, multilingual text dataset, comprising highly general content originating from publicly available text datasets and user data, as well as highly specialized and technical datasets, both synthetically generated and human-curated by third-party providers.",
        "update_note": "updated 2026-09-18 from raw.githubusercontent.com: Mistral operates the documented user agents MistralAI-Training, MistralAI-Index and MistralAI-User (all carried in the registry), and names none of them in its own EU summary. The live summary URL on legal.cms.mistral.ai is dead: the host answers 'no Route matched with those values'. crawler_named deliberately left unset: the summary names no craw",
        "url": "https://legal.cms.mistral.ai/assets/36afc281-be9c-4cd0-9763-81cc19540895",
        "uses_interaction_data": false,
        "uses_other_service_data": true,
        "weights": "open"
      },
      "id": "model:ministral-3-14b-base-instruct-reasoning",
      "name": "Ministral 3 14B (Base, Instruct, Reasoning)",
      "notes": {
        "_cutoff_date_precision": "month",
        "_org_roles": {
          "Mistral AI": [
            "developer"
          ]
        },
        "_release_date_precision": "day"
      },
      "provenance": {
        "consent_license": "third-party-derived",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-update",
        "source_hosts": [
          "legal.cms.mistral.ai",
          "raw.githubusercontent.com"
        ],
        "source_urls": [
          "https://legal.cms.mistral.ai/assets/36afc281-be9c-4cd0-9763-81cc19540895",
          "https://raw.githubusercontent.com/AIAccountabilityLab/gpai-training-transparency/HEAD/public/archive/Ministral_3_14B_2026_08_03.pdf"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.923,
        "confidence": "high",
        "corroborated": true,
        "flags": [],
        "grade": "A",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 2,
        "stale": false,
        "stale_after_days": 120
      },
      "type": "model",
      "url": "https://legal.cms.mistral.ai/assets/36afc281-be9c-4cd0-9763-81cc19540895"
    }
  ],
  "_meta": {
    "source": "Blomega Data Refinery",
    "url": "https://data.blomega.com",
    "publisher": "Blomega",
    "publisher_url": "https://blomegalab.com",
    "wikidata": "Q141048865",
    "license": "CC BY 4.0",
    "license_url": "https://creativecommons.org/licenses/by/4.0/",
    "cite_as": "Blomega Data Refinery (https://data.blomega.com), CC BY 4.0. Cite the registry and the record id.",
    "attribution_required": true
  }
}
