{
  "total": 62,
  "limit": 50,
  "offset": 0,
  "records": [
    {
      "fields": {
        "access_gated": "auto",
        "commercial_use": null,
        "downloads_30d": 721194,
        "license": "CC-BY-SA-4.0",
        "license_family": "unclear-scope",
        "license_spdx": "CC-BY-SA-4.0",
        "name": "10Kh-RealOmin-OpenData",
        "note": "Added from the Hugging Face datasets API on 2026-09-17 as one of the most-downloaded corpora whose card tag permits commercial use. 721,194 downloads in 30 days, 270 likes, card last modified 2026-04-24. The licence is the uploader's card tag: the repo tree was walked and holds no LICENSE file, so commercial use is NOT established. Size, hours and episodes are null: the API does not state them.",
        "org": "GenRobot",
        "orgs": [
          "GenRobot"
        ],
        "redistribution": null,
        "url": "https://huggingface.co/datasets/genrobot2025/10Kh-RealOmin-OpenData"
      },
      "id": "dataset:10kh-realomin-opendata",
      "name": "10Kh-RealOmin-OpenData",
      "notes": {
        "_license_note": "UNCHANGED, and specifically the gate did NOT open. Re-checked live: still gated 'auto', lastModified still 2026-04-24T05:02:26Z. The tag is readable from public metadata; the card body and whatever agreement the gate presents are not. A card tag on a gated repo does not establish commercial use, and the word 'OpenData' in the name establishes nothing. Commercial use cannot be established.",
        "_org_roles": {
          "GenRobot": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-mine",
        "source_hosts": [
          "huggingface.co"
        ],
        "source_urls": [
          "https://huggingface.co/datasets/genrobot2025/10Kh-RealOmin-OpenData",
          "https://huggingface.co/api/datasets/genrobot2025/10Kh-RealOmin-OpenData"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.556,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "license-verified",
          "single-source"
        ],
        "grade": "C",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://huggingface.co/datasets/genrobot2025/10Kh-RealOmin-OpenData"
    },
    {
      "fields": {
        "commercial_use": null,
        "episodes": 1275,
        "has_force": true,
        "hours": 2.32,
        "license": "No licence at any layer: the HuggingFace card YAML has only pretty_name and tags, the README states no terms, and the re...",
        "license_family": "unknown",
        "license_note": "Re-checked today and extended past the card YAML to the file tree: eleven task directories, .gitattributes and README.md, and no LICENSE, LICENSE.txt or COPYING. The card was last modified 2026-09-03 and is not gated, so the absence is not an access artefact. No grant means default copyright and commercial use cannot be read as permitted.",
        "license_spdx": null,
        "modalities": [
          "force",
          "rgb",
          "tactile"
        ],
        "name": "AetheRock demonstration set",
        "notes": "6 tasks, 138.91 minutes total: Clamp Seal 229 eps, Towel Hanging 206, Pick Bread 199, Erase Board 215, Pick Block 214, Insert Flower 212.",
        "org": [
          "Shanghai Jiao Tong University",
          "Ant Group"
        ],
        "orgs": [
          "Shanghai Jiao Tong University",
          "Ant Group"
        ],
        "redistribution": null,
        "url": "https://arxiv.org/abs/2606.09777"
      },
      "id": "dataset:aetherock-demonstration-set",
      "name": "AetheRock demonstration set",
      "notes": {
        "_license_note": "licence string did not match any known family",
        "_org_roles": {
          "Ant Group": [
            "org"
          ],
          "Shanghai Jiao Tong University": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-09",
        "last_seen": "2026-09-17",
        "method": "refinery-mine",
        "source_hosts": [
          "arxiv.org",
          "huggingface.co",
          "open-x-tactile.github.io"
        ],
        "source_urls": [
          "https://huggingface.co/datasets/lihong-cs/aetherock",
          "https://open-x-tactile.github.io/",
          "https://arxiv.org/abs/2606.09777"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.889,
        "confidence": "high",
        "corroborated": true,
        "flags": [],
        "grade": "A",
        "hosts": 3,
        "independent_hosts": 3,
        "sources": 3,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://arxiv.org/abs/2606.09777"
    },
    {
      "fields": {
        "commercial_use": false,
        "embodiment": "humanoid",
        "episodes": 1001552,
        "format": "custom",
        "has_force": null,
        "hours": 2976,
        "license": "CC BY-NC (check)",
        "license_family": "noncommercial",
        "license_spdx": "CC-BY-NC-4.0",
        "modalities": [
          "depth",
          "proprioception",
          "rgb",
          "text"
        ],
        "name": "AgiBot World",
        "org": "AgiBot",
        "orgs": [
          "AgiBot"
        ],
        "redistribution": true,
        "url": "https://arxiv.org/abs/2503.06669",
        "year": 2024
      },
      "id": "dataset:agibot-world",
      "name": "AgiBot World",
      "notes": {
        "_license_note": "source hedges on the licence; verify at primary source",
        "_org_roles": {
          "AgiBot": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-09-17",
        "method": "refinery-corroborate",
        "source_hosts": [
          "arxiv.org",
          "huggingface.co"
        ],
        "source_urls": [
          "https://huggingface.co/datasets/agibot-world/AgiBotWorld-Beta",
          "https://arxiv.org/abs/2503.06669"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.778,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "license-needs-review",
          "single-source"
        ],
        "grade": "B",
        "hosts": 2,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://arxiv.org/abs/2503.06669"
    },
    {
      "fields": {
        "commercial_use": false,
        "embodiment": "Franka Research 3 (simulated)",
        "episodes": 50000,
        "format": "zarr",
        "has_force": false,
        "hours": null,
        "license": "other (gated, non-commercial academic/education only)",
        "license_family": "research-only",
        "license_spdx": null,
        "modalities": [
          "action",
          "proprioception",
          "rgb",
          "text"
        ],
        "name": "AXIS Franka Dataset (Axis Sim Dataset V1)",
        "notes": "Verified 2026-09-09 from the HF tree API: data/ contains exactly action, head_camera, state; meta/ contains episode_ends. No wrench, effort, tactile or depth array. 100 root shards, names sum to 66,112 trajectories. 2.57 TB, 99,120 files, 2,162 trailing-30-day downloads.",
        "org": "Axis Robotics",
        "orgs": [
          "Axis Robotics"
        ],
        "redistribution": null,
        "url": "https://huggingface.co/datasets/axisrobotics/Franka-Dataset",
        "year": 2026
      },
      "id": "dataset:axis-franka-dataset-axis-sim-dataset-v1",
      "name": "AXIS Franka Dataset (Axis Sim Dataset V1)",
      "notes": {
        "_org_roles": {
          "Axis Robotics": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "unknown",
        "first_seen": "2026-09-09",
        "last_seen": "2026-09-09",
        "method": "api",
        "source_hosts": [
          "axisaiorg.github.io",
          "huggingface.co"
        ],
        "source_urls": [
          "https://huggingface.co/api/datasets/axisrobotics/Franka-Dataset",
          "https://axisaiorg.github.io/AXIS-V1",
          "https://huggingface.co/datasets/axisrobotics/Franka-Dataset"
        ]
      },
      "quality": {
        "age_days": 9,
        "completeness": 0.778,
        "confidence": "high",
        "corroborated": true,
        "flags": [],
        "grade": "B",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 3,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://huggingface.co/datasets/axisrobotics/Franka-Dataset"
    },
    {
      "fields": {
        "commercial_use": null,
        "episodes": 1300,
        "has_force": false,
        "license": "MIT",
        "license_family": "unclear-scope",
        "license_spdx": "MIT",
        "modalities": [
          "bounding-boxes",
          "depth",
          "object-state",
          "occupancy",
          "proprioception",
          "rgb",
          "tactile"
        ],
        "name": "Bench2Dex",
        "notes": "Isaac Lab simulation benchmark, 12 bimanual dexterous embodiments, 26 long-horizon tasks, ~1.3K teleoperated demos via Manus glove + DexPilot + ARKit. Tactile is ray-cast penetration depth, explicitly not calibrated to any physical sensor and reported with no force units. All 20,800 evaluation episodes ran on RGB + proprioception; no tactile-conditioned baseline reported. Code: github.com/TriWorldBench/TriWorldBench",
        "org": "Shanghai Jiao Tong University",
        "orgs": [
          "Shanghai Jiao Tong University"
        ],
        "redistribution": null,
        "url": "https://arxiv.org/abs/2609.15726"
      },
      "id": "dataset:bench2dex",
      "name": "Bench2Dex",
      "notes": {
        "_license_conflict": "https://modelscope.cn/api/v1/datasets/Bench2Dex/teleopdata says License: 'Apache License 2.0'; ReadmeContent is ModelScope's default template, which states in Chinese that the contributor supplied no detailed dataset description; root tree is ['dataset', '.gitattributes', '.gitignore', 'README.md'] with no LICENSE file. 51,615 downloads, 472 GB. The sibling assets repo Bench2Dex/Bench2Dex carries ",
        "_license_note": "Correction: the row carried commercial_use true derived from the code licence while its own note said the data licence was unresolved. A code licence is not a data grant, so commercial use of the data is not established.",
        "_org_roles": {
          "Shanghai Jiao Tong University": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-18",
        "method": "refinery-mine",
        "source_hosts": [
          "arxiv.org",
          "github.com",
          "huggingface.co",
          "modelscope.cn"
        ],
        "source_urls": [
          "https://github.com/Bench2Dex/Bench2Dex/blob/main/LICENSE",
          "https://github.com/Bench2Dex/Bench2Dex",
          "https://modelscope.cn/api/v1/datasets/Bench2Dex/teleopdata",
          "https://huggingface.co/api/datasets/Bench2Dex/teleopdata",
          "https://arxiv.org/abs/2609.15726",
          "https://modelscope.cn/datasets/Bench2Dex/teleopdata",
          "https://huggingface.co/datasets/Bench2Dex/teleopdata"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.778,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "conflicted",
          "license-conflict",
          "license-verified"
        ],
        "grade": "B",
        "hosts": 4,
        "independent_hosts": 3,
        "sources": 7,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://arxiv.org/abs/2609.15726"
    },
    {
      "fields": {
        "commercial_use": false,
        "episodes": 55,
        "has_force": true,
        "hours": 19.3,
        "license": "not released",
        "license_family": "unreleased",
        "license_spdx": null,
        "modalities": [
          "tactile"
        ],
        "name": "Beyond Gestures HOM (hand-object manipulation)",
        "org": "Meta Reality Labs Research",
        "orgs": [
          "Meta Reality Labs Research"
        ],
        "redistribution": false,
        "url": "https://arxiv.org/abs/2609.16518"
      },
      "id": "dataset:beyond-gestures-hom-hand-object-manipulation",
      "name": "Beyond Gestures HOM (hand-object manipulation)",
      "notes": {
        "_modalities_unmapped": [
          "OptiTrack hand pose"
        ],
        "_org_roles": {
          "Meta Reality Labs Research": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "primary-source",
        "source_hosts": [
          "arxiv.org"
        ],
        "source_urls": [
          "https://arxiv.org/html/2609.16518v1",
          "https://arxiv.org/abs/2609.16518"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.889,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "modality-unmapped",
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://arxiv.org/abs/2609.16518"
    },
    {
      "fields": {
        "commercial_use": false,
        "episodes": 38,
        "has_force": true,
        "hours": 12.7,
        "license": "not released",
        "license_family": "unreleased",
        "license_spdx": null,
        "modalities": [
          "force"
        ],
        "name": "Beyond Gestures HP + FF (single participant)",
        "org": "Meta Reality Labs Research",
        "orgs": [
          "Meta Reality Labs Research"
        ],
        "redistribution": false,
        "url": "https://arxiv.org/abs/2609.16518"
      },
      "id": "dataset:beyond-gestures-hp-ff-single-participant",
      "name": "Beyond Gestures HP + FF (single participant)",
      "notes": {
        "_modalities_unmapped": [
          "HP has OptiTrack pose only",
          "wrist pressure"
        ],
        "_org_roles": {
          "Meta Reality Labs Research": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "primary-source",
        "source_hosts": [
          "arxiv.org"
        ],
        "source_urls": [
          "https://arxiv.org/html/2609.16518v1",
          "https://arxiv.org/abs/2609.16518"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.889,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "modality-unmapped",
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://arxiv.org/abs/2609.16518"
    },
    {
      "fields": {
        "commercial_use": true,
        "hours": 21.7,
        "license": "CC-BY-4.0",
        "license_family": "permissive",
        "license_spdx": "CC-BY-4.0",
        "modalities": [
          "audio"
        ],
        "name": "BioDCASE 2026 Task 4: ATBFL",
        "notes": "19,633 five-second multilabel segments, 7 Antarctic blue and fin whale call types, 79.8% positive, 11 site-year deployments 2005-2017",
        "org": [
          "BioDCASE",
          "Australian Antarctic Data Centre"
        ],
        "orgs": [
          "BioDCASE",
          "Australian Antarctic Data Centre"
        ],
        "redistribution": true,
        "url": "https://doi.org/10.5281/zenodo.19133112"
      },
      "id": "dataset:biodcase-2026-task-4-atbfl",
      "name": "BioDCASE 2026 Task 4: ATBFL",
      "notes": {
        "_license_note": "Zenodo API for record 19133112 returns license id 'cc-by-4.0' and access_right 'open' for 'BioDCASE 2026 Task 4: ATBFL Dataset' v1. Commercial use permitted with attribution to the creators. The recorded 'public (Zenodo)' was a hosting fact, not a licence.",
        "_license_previously_recorded": "public (Zenodo)",
        "_org_roles": {
          "Australian Antarctic Data Centre": [
            "org"
          ],
          "BioDCASE": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery",
        "source_hosts": [
          "arxiv.org",
          "doi.org",
          "zenodo.org"
        ],
        "source_urls": [
          "https://doi.org/10.5281/zenodo.19133112",
          "https://zenodo.org/records/19133112",
          "https://arxiv.org/abs/2609.15255"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "license-verified"
        ],
        "grade": "B",
        "hosts": 3,
        "independent_hosts": 3,
        "sources": 3,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://doi.org/10.5281/zenodo.19133112"
    },
    {
      "fields": {
        "commercial_use": null,
        "license": "CC-BY-4.0",
        "license_family": "unclear-scope",
        "license_spdx": "CC-BY-4.0",
        "modalities": [
          "audio"
        ],
        "name": "BioDCASE 2026 Task 4: BirdSet subsets (HSN, POW, UHH)",
        "notes": "HSN 12,000 / POW 4,560 / UHH 36,637 five-second segments; 19 / 41 / 25 classes; 0.52 / 2.83 / 1.05 labels per segment; PerchV2 1536-d embeddings supplied instead of audio",
        "org": [
          "BioDCASE",
          "BirdSet"
        ],
        "orgs": [
          "BioDCASE",
          "BirdSet"
        ],
        "redistribution": true,
        "url": "https://doi.org/10.5281/zenodo.19340660"
      },
      "id": "dataset:biodcase-2026-task-4-birdset-subsets-hsn-pow-uhh",
      "name": "BioDCASE 2026 Task 4: BirdSet subsets (HSN, POW, UHH)",
      "notes": {
        "_license_conflict": "https://huggingface.co/datasets/DBD-research-group/BirdSet says Card licence cc-by-nc-4.0; researchers shall use this dataset only for non-commercial research and educational purposes, although each test dataset is licensed under CC BY 4.0; https://zenodo.org/records/7525805 says HSN source soundscapes (Sierra Nevada) are CC BY 4.0; https://zenodo.org/records/4656848 says POW source recordings (Ea",
        "_license_note": "Zenodo 19340660 returns cc-by-4.0 and ships Perch v2 embeddings, not audio. It says it assigns no single new data licence and use stays subject to BirdSet terms. Upstream HSN/UHH are CC BY 4.0 and POW CC0, which allow commercial use, but the BirdSet card adds a blanket non-commercial clause. Unresolved.",
        "_license_previously_recorded": "public (Zenodo)",
        "_org_roles": {
          "BioDCASE": [
            "org"
          ],
          "BirdSet": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery",
        "source_hosts": [
          "arxiv.org",
          "doi.org",
          "huggingface.co",
          "zenodo.org"
        ],
        "source_urls": [
          "https://doi.org/10.5281/zenodo.19340660",
          "https://zenodo.org/records/7525805",
          "https://zenodo.org/api/records/4656848",
          "https://zenodo.org/records/4656848",
          "https://huggingface.co/api/datasets/DBD-research-group/BirdSet",
          "https://zenodo.org/api/records/7525805",
          "https://arxiv.org/abs/2609.15255",
          "https://huggingface.co/datasets/DBD-research-group/BirdSet/raw/main/README.md",
          "https://zenodo.org/api/records/7078499",
          "https://huggingface.co/datasets/DBD-research-group/BirdSet",
          "https://zenodo.org/records/19340660",
          "https://zenodo.org/api/records/19340660",
          "https://zenodo.org/records/7078499"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.556,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "conflicted",
          "license-conflict",
          "license-verified"
        ],
        "grade": "B",
        "hosts": 4,
        "independent_hosts": 4,
        "sources": 13,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://doi.org/10.5281/zenodo.19340660"
    },
    {
      "fields": {
        "commercial_use": true,
        "embodiment": "human",
        "episodes": null,
        "format": "LeRobot/RLDS export",
        "has_force": true,
        "hours": null,
        "license": "commercial",
        "license_family": "commercial",
        "license_spdx": null,
        "modalities": [
          "force",
          "hand-pose",
          "imu",
          "rgb-egocentric"
        ],
        "name": "Blomega GX-1 capture",
        "org": "Blomega",
        "orgs": [
          "Blomega"
        ],
        "redistribution": null,
        "url": "https://blomegalab.com",
        "year": 2026
      },
      "id": "dataset:blomega-gx-1-capture",
      "name": "Blomega GX-1 capture",
      "notes": {
        "_org_roles": {
          "Blomega": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-08-14",
        "method": "import",
        "source_hosts": [
          "blomegalab.com"
        ],
        "source_urls": [
          "https://blomegalab.com"
        ]
      },
      "quality": {
        "age_days": 35,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://blomegalab.com"
    },
    {
      "fields": {
        "commercial_use": true,
        "embodiment": "single-arm",
        "episodes": 60096,
        "format": "RLDS",
        "has_force": false,
        "license": "CC BY 4.0",
        "license_family": "permissive",
        "license_spdx": "CC-BY-4.0",
        "modalities": [
          "proprioception",
          "rgb",
          "text"
        ],
        "name": "BridgeData V2",
        "org": "UC Berkeley",
        "orgs": [
          "UC Berkeley"
        ],
        "redistribution": true,
        "url": "https://rail-berkeley.github.io/bridgedata",
        "year": 2023
      },
      "id": "dataset:bridgedata-v2",
      "name": "BridgeData V2",
      "notes": {
        "_org_roles": {
          "UC Berkeley": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-09-16",
        "method": "refinery-corroborate",
        "source_hosts": [
          "blomega.com",
          "rail-berkeley.github.io",
          "therobotreport.com"
        ],
        "source_urls": [
          "https://blomega.com/guides/licensable-robotics-training-datasets",
          "https://rail-berkeley.github.io/bridgedata",
          "https://www.therobotreport.com/berkeley-released-bridgedata-v2-dataset-for-robot-learning-at-scale"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.778,
        "confidence": "high",
        "corroborated": true,
        "flags": [],
        "grade": "B",
        "hosts": 3,
        "independent_hosts": 3,
        "sources": 3,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://rail-berkeley.github.io/bridgedata"
    },
    {
      "fields": {
        "access_gated": "no",
        "commercial_use": true,
        "downloads_30d": 1371330,
        "license": "ODC-By-1.0",
        "license_family": "permissive",
        "license_spdx": "ODC-By-1.0",
        "name": "c4",
        "note": "Added from the Hugging Face datasets API on 2026-09-17 as one of the most-downloaded corpora whose card tag permits commercial use. 1,371,330 downloads in 30 days, 668 likes, card last modified 2024-01-09. The licence is the uploader's card tag: the repo tree was walked and holds no LICENSE file, so commercial use is NOT established. Size, hours and episodes are null: the API does not state them.",
        "org": "Allen Institute for AI",
        "orgs": [
          "Allen Institute for AI"
        ],
        "redistribution": true,
        "url": "https://huggingface.co/datasets/allenai/c4"
      },
      "id": "dataset:c4",
      "name": "c4",
      "notes": {
        "_license_note": "The repo has no LICENSE file, but the README carries a real grant, not a bare tag: 'We are releasing this dataset under the terms of ODC-BY. By using this, you are also bound by the Common Crawl terms of use in respect of the content contained in the dataset.' ODC-BY permits commercial use of the database and requires attribution. It conveys no rights in the underlying web text, and the Common Cra",
        "_org_roles": {
          "Allen Institute for AI": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-mine",
        "source_hosts": [
          "commoncrawl.org",
          "huggingface.co"
        ],
        "source_urls": [
          "https://huggingface.co/datasets/allenai/c4",
          "https://commoncrawl.org/terms-of-use"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.556,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "license-verified"
        ],
        "grade": "B",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://huggingface.co/datasets/allenai/c4"
    },
    {
      "fields": {
        "all_five_request_types_received": 194943316,
        "collects_biometric": 12,
        "collects_gender_identity": 69,
        "collects_govt_id": 42,
        "collects_minors": 18,
        "collects_precise_geolocation": 115,
        "collects_reproductive_health": 8,
        "columns": 77,
        "commercial_use": true,
        "delete_requests_denied": 890712,
        "delete_requests_received": 58031327,
        "largest_single_filer": "Cuebiq Group LLC, 26285436 delete requests, 45.3% of total",
        "license": "'In general, information presented on this website, unless otherwise indicated, is considered in the public domain. It m...",
        "license_family": "permissive",
        "license_spdx": null,
        "measured": "2026-09-10",
        "name": "California Data Broker Registry 2026 (first-party count, Talika Broker-to-Model Count)",
        "note": "GenAI field is new for 2026 via SB 361; the 2025 and 2024 registries do not contain it, so no year-over-year comparison exists. Request-metric columns are labelled 'in 2024' in the published CSV and are reproduced as published.",
        "optout_requests_received": 98815966,
        "org": "California Privacy Protection Agency",
        "orgs": [
          "California Privacy Protection Agency"
        ],
        "page_last_updated": "2026-07-29",
        "payment_fields_in_form": 0,
        "redistribution": true,
        "registered_brokers": 603,
        "sold_to_federal_government": 55,
        "sold_to_foreign_actor": 26,
        "sold_to_genai_developer": 32,
        "sold_to_law_enforcement": 28,
        "sold_to_other_state_governments": 54,
        "url": "https://cppa.ca.gov/data_broker_registry/registry.csv"
      },
      "id": "dataset:california-data-broker-registry-2026-first-party-count-talika-broker-to-model-count",
      "name": "California Data Broker Registry 2026 (first-party count, Talika Broker-to-Model Count)",
      "notes": {
        "_license_note": "Not an SPDX licence: it is a US state public-domain assertion, reached from the 'Conditions of Use' link in the footer of cppa.ca.gov/data_broker_registry, which is the governing terms page for that site. Public-domain records may be used commercially and redistributed. Two caveats worth carrying in the registry: (1) the clause is 'unless otherwise indicated' and expressly excludes third-party cop",
        "_org_roles": {
          "California Privacy Protection Agency": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "unknown",
        "first_seen": "2026-09-10",
        "last_seen": "2026-09-16",
        "method": "first-party-measurement",
        "source_hosts": [
          "ca.gov",
          "cppa.ca.gov"
        ],
        "source_urls": [
          "https://cppa.ca.gov/data_broker_registry/registry.csv",
          "https://www.ca.gov/legal/conditions-of-use"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.444,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "license-verified",
          "sparse"
        ],
        "grade": "B",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://cppa.ca.gov/data_broker_registry/registry.csv"
    },
    {
      "fields": {
        "access_gated": "no",
        "commercial_use": null,
        "downloads_30d": 584516,
        "license": "CC-BY-4.0",
        "license_family": "permissive",
        "license_spdx": "CC-BY-4.0",
        "name": "dclm-baseline-1.0",
        "note": "Added from the Hugging Face datasets API on 2026-09-17 as one of the most-downloaded corpora whose card tag permits commercial use. 584,516 downloads in 30 days, 314 likes, card last modified 2024-07-22. The licence is the uploader's card tag: the repo tree was walked and holds no LICENSE file, so commercial use is NOT established. Size, hours and episodes are null: the API does not state them.",
        "org": "ML Foundations",
        "orgs": [
          "ML Foundations"
        ],
        "redistribution": true,
        "url": "https://huggingface.co/datasets/mlfoundations/dclm-baseline-1.0"
      },
      "id": "dataset:dclm-baseline-1-0",
      "name": "dclm-baseline-1.0",
      "notes": {
        "_license_conflict": "https://huggingface.co/api/datasets/mlfoundations/dclm-baseline-1.0 says cardData.license cc-by-4.0, tag license:cc-by-4.0, lastModified 2024-07-22T15:27:52Z, gated false, 27,840 siblings of which the only markdown or licence-shaped file is README.md. No LICENSE file has been added.; https://github.com/mlfoundations/dclm says MIT, on the construction-code repository, which covers the pipeline and ",
        "_license_note": "UNCHANGED. Re-checked live: lastModified is still 2024-07-22T15:27:52Z, identical to round 5, and the full 27,840-entry sibling list still contains no LICENSE. One card still says two things. Commercial use stays null and the conflict stands.",
        "_license_previously_recorded": "Card tag 'License: CC-by-4.0' with, in the same card, 'the dataset is intended for research use only'. No LICENSE file i...",
        "_org_roles": {
          "ML Foundations": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-mine",
        "source_hosts": [
          "api.github.com",
          "commoncrawl.org",
          "github.com",
          "huggingface.co"
        ],
        "source_urls": [
          "https://github.com/mlfoundations/dclm",
          "https://commoncrawl.org/terms-of-use",
          "https://huggingface.co/datasets/mlfoundations/dclm-baseline-1.0",
          "https://api.github.com/repos/mlfoundations/dclm/license",
          "https://huggingface.co/datasets/mlfoundations/dclm-baseline-1.0/blob/main/README.md",
          "https://huggingface.co/datasets/mlfoundations/dclm-baseline-1.0/raw/main/README.md",
          "https://huggingface.co/api/datasets/mlfoundations/dclm-baseline-1.0"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.556,
        "confidence": "medium",
        "corroborated": true,
        "flags": [
          "conflicted",
          "license-conflict",
          "license-needs-review",
          "license-verified"
        ],
        "grade": "B",
        "hosts": 4,
        "independent_hosts": 4,
        "sources": 7,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://huggingface.co/datasets/mlfoundations/dclm-baseline-1.0"
    },
    {
      "fields": {
        "commercial_use": true,
        "embodiment": "human",
        "episodes": null,
        "format": "custom",
        "has_force": false,
        "hours": null,
        "license": "CC-BY-4.0",
        "license_family": "permissive",
        "license_spdx": "CC-BY-4.0",
        "modalities": [
          "hand-pose",
          "rgb"
        ],
        "name": "DexCap",
        "org": "Stanford",
        "orgs": [
          "Stanford"
        ],
        "redistribution": true,
        "url": "https://dex-cap.github.io/",
        "year": 2024
      },
      "id": "dataset:dexcap",
      "name": "DexCap",
      "notes": {
        "_license_note": "The DATA is CC BY 4.0 per the official HuggingFace dataset card (linked from the project page). The CODE repository github.com/j96w/DexCap is separately MIT (Copyright (c) 2024 Chen Wang). These govern different artifacts, so this is not a conflict. Commercial use is permitted with attribution. The previously recorded string 'open' was too vague to act on.",
        "_org_roles": {
          "Stanford": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-09-16",
        "method": "refinery-mine",
        "source_hosts": [
          "dex-cap.github.io",
          "huggingface.co"
        ],
        "source_urls": [
          "https://huggingface.co/datasets/chenwangj/DexCap-Data/raw/main/README.md",
          "https://dex-cap.github.io/"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "license-verified"
        ],
        "grade": "B",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://dex-cap.github.io/"
    },
    {
      "fields": {
        "commercial_use": true,
        "license": "OTS (healthcare DICOM paired with clinical reports)",
        "license_family": "ots",
        "license_spdx": null,
        "modalities": [
          "rgb",
          "text"
        ],
        "name": "DICOM Medical Imaging Dataset with Clinical Reports",
        "org": "BLOMEGA",
        "orgs": [
          "BLOMEGA"
        ],
        "redistribution": null,
        "url": "https://blomega.com/explore-ots-datasets"
      },
      "id": "dataset:dicom-medical-imaging-dataset-with-clinical-reports",
      "name": "DICOM Medical Imaging Dataset with Clinical Reports",
      "notes": {
        "_org_roles": {
          "BLOMEGA": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-08-14",
        "method": "import",
        "source_hosts": [
          "blomega.com"
        ],
        "source_urls": [
          "https://blomega.com/datasets.json",
          "https://blomega.com/explore-ots-datasets"
        ]
      },
      "quality": {
        "age_days": 35,
        "completeness": 0.556,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "C",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://blomega.com/explore-ots-datasets"
    },
    {
      "fields": {
        "commercial_use": true,
        "episodes": 7470,
        "license": "OTS (booking emails: confirmation/cancellation/modification)",
        "license_family": "ots",
        "license_spdx": null,
        "modalities": [
          "text"
        ],
        "name": "Doctor Appointment Emails",
        "org": "BLOMEGA",
        "orgs": [
          "BLOMEGA"
        ],
        "redistribution": null,
        "url": "https://blomega.com/explore-ots-datasets"
      },
      "id": "dataset:doctor-appointment-emails",
      "name": "Doctor Appointment Emails",
      "notes": {
        "_org_roles": {
          "BLOMEGA": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-08-14",
        "method": "import",
        "source_hosts": [
          "blomega.com"
        ],
        "source_urls": [
          "https://blomega.com/datasets.json",
          "https://blomega.com/explore-ots-datasets"
        ]
      },
      "quality": {
        "age_days": 35,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://blomega.com/explore-ots-datasets"
    },
    {
      "fields": {
        "commercial_use": true,
        "episodes": 408000,
        "license": "OTS (408K clips)",
        "license_family": "ots",
        "license_spdx": null,
        "modalities": [
          "rgb"
        ],
        "name": "Domestic Animal Behavior",
        "org": "BLOMEGA",
        "orgs": [
          "BLOMEGA"
        ],
        "redistribution": null,
        "url": "https://blomega.com/explore-ots-datasets"
      },
      "id": "dataset:domestic-animal-behavior",
      "name": "Domestic Animal Behavior",
      "notes": {
        "_org_roles": {
          "BLOMEGA": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-08-14",
        "method": "import",
        "source_hosts": [
          "blomega.com"
        ],
        "source_urls": [
          "https://blomega.com/datasets.json",
          "https://blomega.com/explore-ots-datasets"
        ]
      },
      "quality": {
        "age_days": 35,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://blomega.com/explore-ots-datasets"
    },
    {
      "fields": {
        "commercial_use": true,
        "embodiment": "single-arm",
        "episodes": 76000,
        "format": "RLDS",
        "has_force": false,
        "hours": 350,
        "license": "CC-BY-4.0",
        "license_family": "permissive",
        "license_spdx": "CC-BY-4.0",
        "modalities": [
          "proprioception",
          "rgb-stereo",
          "text"
        ],
        "name": "DROID",
        "org": "Stanford",
        "orgs": [
          "Stanford"
        ],
        "redistribution": true,
        "url": "https://droid-dataset.github.io/",
        "year": 2024
      },
      "id": "dataset:droid",
      "name": "DROID",
      "notes": {
        "_license_note": "Two independent sources agree, so no conflict. The HuggingFace dataset card linked from the official project page (maintained by DROID co-lead Karl Pertsch) states cc-by-4.0, and the paper states the same in two places (https://arxiv.org/html/2403.12945v2). The DATA is therefore CC BY 4.0: commercial use and redistribution permitted with attribution. The previously recorded 'MIT / CC (per subset)'",
        "_license_previously_recorded": "MIT / CC (per subset)",
        "_org_roles": {
          "Stanford": [
            "org"
          ]
        },
        "_orgs_dropped": [
          "consortium"
        ]
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-09-16",
        "method": "import",
        "source_hosts": [
          "blomega.com",
          "droid-dataset.github.io",
          "emergentmind.com",
          "huggingface.co"
        ],
        "source_urls": [
          "https://blomega.com/guides/licensable-robotics-training-datasets",
          "https://droid-dataset.github.io/",
          "https://www.emergentmind.com/topics/droid-a-large-scale-in-the-wild-robot-manipulation-dataset",
          "https://huggingface.co/datasets/KarlP/droid/raw/main/README.md"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.889,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "license-needs-review",
          "license-verified"
        ],
        "grade": "A",
        "hosts": 4,
        "independent_hosts": 3,
        "sources": 4,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://droid-dataset.github.io/"
    },
    {
      "fields": {
        "commercial_use": false,
        "embodiment": "human",
        "episodes": null,
        "format": "proprietary",
        "has_force": false,
        "hours": 1000000,
        "license": "proprietary (no public release)",
        "license_family": "unreleased",
        "license_spdx": null,
        "modalities": [
          "rgb-egocentric"
        ],
        "name": "DYNA-2 corpus",
        "org": "Dyna Robotics",
        "orgs": [
          "Dyna Robotics"
        ],
        "redistribution": false,
        "url": "https://www.dyna.co/dyna-2",
        "year": 2026
      },
      "id": "dataset:dyna-2-corpus",
      "name": "DYNA-2 corpus",
      "notes": {
        "_org_roles": {
          "Dyna Robotics": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-08-14",
        "method": "import",
        "source_hosts": [
          "dyna.co"
        ],
        "source_urls": [
          "https://www.dyna.co/dyna-2"
        ]
      },
      "quality": {
        "age_days": 35,
        "completeness": 0.778,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://www.dyna.co/dyna-2"
    },
    {
      "fields": {
        "commercial_use": true,
        "embodiment": "human",
        "format": "custom",
        "has_force": false,
        "hours": 1286,
        "license": "Bespoke Ego-Exo4D licence agreement, 12 licensors: commercial product development allowed for enumerated purposes; no su...",
        "license_family": "commercial",
        "license_spdx": null,
        "modalities": [
          "gaze",
          "imu",
          "point-cloud",
          "rgb-egocentric",
          "rgb-exocentric"
        ],
        "name": "Ego-Exo4D",
        "org": "Meta",
        "orgs": [
          "Meta"
        ],
        "redistribution": false,
        "url": "https://arxiv.org/abs/2311.18259",
        "year": 2023
      },
      "id": "dataset:ego-exo4d",
      "name": "Ego-Exo4D",
      "notes": {
        "_license_note": "Supersedes the earlier null. The published 'DRAFT FOR REVIEW' PDF holds 12 agreements (CMU, Meta, Georgia Tech, IIIT-H, Indiana, Los Andes, UNC, UPenn, NUS, SFU, Minnesota, Tokyo); all 12 permit uses 2(a)-2(c) 'for academic research, commercial or noncommercial product development'. Data may not appear in any product; no sublicense or transfer. Signed version is gated at ego4d.dev.",
        "_org_roles": {
          "Meta": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-09-17",
        "method": "refinery-mine",
        "source_hosts": [
          "arxiv.org",
          "blomega.com",
          "docs.ego-exo4d-data.org",
          "ego4d-data.org",
          "emergentmind.com"
        ],
        "source_urls": [
          "https://www.emergentmind.com/topics/egoexo4d-dataset",
          "https://blomega.com/guides/licensable-robotics-training-datasets",
          "https://arxiv.org/abs/2311.18259",
          "https://ego4d-data.org/pdfs/Ego-Exo4D-Model-License.pdf",
          "https://docs.ego-exo4d-data.org/getting-started"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.778,
        "confidence": "medium",
        "corroborated": true,
        "flags": [
          "license-verified"
        ],
        "grade": "B",
        "hosts": 5,
        "independent_hosts": 4,
        "sources": 5,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://arxiv.org/abs/2311.18259"
    },
    {
      "fields": {
        "commercial_use": false,
        "episodes": null,
        "has_force": false,
        "hours": 18561,
        "license": "not released (project page code link is '#')",
        "license_family": "unreleased",
        "license_spdx": null,
        "modalities": [
          "end-effector-pose",
          "gripper-state",
          "rgb-rendered"
        ],
        "morphologies": 15,
        "name": "Ego2Robot synthesized robot data",
        "org": [
          "Qwen Team",
          "Renmin University AIM3",
          "ShanghaiTech",
          "BIGAI"
        ],
        "orgs": [
          "Qwen Team",
          "Renmin University AIM3",
          "ShanghaiTech",
          "BIGAI"
        ],
        "redistribution": false,
        "robot_data_hours": 6565,
        "robotwin_main_1to1_vs_robot_only": {
          "clean": [
            62.2,
            68.1
          ],
          "ebench": [
            39.6,
            49.8
          ],
          "embody": [
            23.8,
            27.2
          ],
          "franka_perturb": [
            7,
            5.3
          ],
          "rand": [
            50.9,
            53.5
          ]
        },
        "robotwin_randomized_ego_only_ablation": {
          "ego2r_15_morph": 33.5,
          "ego2r_15_plus_raw": 37.3,
          "ego2r_1_morph": 31.7,
          "raw_ego": 28.1
        },
        "source_ego_hours": {
          "ANT": 7,
          "EgoDex": 732,
          "EgoVerse": 954,
          "ViTRA": 249,
          "total": 1940
        },
        "url": "https://arxiv.org/abs/2608.02580"
      },
      "id": "dataset:ego2robot-synthesized-robot-data",
      "name": "Ego2Robot synthesized robot data",
      "notes": {
        "_org_roles": {
          "BIGAI": [
            "org"
          ],
          "Qwen Team": [
            "org"
          ],
          "Renmin University AIM3": [
            "org"
          ],
          "ShanghaiTech": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-15",
        "last_seen": "2026-09-15",
        "method": "refinery",
        "source_hosts": [
          "arxiv.org"
        ],
        "source_urls": [
          "https://arxiv.org/html/2608.02580",
          "https://arxiv.org/abs/2608.02580"
        ]
      },
      "quality": {
        "age_days": 3,
        "completeness": 0.778,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://arxiv.org/abs/2608.02580"
    },
    {
      "fields": {
        "commercial_use": true,
        "embodiment": "human",
        "format": "custom",
        "has_force": false,
        "hours": 3670,
        "license": "Ego4D license",
        "license_family": "commercial-restricted",
        "license_spdx": null,
        "modalities": [
          "audio",
          "gaze",
          "imu",
          "rgb"
        ],
        "name": "Ego4D",
        "org": "Meta",
        "orgs": [
          "Meta"
        ],
        "redistribution": false,
        "url": "https://arxiv.org/abs/2110.07058",
        "year": 2021
      },
      "id": "dataset:ego4d",
      "name": "Ego4D",
      "notes": {
        "_license_note": "Bespoke agreement, not an SPDX licence, and access is gated: it must be signed at ego4d.dev and credentials arrive in about 48 hours. The 36-page document bundles 12 separate agreements, one per partner institution (University of Bristol, Carnegie Mellon, Georgia Tech Research Corporation, IIIT, Indiana University, KAUST, Universidad de los Andes, National University of Singapore, Regents of the U",
        "_org_roles": {
          "Meta": [
            "org"
          ]
        },
        "_orgs_dropped": [
          "consortium"
        ]
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-09-16",
        "method": "import",
        "source_hosts": [
          "arxiv.org",
          "blomega.com",
          "ego4d-data.org"
        ],
        "source_urls": [
          "https://ego4d-data.org/pdfs/Ego4D-Licenses-Draft.pdf",
          "https://arxiv.org/abs/2110.07058",
          "https://blomega.com/guides/licensable-robotics-training-datasets"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.778,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "license-verified"
        ],
        "grade": "B",
        "hosts": 3,
        "independent_hosts": 3,
        "sources": 3,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://arxiv.org/abs/2110.07058"
    },
    {
      "fields": {
        "commercial_use": null,
        "embodiment": "human",
        "episodes": null,
        "format": "custom",
        "has_force": false,
        "hours": 20854,
        "license": "No dataset licence published: the paper states no terms and offers no download, no repository or card exists for it, and...",
        "license_family": "unknown",
        "license_note": "arXiv 2602.16710v1, 18 Feb 2026, NVIDIA with UC Berkeley, UT Austin, Stanford and Georgia Tech. Paper licence CC BY 4.0 covers the paper, not the 20,854 hours of egocentric video. The full text names no project page, repository or host, and a Hugging Face search returns no EgoScale dataset. NVIDIA commercially licenses a model pretrained on it while publishing no terms for the data, which is a gap",
        "license_spdx": null,
        "modalities": [
          "hand-pose",
          "rgb"
        ],
        "name": "EgoScale",
        "org": "UT Austin RPL",
        "orgs": [
          "UT Austin RPL"
        ],
        "redistribution": null,
        "url": "https://arxiv.org/abs/2602.16710",
        "year": 2026
      },
      "id": "dataset:egoscale",
      "name": "EgoScale",
      "notes": {
        "_license_note": "source hedges on the licence; verify at primary source",
        "_org_roles": {
          "UT Austin RPL": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-09-17",
        "method": "refinery-mine",
        "source_hosts": [
          "arxiv.org",
          "github.com"
        ],
        "source_urls": [
          "https://github.com/NVIDIA/Isaac-GR00T",
          "https://arxiv.org/abs/2602.16710",
          "https://arxiv.org/html/2602.16710v1"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.778,
        "confidence": "medium",
        "corroborated": true,
        "flags": [
          "license-needs-review"
        ],
        "grade": "B",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 3,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://arxiv.org/abs/2602.16710"
    },
    {
      "fields": {
        "commercial_use": true,
        "embodiment": "human",
        "episodes": 80000,
        "format": "custom",
        "has_force": false,
        "hours": 1362,
        "license": "MIT",
        "license_family": "permissive",
        "license_spdx": "MIT",
        "modalities": [
          "hand-pose",
          "head-pose",
          "rgb"
        ],
        "name": "EgoVerse",
        "org": [
          "Georgia Tech",
          "Stanford",
          "UC San Diego",
          "ETH",
          "Massachusetts Institute of Technology",
          "Meta",
          "Scale"
        ],
        "orgs": [
          "Georgia Tech",
          "Stanford",
          "UC San Diego",
          "ETH",
          "Massachusetts Institute of Technology",
          "Meta",
          "Scale"
        ],
        "redistribution": true,
        "url": "https://arxiv.org/abs/2604.07607",
        "year": 2026
      },
      "id": "dataset:egoverse",
      "name": "EgoVerse",
      "notes": {
        "_license_conflict": "https://github.com/GaTech-RL2/EgoVerse/blob/main/LICENSE says MIT License, Copyright (c) 2025 EgoMimic-Team. This is the repository-root LICENSE file and on its face covers the repository.; https://raw.githubusercontent.com/GaTech-RL2/EgoVerse/main/CONTRIBUTING_DATA.md says The Resources table in the consortium's own data-contribution document lists 'License | CC BY-SA 4.0' alongside the website, ",
        "_license_note": "Two licences are stated inside the same repository and no winner is picked here. The repo-root LICENSE says MIT; CONTRIBUTING_DATA.md says CC BY-SA 4.0 for the project's resources. The README has no licence section at all. Both MIT and CC BY-SA 4.0 permit commercial use and redistribution, so the commercial verdict is safe either way, BUT the obligations differ materially: CC BY-SA 4.0 is copyleft",
        "_license_previously_recorded": "Repository LICENSE file: 'MIT License. Copyright (c) 2025 EgoMimic-Team'. Repository data documentation: '| License | CC...",
        "_org_roles": {
          "ETH": [
            "org"
          ],
          "Georgia Tech": [
            "org"
          ],
          "Massachusetts Institute of Technology": [
            "org"
          ],
          "Meta": [
            "org"
          ],
          "Scale": [
            "org"
          ],
          "Stanford": [
            "org"
          ],
          "UC San Diego": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-09-16",
        "method": "import",
        "source_hosts": [
          "arxiv.org",
          "github.com",
          "raw.githubusercontent.com"
        ],
        "source_urls": [
          "https://raw.githubusercontent.com/GaTech-RL2/EgoVerse/main/CONTRIBUTING_DATA.md",
          "https://arxiv.org/abs/2604.07607",
          "https://github.com/GaTech-RL2/EgoVerse/blob/main/LICENSE"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.889,
        "confidence": "medium",
        "corroborated": true,
        "flags": [
          "conflicted",
          "license-conflict",
          "license-verified"
        ],
        "grade": "B",
        "hosts": 3,
        "independent_hosts": 3,
        "sources": 3,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://arxiv.org/abs/2604.07607"
    },
    {
      "fields": {
        "commercial_use": true,
        "hours": 8500,
        "license": "OTS, en",
        "license_family": "ots",
        "license_spdx": null,
        "modalities": [
          "rgb"
        ],
        "name": "English Animation Videos",
        "org": "BLOMEGA",
        "orgs": [
          "BLOMEGA"
        ],
        "redistribution": null,
        "url": "https://blomega.com/explore-ots-datasets"
      },
      "id": "dataset:english-animation-videos",
      "name": "English Animation Videos",
      "notes": {
        "_org_roles": {
          "BLOMEGA": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-08-14",
        "method": "import",
        "source_hosts": [
          "blomega.com"
        ],
        "source_urls": [
          "https://blomega.com/datasets.json",
          "https://blomega.com/explore-ots-datasets"
        ]
      },
      "quality": {
        "age_days": 35,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://blomega.com/explore-ots-datasets"
    },
    {
      "fields": {
        "commercial_use": true,
        "episodes": 22500000,
        "license": "ODbL-1.0",
        "license_family": "permissive",
        "license_spdx": "ODbL-1.0",
        "modalities": [
          "text"
        ],
        "name": "English-French Translation Pairs",
        "org": "BLOMEGA",
        "orgs": [
          "BLOMEGA"
        ],
        "redistribution": true,
        "url": "https://blomega.com/explore-ots-datasets"
      },
      "id": "dataset:english-french-translation-pairs",
      "name": "English-French Translation Pairs",
      "notes": {
        "_license_note": "Listing text: 'Parallel EN<->FR corpus for NMT training (Tatoeba + OPUS). 22.5M sentence pairs', price FREE, no licence. Tatoeba downloads page: 'These files are released under CC BY 2.0 FR'. OPUS publishes no blanket licence; each of its corpora carries its own. Commercial use cannot be settled until the OPUS subsets used are named. 'Free' is a price, not a licence.",
        "_license_previously_recorded": "free (isAccessibleForFree), 22.5M sentence pairs, en-fr",
        "_org_roles": {
          "BLOMEGA": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-09-18",
        "method": "import",
        "source_hosts": [
          "blomega.com",
          "kaggle.com",
          "tatoeba.org"
        ],
        "source_urls": [
          "https://www.kaggle.com/datasets/dhruvildave/en-fr-translation-dataset",
          "https://tatoeba.org/en/downloads",
          "https://blomega.com/explore-ots-datasets",
          "https://blomega.com/datasets.json"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "license-verified"
        ],
        "grade": "B",
        "hosts": 3,
        "independent_hosts": 3,
        "sources": 4,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://blomega.com/explore-ots-datasets"
    },
    {
      "fields": {
        "commercial_use": false,
        "embodiment": "human",
        "episodes": null,
        "format": "custom",
        "has_force": false,
        "hours": 100,
        "license": "CC BY-NC",
        "license_family": "noncommercial",
        "license_spdx": "CC-BY-NC-4.0",
        "modalities": [
          "audio",
          "rgb"
        ],
        "name": "EPIC-KITCHENS",
        "org": "University of Bristol",
        "orgs": [
          "University of Bristol"
        ],
        "redistribution": true,
        "url": "https://epic-kitchens.github.io/",
        "year": 2018
      },
      "id": "dataset:epic-kitchens",
      "name": "EPIC-KITCHENS",
      "notes": {
        "_org_roles": {
          "University of Bristol": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-08-14",
        "method": "import",
        "source_hosts": [
          "epic-kitchens.github.io"
        ],
        "source_urls": [
          "https://epic-kitchens.github.io/"
        ]
      },
      "quality": {
        "age_days": 35,
        "completeness": 0.778,
        "confidence": "medium",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://epic-kitchens.github.io/"
    },
    {
      "fields": {
        "commercial_use": true,
        "episodes": 2000000,
        "license": "free; 2M pairs across 21 EU languages",
        "license_family": "permissive",
        "license_spdx": null,
        "modalities": [
          "text"
        ],
        "name": "Europarl Parallel Corpus",
        "org": "BLOMEGA",
        "orgs": [
          "BLOMEGA"
        ],
        "redistribution": true,
        "url": "https://blomega.com/explore-ots-datasets"
      },
      "id": "dataset:europarl-parallel-corpus",
      "name": "Europarl Parallel Corpus",
      "notes": {
        "_license_note": "The corpus page (statmt.org) does not address commercial use, saying only 'We are not aware of any copyright restrictions of the material.' The rights holder's legal notice authorises reuse of EP text 'for personal use or for further non-commercial or commercial dissemination, provided that the entire item is reproduced and the source is acknowledged'; partial reproductions must cite the URL.",
        "_org_roles": {
          "BLOMEGA": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-09-17",
        "method": "import",
        "source_hosts": [
          "blomega.com",
          "europarl.europa.eu",
          "statmt.org"
        ],
        "source_urls": [
          "https://www.statmt.org/europarl",
          "https://www.europarl.europa.eu/legal-notice/en",
          "https://blomega.com/datasets.json",
          "https://blomega.com/explore-ots-datasets"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "license-verified"
        ],
        "grade": "B",
        "hosts": 3,
        "independent_hosts": 3,
        "sources": 4,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://blomega.com/explore-ots-datasets"
    },
    {
      "fields": {
        "commercial_use": true,
        "license": "Apache-2.0",
        "license_family": "permissive",
        "license_spdx": "Apache-2.0",
        "modalities": [
          "text"
        ],
        "name": "FarsTail",
        "note": "10,367 Persian NLI examples. Web-normalized annotation density 0.60, the only one of four measured Persian tasks below the web-proportional baseline.",
        "org": "Iranian NLP community",
        "orgs": [
          "Iranian NLP community"
        ],
        "redistribution": true,
        "url": "https://arxiv.org/abs/2608.24698"
      },
      "id": "dataset:farstail",
      "name": "FarsTail",
      "notes": {
        "_license_note": "The URL previously recorded for this dataset (arXiv:2608.24698) is wrong: that paper is 'The Annotation Bottleneck in Persian Text NLP', a 2026 survey that reviews 34 Persian resources, not FarsTail's own source. The primary source is the FarsTail repository from the Data Mining Lab, University of Qom, whose root LICENSE is Apache-2.0 and whose repository contains the dataset itself in a data/ dir",
        "_org_roles": {
          "Iranian NLP community": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery",
        "source_hosts": [
          "arxiv.org",
          "github.com"
        ],
        "source_urls": [
          "https://arxiv.org/abs/2608.24698",
          "https://github.com/dml-qom/FarsTail/blob/master/LICENSE"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.556,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "license-verified"
        ],
        "grade": "B",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://arxiv.org/abs/2608.24698"
    },
    {
      "fields": {
        "access_gated": "no",
        "commercial_use": null,
        "downloads_30d": 3030510,
        "license": "Apache-2.0",
        "license_family": "permissive",
        "license_spdx": "Apache-2.0",
        "name": "FineFineWeb",
        "note": "Added from the Hugging Face datasets API on 2026-09-17 as one of the most-downloaded corpora whose card tag permits commercial use. 3,030,510 downloads in 30 days, 189 likes, card last modified 2024-12-19. The licence is the uploader's card tag: the repo tree was walked and holds no LICENSE file, so commercial use is NOT established. Size, hours and episodes are null: the API does not state them.",
        "org": "M-A-P",
        "orgs": [
          "M-A-P"
        ],
        "redistribution": null,
        "url": "https://huggingface.co/datasets/m-a-p/FineFineWeb"
      },
      "id": "dataset:finefineweb",
      "name": "FineFineWeb",
      "notes": {
        "_license_conflict": "https://huggingface.co/api/datasets/m-a-p/FineFineWeb says cardData.license apache-2.0, tag license:apache-2.0, lastModified 2024-12-19T11:34:03Z, gated false, 66,109 siblings with no LICENSE file.; https://huggingface.co/datasets/HuggingFaceFW/fineweb/raw/main/README.md says The upstream corpus is released under ODC-By v1.0 and is additionally subject to the Common Crawl Terms of Use.",
        "_license_note": "UNCHANGED. Re-checked live: lastModified still 2024-12-19T11:34:03Z, no LICENSE file among 66,109 files. A card tag alone never establishes commercial use, and a derivative of an ODC-By corpus cannot shed the attribution obligation by being relabelled Apache-2.0. Commercial use stays null and the conflict stands.",
        "_license_previously_recorded": "Card tag apache-2.0 only. No LICENSE file, no Licensing Information section, no written grant by any party with rights i...",
        "_org_roles": {
          "M-A-P": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-mine",
        "source_hosts": [
          "commoncrawl.org",
          "huggingface.co"
        ],
        "source_urls": [
          "https://huggingface.co/datasets/m-a-p/FineFineWeb",
          "https://huggingface.co/api/datasets/m-a-p/FineFineWeb",
          "https://huggingface.co/datasets/HuggingFaceFW/fineweb/raw/main/README.md",
          "https://commoncrawl.org/terms-of-use",
          "https://huggingface.co/datasets/HuggingFaceFW/fineweb"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.556,
        "confidence": "medium",
        "corroborated": true,
        "flags": [
          "conflicted",
          "license-conflict",
          "license-needs-review",
          "license-verified"
        ],
        "grade": "B",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 5,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://huggingface.co/datasets/m-a-p/FineFineWeb"
    },
    {
      "fields": {
        "commercial_use": true,
        "hours": 5000,
        "license": "OTS, en",
        "license_family": "ots",
        "license_spdx": null,
        "modalities": [
          "rgb"
        ],
        "name": "Food Videos",
        "org": "BLOMEGA",
        "orgs": [
          "BLOMEGA"
        ],
        "redistribution": null,
        "url": "https://blomega.com/explore-ots-datasets"
      },
      "id": "dataset:food-videos",
      "name": "Food Videos",
      "notes": {
        "_org_roles": {
          "BLOMEGA": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-08-14",
        "method": "import",
        "source_hosts": [
          "blomega.com"
        ],
        "source_urls": [
          "https://blomega.com/datasets.json",
          "https://blomega.com/explore-ots-datasets"
        ]
      },
      "quality": {
        "age_days": 35,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://blomega.com/explore-ots-datasets"
    },
    {
      "fields": {
        "commercial_use": false,
        "episodes": null,
        "has_force": true,
        "hours": 10,
        "license": "unreleased, listed as under preparation",
        "license_family": "unreleased",
        "license_spdx": null,
        "modalities": [
          "force",
          "imu",
          "rgb",
          "semg"
        ],
        "name": "ForceBand EMG2Force pretraining dataset",
        "notes": "4 users. Repo ships code, BOM, checkpoints and an eval slice (data/demos, data/splits, data/val); the full corpus is an unchecked roadmap item as of 2026-09-09.",
        "org": [
          "Amazon FAR",
          "University of Maryland",
          "Johns Hopkins"
        ],
        "orgs": [
          "Amazon FAR",
          "University of Maryland",
          "Johns Hopkins"
        ],
        "redistribution": false,
        "url": "https://github.com/Bottle101/ForceBand"
      },
      "id": "dataset:forceband-emg2force-pretraining-dataset",
      "name": "ForceBand EMG2Force pretraining dataset",
      "notes": {
        "_org_roles": {
          "Amazon FAR": [
            "org"
          ],
          "Johns Hopkins": [
            "org"
          ],
          "University of Maryland": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-09",
        "last_seen": "2026-09-09",
        "method": "refinery",
        "source_hosts": [
          "github.com"
        ],
        "source_urls": [
          "https://github.com/Bottle101/ForceBand"
        ]
      },
      "quality": {
        "age_days": 9,
        "completeness": 0.778,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://github.com/Bottle101/ForceBand"
    },
    {
      "fields": {
        "commercial_use": true,
        "license": "MIT",
        "license_family": "permissive",
        "license_spdx": "MIT",
        "modalities": [
          "text"
        ],
        "name": "GAPA (Gender Associations of Physical Attributes)",
        "notes": "316 physical attributes, 14,706 gender-association ratings, 304 US annotators, 7-point Likert, cross-classified; 53% of attributes significantly gender-associated; released predictor olmo2-7b-base r=0.764",
        "org": "University of California, Los Angeles",
        "orgs": [
          "University of California, Los Angeles"
        ],
        "redistribution": true,
        "url": "https://github.com/Yingjia-Wan/GAPA"
      },
      "id": "dataset:gapa-gender-associations-of-physical-attributes",
      "name": "GAPA (Gender Associations of Physical Attributes)",
      "notes": {
        "_license_note": "LICENSE says the MIT grant covers both code and data, including the gender-association ratings; the HF card agrees (mit). GitHub shows NOASSERTION only because of added text. Carve-outs: LitBank extracts stay CC BY 4.0, quoted in-copyright novel sentences grant no rights, predictor model is Apache-2.0.",
        "_license_previously_recorded": "public (GitHub)",
        "_org_roles": {
          "University of California, Los Angeles": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery",
        "source_hosts": [
          "arxiv.org",
          "github.com",
          "huggingface.co"
        ],
        "source_urls": [
          "https://arxiv.org/abs/2609.16366",
          "https://github.com/Yingjia-Wan/GAPA",
          "https://github.com/Yingjia-Wan/GAPA/blob/main/LICENSE",
          "https://huggingface.co/datasets/alisa-yingjia-wan/gapa"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.556,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "license-verified"
        ],
        "grade": "B",
        "hosts": 3,
        "independent_hosts": 3,
        "sources": 4,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://github.com/Yingjia-Wan/GAPA"
    },
    {
      "fields": {
        "commercial_use": true,
        "episodes": 5800,
        "license": "OTS (5.8K thermographic images for predictive maintenance)",
        "license_family": "ots",
        "license_spdx": null,
        "modalities": [
          "rgb"
        ],
        "name": "Industrial Electric Motor Thermography",
        "org": "BLOMEGA",
        "orgs": [
          "BLOMEGA"
        ],
        "redistribution": null,
        "url": "https://blomega.com/explore-ots-datasets"
      },
      "id": "dataset:industrial-electric-motor-thermography",
      "name": "Industrial Electric Motor Thermography",
      "notes": {
        "_org_roles": {
          "BLOMEGA": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-08-14",
        "method": "import",
        "source_hosts": [
          "blomega.com"
        ],
        "source_urls": [
          "https://blomega.com/datasets.json",
          "https://blomega.com/explore-ots-datasets"
        ]
      },
      "quality": {
        "age_days": 35,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://blomega.com/explore-ots-datasets"
    },
    {
      "fields": {
        "access_gated": "no",
        "commercial_use": null,
        "downloads_30d": 723810,
        "license": "Apache-2.0",
        "license_family": "permissive",
        "license_spdx": "Apache-2.0",
        "name": "LLaVA-OneVision-1.5-Mid-Training-85M",
        "note": "Added from the Hugging Face datasets API on 2026-09-17 as one of the most-downloaded corpora whose card tag permits commercial use. 723,810 downloads in 30 days, 111 likes, card last modified 2026-07-16. The licence is the uploader's card tag: the repo tree was walked and holds no LICENSE file, so commercial use is NOT established. Size, hours and episodes are null: the API does not state them.",
        "org": "MVP Lab",
        "orgs": [
          "MVP Lab"
        ],
        "redistribution": null,
        "url": "https://huggingface.co/datasets/mvp-lab/LLaVA-OneVision-1.5-Mid-Training-85M"
      },
      "id": "dataset:llava-onevision-1-5-mid-training-85m",
      "name": "LLaVA-OneVision-1.5-Mid-Training-85M",
      "notes": {
        "_license_conflict": "https://huggingface.co/api/datasets/mvp-lab/LLaVA-OneVision-1.5-Mid-Training-85M says cardData.license apache-2.0, tag license:apache-2.0, lastModified 2026-07-16T16:17:14Z, gated false, 12,126 siblings with no LICENSE file.; https://www.image-net.org/download.php says ImageNet terms of access: 'Researcher shall use the Database only for non-commercial research and educational purposes.'",
        "_license_note": "UNCHANGED. Re-checked live: lastModified still 2026-07-16T16:17:14Z, no LICENSE added. The earlier pass flagged that the other seven upstreams, including SA-1B, had not been re-read; that gap is NOT closed this round. The SA-1B terms live behind a JavaScript-rendered Meta page and a PDF path that now 404s on the segment-anything repository, and both attempts to fetch them returned no licence text.",
        "_license_previously_recorded": "Card tag apache-2.0 only, over a redistribution of upstream corpora whose own terms of access are non-commercial. No LIC...",
        "_org_roles": {
          "MVP Lab": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-mine",
        "source_hosts": [
          "huggingface.co",
          "image-net.org"
        ],
        "source_urls": [
          "https://huggingface.co/api/datasets/mvp-lab/LLaVA-OneVision-1.5-Mid-Training-85M",
          "https://www.image-net.org/download.php",
          "https://huggingface.co/datasets/mvp-lab/LLaVA-OneVision-1.5-Mid-Training-85M"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.556,
        "confidence": "medium",
        "corroborated": true,
        "flags": [
          "conflicted",
          "license-conflict",
          "license-needs-review",
          "license-verified"
        ],
        "grade": "B",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 3,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://huggingface.co/datasets/mvp-lab/LLaVA-OneVision-1.5-Mid-Training-85M"
    },
    {
      "fields": {
        "commercial_use": true,
        "hours": 99500,
        "license": "OTS, consent-verified",
        "license_family": "ots",
        "license_spdx": null,
        "modalities": [
          "audio",
          "rgb",
          "text"
        ],
        "name": "Meeting Recordings",
        "org": "BLOMEGA",
        "orgs": [
          "BLOMEGA"
        ],
        "redistribution": null,
        "url": "https://blomega.com/explore-ots-datasets"
      },
      "id": "dataset:meeting-recordings",
      "name": "Meeting Recordings",
      "notes": {
        "_org_roles": {
          "BLOMEGA": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-08-14",
        "method": "import",
        "source_hosts": [
          "blomega.com"
        ],
        "source_urls": [
          "https://blomega.com/datasets.json",
          "https://blomega.com/explore-ots-datasets"
        ]
      },
      "quality": {
        "age_days": 35,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://blomega.com/explore-ots-datasets"
    },
    {
      "fields": {
        "commercial_use": true,
        "embodiment": "bimanual",
        "episodes": null,
        "format": "custom",
        "has_force": false,
        "hours": null,
        "license": "MIT",
        "license_family": "permissive",
        "license_spdx": "MIT",
        "modalities": [
          "proprioception",
          "rgb"
        ],
        "name": "Mobile ALOHA",
        "org": "Stanford",
        "orgs": [
          "Stanford"
        ],
        "redistribution": true,
        "url": "https://mobile-aloha.github.io/",
        "year": 2024
      },
      "id": "dataset:mobile-aloha",
      "name": "Mobile ALOHA",
      "notes": {
        "_org_roles": {
          "Stanford": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-09-16",
        "method": "refinery-corroborate",
        "source_hosts": [
          "marktechpost.com",
          "mobile-aloha.github.io"
        ],
        "source_urls": [
          "https://www.marktechpost.com/2024/01/11/researchers-from-stanford-present-mobile-aloha-a-low-cost-and-whole-body-teleoperation-system-for-data-collection",
          "https://mobile-aloha.github.io/"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": true,
        "flags": [],
        "grade": "B",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://mobile-aloha.github.io/"
    },
    {
      "fields": {
        "commercial_use": true,
        "episodes": 478000,
        "license": "OTS (478K clips)",
        "license_family": "ots",
        "license_spdx": null,
        "modalities": [
          "rgb"
        ],
        "name": "Multi-Person Social Interaction",
        "org": "BLOMEGA",
        "orgs": [
          "BLOMEGA"
        ],
        "redistribution": null,
        "url": "https://blomega.com/explore-ots-datasets"
      },
      "id": "dataset:multi-person-social-interaction",
      "name": "Multi-Person Social Interaction",
      "notes": {
        "_org_roles": {
          "BLOMEGA": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-08-14",
        "method": "import",
        "source_hosts": [
          "blomega.com"
        ],
        "source_urls": [
          "https://blomega.com/datasets.json",
          "https://blomega.com/explore-ots-datasets"
        ]
      },
      "quality": {
        "age_days": 35,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://blomega.com/explore-ots-datasets"
    },
    {
      "fields": {
        "commercial_use": true,
        "hours": 976000,
        "license": "consent-verified, license-clear (OTS)",
        "license_family": "ots",
        "license_spdx": null,
        "modalities": [
          "audio",
          "speech"
        ],
        "name": "Multilingual Conversations",
        "org": "BLOMEGA",
        "orgs": [
          "BLOMEGA"
        ],
        "redistribution": null,
        "url": "https://blomega.com/explore-ots-datasets"
      },
      "id": "dataset:multilingual-conversations",
      "name": "Multilingual Conversations",
      "notes": {
        "_org_roles": {
          "BLOMEGA": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-08-14",
        "method": "import",
        "source_hosts": [
          "blomega.com"
        ],
        "source_urls": [
          "https://blomega.com/datasets.json",
          "https://blomega.com/explore-ots-datasets"
        ]
      },
      "quality": {
        "age_days": 35,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://blomega.com/explore-ots-datasets"
    },
    {
      "fields": {
        "commercial_use": true,
        "episodes": 11000000,
        "license": "free; 11M pairs across 6 official languages",
        "license_family": "permissive",
        "license_spdx": null,
        "modalities": [
          "text"
        ],
        "name": "Multilingual UN Parallel Corpus",
        "org": "BLOMEGA",
        "orgs": [
          "BLOMEGA"
        ],
        "redistribution": true,
        "url": "https://blomega.com/explore-ots-datasets"
      },
      "id": "dataset:multilingual-un-parallel-corpus",
      "name": "Multilingual UN Parallel Corpus",
      "notes": {
        "_license_note": "The UN corpus page lists four conditions only (no warranty, no liability, attribute the UN, UN privileges reserved) and imposes no use restriction. The UN authors' LREC 2016 paper, section 2: documents 'are in the public domain' and the disclaimer applies with 'no other restrictions apply'. Page and paper agree. Commercial use permitted with attribution.",
        "_org_roles": {
          "BLOMEGA": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-09-17",
        "method": "import",
        "source_hosts": [
          "blomega.com",
          "lrec-conf.org",
          "un.org"
        ],
        "source_urls": [
          "https://www.lrec-conf.org/proceedings/lrec2016/pdf/1195_Paper.pdf",
          "https://www.un.org/dgacm/en/content/uncorpus",
          "https://blomega.com/datasets.json",
          "https://blomega.com/explore-ots-datasets"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "license-verified"
        ],
        "grade": "B",
        "hosts": 3,
        "independent_hosts": 3,
        "sources": 4,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://blomega.com/explore-ots-datasets"
    },
    {
      "fields": {
        "commercial_use": true,
        "has_force": false,
        "license": "OTS (3D and Lidar multimodal household robotics data)",
        "license_family": "ots",
        "license_spdx": null,
        "modalities": [
          "lidar"
        ],
        "name": "Multimodal Dataset for Household Robotics",
        "org": "BLOMEGA",
        "orgs": [
          "BLOMEGA"
        ],
        "redistribution": null,
        "url": "https://blomega.com/explore-ots-datasets"
      },
      "id": "dataset:multimodal-dataset-for-household-robotics",
      "name": "Multimodal Dataset for Household Robotics",
      "notes": {
        "_modalities_dropped": [
          "multimodal"
        ],
        "_org_roles": {
          "BLOMEGA": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-08-14",
        "method": "import",
        "source_hosts": [
          "blomega.com"
        ],
        "source_urls": [
          "https://blomega.com/datasets.json",
          "https://blomega.com/explore-ots-datasets"
        ]
      },
      "quality": {
        "age_days": 35,
        "completeness": 0.667,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "B",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://blomega.com/explore-ots-datasets"
    },
    {
      "fields": {
        "commercial_use": null,
        "license": "No licence published: the task organisers' repository carries no LICENSE file and the README states none; training part ...",
        "license_family": "unknown",
        "license_note": "Organiser repo (last push 2021-08-04) holds training rar and test zips, README gives a citation only, GitHub API licence null. Training = PEYMA 300K tokens + 600K new; test 150K. No grant means default copyright, so commercial use cannot be read as permitted. Third-party HF re-uploads of PEYMA tagged apache-2.0/mit are not the rights holder.",
        "license_spdx": null,
        "modalities": [
          "text"
        ],
        "name": "NSURL-2019 Persian NER",
        "note": "1,029,822 tokens. Web-normalized annotation density 197.0 against CoNLL-2003 English (301,418 tokens) at Common Crawl CC-MAIN-2026-30 page-share ratio 0.01735.",
        "org": "NSURL",
        "orgs": [
          "NSURL"
        ],
        "redistribution": null,
        "url": "https://arxiv.org/abs/2608.24698"
      },
      "id": "dataset:nsurl-2019-persian-ner",
      "name": "NSURL-2019 Persian NER",
      "notes": {
        "_license_note": "licence string did not match any known family",
        "_org_roles": {
          "NSURL": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-17",
        "method": "refinery-mine",
        "source_hosts": [
          "arxiv.org",
          "github.com"
        ],
        "source_urls": [
          "https://arxiv.org/abs/2608.24698",
          "https://github.com/nasrin-taghizadeh/NSURL-Persian-NER"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.556,
        "confidence": "high",
        "corroborated": true,
        "flags": [],
        "grade": "B",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://arxiv.org/abs/2608.24698"
    },
    {
      "fields": {
        "access_gated": "no",
        "commercial_use": false,
        "downloads_30d": 726554,
        "license": "README License section: the dataset as a whole is ODC-By v1.0, but individual objects carry their own Creative Commons l...",
        "license_family": "noncommercial",
        "license_spdx": null,
        "name": "objaverse",
        "note": "Added from the Hugging Face datasets API on 2026-09-17 as one of the most-downloaded corpora whose card tag permits commercial use. 726,554 downloads in 30 days, 469 likes, card last modified 2023-03-31. The licence is the uploader's card tag: the repo tree was walked and holds no LICENSE file, so commercial use is NOT established. Size, hours and episodes are null: the API does not state them.",
        "org": "Allen Institute for AI",
        "orgs": [
          "Allen Institute for AI"
        ],
        "redistribution": true,
        "url": "https://huggingface.co/datasets/allenai/objaverse"
      },
      "id": "dataset:objaverse",
      "name": "objaverse",
      "notes": {
        "_license_note": "More than a tag: the README has a real License section. ODC-By covers the compilation and permits commercial use of the database, but roughly 77K of the ~818K objects are NC-licensed, so the dataset as distributed cannot be used commercially in full. A commercially usable subset exists (721K CC-BY + 16K CC-BY-SA + 3.5K CC0) and the per-object licence is in the metadata, so the filter is mechanical",
        "_org_roles": {
          "Allen Institute for AI": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-17",
        "method": "refinery-mine",
        "source_hosts": [
          "huggingface.co"
        ],
        "source_urls": [
          "https://huggingface.co/datasets/allenai/objaverse/blob/main/README.md",
          "https://huggingface.co/datasets/allenai/objaverse"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.556,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "license-verified",
          "single-source"
        ],
        "grade": "C",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://huggingface.co/datasets/allenai/objaverse"
    },
    {
      "fields": {
        "commercial_use": false,
        "license": "cc-by-nc-nd-4.0",
        "license_family": "noncommercial",
        "license_spdx": "CC-BY-NC-ND-4.0",
        "modalities": [
          "text"
        ],
        "name": "Objective Projection (v7.2)",
        "note": "500 Turkish-English scene pairs; applied_rules annotation layer measured at Cohen's kappa 0.004 to 0.027 against human labels on materialized metaphor (arXiv:2609.13936). Not gated; 337 downloads as of 2026-09-16.",
        "org": "Levent Bulut",
        "orgs": [
          "Levent Bulut"
        ],
        "redistribution": false,
        "url": "https://huggingface.co/datasets/leventbulut/objective-projection"
      },
      "id": "dataset:objective-projection-v72",
      "name": "Objective Projection (v7.2)",
      "notes": {
        "_org_roles": {
          "Levent Bulut": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-16",
        "last_seen": "2026-09-16",
        "method": "refinery",
        "source_hosts": [
          "huggingface.co"
        ],
        "source_urls": [
          "https://huggingface.co/datasets/leventbulut/objective-projection"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.556,
        "confidence": "high",
        "corroborated": false,
        "flags": [
          "single-source"
        ],
        "grade": "C",
        "hosts": 1,
        "independent_hosts": 1,
        "sources": 1,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://huggingface.co/datasets/leventbulut/objective-projection"
    },
    {
      "fields": {
        "byte_share_mano_files_20_sample": 0.727,
        "byte_share_mano_incl_visualisations_20_sample": 9.882,
        "commercial_use": true,
        "contributors": "500+",
        "devices": "400+ smartphone models",
        "has_force": false,
        "hf_created": "2026-07-07",
        "hf_downloads_2026_09_15": 560619,
        "hf_last_modified": "2026-09-04",
        "hours": 2000,
        "license": "Open-AoE Dataset License v1.0 (commercial with attribution) for video+annotations; hands.npz and camera_traj.npz additionally subject to MANO license (non-commercial research only). Paper states CC BY 4.0.",
        "license_family": "mixed",
        "license_spdx": null,
        "median_minutes_per_contributor_id_100h_sample": 2.6,
        "modalities": [
          "camera-trajectory",
          "hand-pose",
          "rgb",
          "text"
        ],
        "name": "Open-AoE (OpenAoE-2000h)",
        "org": [
          "Ant Group",
          "inclusionAI"
        ],
        "orgs": [
          "Ant Group",
          "inclusionAI"
        ],
        "redistribution": null,
        "top10_contributor_share_100h_sample": 13.7,
        "top_level_sample_dirs_2026_09_15": 26736,
        "uploaded_hours_logged_by_2026_09_03": 1206,
        "uploaded_hours_per_release_notes": {
          "2026-07-31": 323,
          "2026-08-12": 694,
          "2026-09-03": 189
        },
        "url": "https://huggingface.co/datasets/inclusionAI/OpenAoE-2000h"
      },
      "id": "dataset:open-aoe-openaoe-2000h",
      "name": "Open-AoE (OpenAoE-2000h)",
      "notes": {
        "_license_note": "The previously recorded string is CONFIRMED correct and is now verified end to end. Video plus annotations: commercial use permitted with attribution to Open-AoE (dataset inclusionAI/OpenAoE-2000h; paper arXiv:2607.14183). The MANO carve-out was independently checked at the MANO source: its licence is titled 'Software Copyright License for non-commercial scientific research purposes' and grants ri",
        "_org_roles": {
          "Ant Group": [
            "org"
          ],
          "inclusionAI": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-15",
        "last_seen": "2026-09-16",
        "method": "refinery",
        "source_hosts": [
          "arxiv.org",
          "huggingface.co",
          "mano.is.tue.mpg.de"
        ],
        "source_urls": [
          "https://arxiv.org/abs/2607.14183",
          "https://mano.is.tue.mpg.de/license.html",
          "https://huggingface.co/datasets/inclusionAI/OpenAoE-2000h/raw/main/README.md",
          "https://huggingface.co/datasets/inclusionAI/OpenAoE-2000h/blob/main/LICENSE",
          "https://huggingface.co/datasets/inclusionAI/OpenAoE-2000h"
        ]
      },
      "quality": {
        "age_days": 2,
        "completeness": 0.778,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "license-needs-review",
          "license-verified"
        ],
        "grade": "B",
        "hosts": 3,
        "independent_hosts": 3,
        "sources": 5,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://huggingface.co/datasets/inclusionAI/OpenAoE-2000h"
    },
    {
      "fields": {
        "commercial_use": true,
        "embodiment": "multi",
        "episodes": 1000000,
        "format": "RLDS",
        "has_force": false,
        "license": "CC-BY-4.0",
        "license_family": "permissive",
        "license_spdx": "CC-BY-4.0",
        "modalities": [
          "proprioception",
          "rgb",
          "text"
        ],
        "name": "Open X-Embodiment",
        "org": "Google DeepMind",
        "orgs": [
          "Google DeepMind"
        ],
        "redistribution": true,
        "url": "https://robotics-transformer-x.github.io/",
        "year": 2023
      },
      "id": "dataset:open-x-embodiment",
      "name": "Open X-Embodiment",
      "notes": {
        "_license_note": "README 'License and Disclaimer': 'All software is licensed under the Apache License, Version 2.0' and 'All other materials are licensed under the Creative Commons Attribution 4.0 International License (CC-BY)'. The paper (arXiv 2310.08864) states no licence and the dataset spreadsheet has no licence column, so no conflict. Caveat: upstream terms of the 72 contributed subsets not audited.",
        "_org_roles": {
          "Google DeepMind": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-08-14",
        "last_seen": "2026-09-17",
        "method": "refinery-mine",
        "source_hosts": [
          "blomega.com",
          "claru.ai",
          "github.com",
          "robotics-transformer-x.github.io"
        ],
        "source_urls": [
          "https://blomega.com/guides/licensable-robotics-training-datasets",
          "https://github.com/google-deepmind/open_x_embodiment",
          "https://robotics-transformer-x.github.io/",
          "https://claru.ai/glossary/open-x-embodiment"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.778,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "license-verified"
        ],
        "grade": "B",
        "hosts": 4,
        "independent_hosts": 4,
        "sources": 4,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://robotics-transformer-x.github.io/"
    },
    {
      "fields": {
        "access_gated": "no",
        "commercial_use": null,
        "downloads_30d": 6186,
        "license": null,
        "license_family": "unclear-scope",
        "license_spdx": "MIT",
        "name": "OpenEAI-Dataset",
        "note": "Aggregation of 11 upstream datasets published under a single MIT card tag. Its own arxiv tags include 2307.00595 (RH20T) and 2212.06817 (RT-1), both carried here as research-only with commercial_use false, and the card names a licence for only two upstreams (Open X-Embodiment Apache-2.0, UMI MIT). A permissive tag on an aggregate of restrictively licensed data is what a buyer needs caught. The MIT claim is disputed in refinery/disputes.json and is not served as a licence.",
        "org": "OpenEAI",
        "orgs": [
          "OpenEAI"
        ],
        "redistribution": null,
        "size": "3.12 TB",
        "url": "https://huggingface.co/datasets/OpenEAI/OpenEAI-Dataset"
      },
      "id": "dataset:openeai-dataset",
      "name": "OpenEAI-Dataset",
      "notes": {
        "_claimed_license": "MIT",
        "_disputes": [
          {
            "checked": "2026-09-17",
            "claim": "MIT (Hugging Face card tag) for a 3.12 TB aggregate of 11 upstream datasets",
            "field": "license",
            "reason": "The card's own arxiv tags include RH20T (2307.00595) and RT-1 (2212.06817), which this registry records as research-only with commercial use not permitted, and the card names a licence for only two of the eleven upstreams. An uploader cannot relicense data it does not own. The repo does hold a LICENSE file (MIT, added 2026-02-25), but that file itself says third-party dataset terms still apply, so it does not clear the upstreams."
          }
        ],
        "_license_conflict": "https://huggingface.co/datasets/OpenEAI/OpenEAI-Dataset/raw/main/LICENSE says MIT, expressly subject to the licences and use restrictions of the included third-party datasets.; https://huggingface.co/api/datasets/OpenEAI/OpenEAI-Dataset says lastModified 2026-02-25T02:12:11Z, gated false, cardData.license mit, and the sibling list still contains both LICENSE and README.md among 2,954 files. Unchan",
        "_license_note": "UNCHANGED. Re-checked live: lastModified still 2026-02-25T02:12:11Z, the LICENSE file added in the 2026-02-25 commit is still the only licence artefact, and no upstream terms were restated. The MIT grant is real but defeats itself by deferring to upstream terms, and RH20T and RT-1 are not commercially usable. Commercial use stays null and the conflict stands.",
        "_license_previously_recorded": "An MIT LICENSE file, expressly made subject to the licences and use restrictions of the third-party datasets it aggregat...",
        "_org_roles": {
          "OpenEAI": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-mine",
        "source_hosts": [
          "huggingface.co",
          "rh20t.github.io"
        ],
        "source_urls": [
          "https://rh20t.github.io/",
          "https://huggingface.co/datasets/OpenEAI/OpenEAI-Dataset",
          "https://huggingface.co/datasets/OpenEAI/OpenEAI-Dataset/raw/main/LICENSE",
          "https://huggingface.co/datasets/OpenEAI/OpenEAI-Dataset/blob/main/LICENSE",
          "https://huggingface.co/api/datasets/OpenEAI/OpenEAI-Dataset/tree/main",
          "https://huggingface.co/datasets/OpenEAI/OpenEAI-Dataset/blob/main/README.md",
          "https://huggingface.co/api/datasets/OpenEAI/OpenEAI-Dataset"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.556,
        "confidence": "medium",
        "corroborated": true,
        "flags": [
          "conflicted",
          "disputed",
          "license-conflict",
          "license-verified"
        ],
        "grade": "C",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 7,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://huggingface.co/datasets/OpenEAI/OpenEAI-Dataset"
    },
    {
      "fields": {
        "commercial_use": null,
        "episodes": 140,
        "has_force": true,
        "hours": 2,
        "license": "No licence and no release: GitHub's licence detection on jessicayin/osmo_tactile_glove returns null, the repository root...",
        "license_family": "unknown",
        "license_note": "GET /repos/jessicayin/osmo_tactile_glove/license returns a null licence, and the root listing is .gitignore, .nojekyll, README.md, README_archive.md, conda, data, firmware, glovedp, hardware, kinematics, labs, models, pcb, scripts. The README pipeline step reads 'sample data collect for paper can be downloaded via data/download_data.sh (TODO: upload data and update script)', so the demonstrations",
        "license_spdx": null,
        "modalities": [
          "rgb",
          "tactile"
        ],
        "name": "OSMO human demonstration set",
        "notes": "140 demonstrations, approximately two hours.",
        "org": [
          "Meta",
          "University of Pennsylvania",
          "University of Michigan"
        ],
        "orgs": [
          "Meta",
          "University of Pennsylvania",
          "University of Michigan"
        ],
        "redistribution": null,
        "url": "https://arxiv.org/abs/2512.08920"
      },
      "id": "dataset:osmo-human-demonstration-set",
      "name": "OSMO human demonstration set",
      "notes": {
        "_license_note": "licence string did not match any known family",
        "_org_roles": {
          "Meta": [
            "org"
          ],
          "University of Michigan": [
            "org"
          ],
          "University of Pennsylvania": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-09",
        "last_seen": "2026-09-17",
        "method": "refinery-mine",
        "source_hosts": [
          "arxiv.org",
          "github.com",
          "jessicayin.github.io"
        ],
        "source_urls": [
          "https://jessicayin.github.io/osmo_tactile_glove",
          "https://arxiv.org/abs/2512.08920",
          "https://github.com/jessicayin/osmo_tactile_glove"
        ]
      },
      "quality": {
        "age_days": 1,
        "completeness": 0.889,
        "confidence": "high",
        "corroborated": true,
        "flags": [],
        "grade": "A",
        "hosts": 3,
        "independent_hosts": 3,
        "sources": 3,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://arxiv.org/abs/2512.08920"
    },
    {
      "fields": {
        "access_gated": "no",
        "commercial_use": null,
        "downloads_30d": 1250167,
        "license": "CC-BY-4.0",
        "license_family": "permissive",
        "license_spdx": "CC-BY-4.0",
        "name": "PhysicalAI-Robotics-GR00T-X-Embodiment-Sim",
        "note": "Added from the Hugging Face datasets API on 2026-09-17 as one of the most-downloaded corpora whose card tag permits commercial use. 1,250,167 downloads in 30 days, 271 likes, card last modified 2026-03-05. The licence is the uploader's card tag: the repo tree was walked and holds no LICENSE file, so commercial use is NOT established. Size, hours and episodes are null: the API does not state them.",
        "org": "NVIDIA",
        "orgs": [
          "NVIDIA"
        ],
        "redistribution": null,
        "url": "https://huggingface.co/datasets/nvidia/PhysicalAI-Robotics-GR00T-X-Embodiment-Sim"
      },
      "id": "dataset:physicalai-robotics-gr00t-x-embodiment-sim",
      "name": "PhysicalAI-Robotics-GR00T-X-Embodiment-Sim",
      "notes": {
        "_license_note": "UNCHANGED. Re-checked live: lastModified still 2026-03-05T23:36:40Z, gated false, and none of the 97,956 files is a LICENSE. This is the cleanest case in the batch: the tag is first-party, nothing contradicts it, and the only thing standing between it and commercial_use true is the registry's own rule that a card tag alone is not a grant. Recorded as no conflict, commercial use null.",
        "_license_previously_recorded": "Card tag cc-by-4.0, uploaded by NVIDIA itself. No LICENSE file and no separate NVIDIA terms document for this data.",
        "_org_roles": {
          "NVIDIA": [
            "org"
          ]
        }
      },
      "provenance": {
        "consent_license": "public",
        "first_seen": "2026-09-17",
        "last_seen": "2026-09-18",
        "method": "refinery-mine",
        "source_hosts": [
          "github.com",
          "huggingface.co"
        ],
        "source_urls": [
          "https://github.com/NVIDIA/Isaac-GR00T",
          "https://huggingface.co/datasets/nvidia/PhysicalAI-Robotics-GR00T-X-Embodiment-Sim"
        ]
      },
      "quality": {
        "age_days": 0,
        "completeness": 0.556,
        "confidence": "high",
        "corroborated": true,
        "flags": [
          "license-needs-review",
          "license-verified"
        ],
        "grade": "B",
        "hosts": 2,
        "independent_hosts": 2,
        "sources": 2,
        "stale": false,
        "stale_after_days": 365
      },
      "type": "dataset",
      "url": "https://huggingface.co/datasets/nvidia/PhysicalAI-Robotics-GR00T-X-Embodiment-Sim"
    }
  ],
  "_meta": {
    "source": "Blomega Data Refinery",
    "url": "https://data.blomega.com",
    "publisher": "Blomega",
    "publisher_url": "https://blomegalab.com",
    "wikidata": "Q141048865",
    "license": "CC BY 4.0",
    "license_url": "https://creativecommons.org/licenses/by/4.0/",
    "cite_as": "Blomega Data Refinery (https://data.blomega.com), CC BY 4.0. Cite the registry and the record id.",
    "attribution_required": true
  }
}
