{
  "version": "1.0.0",
  "checkedDate": "2026-07-22",
  "recipes": [
    {
      "id": "MIXTURE-OPENVLA-OXE-MAGIC-SOUP-PLUS",
      "model_recipe_id": "openvla-7b / oxe_magic_soup_plus",
      "source_reported_version": "OpenVLA commit c8f03f48af69",
      "normalized_field": "named mixture",
      "normalized_value": "oxe_magic_soup_plus",
      "unit": "relative sampling weight",
      "primary_source_url": "https://github.com/openvla/openvla/blob/c8f03f48af69/prismatic/vla/datasets/rlds/oxe/mixtures.py",
      "source_type": "project",
      "source_id": "project-github-com-openvla-openvla-mixtures",
      "exact_locator": "OXE_NAMED_MIXTURES['oxe_magic_soup_plus']",
      "checked_date": "2026-07-22",
      "retrieval_hash": "git-commit:c8f03f48af69",
      "confidence": "high",
      "status": "human",
      "evidence_basis": "source-reported",
      "filter": "Active tuple entries only; commented broken or omitted entries are excluded by the source config.",
      "normalization": "unknown — mixtures.py points to separate dataset transforms/configs and does not define one shared normalization rule",
      "action_schema_rate": "unknown — heterogeneous component action schemas and control rates are not reported in mixtures.py",
      "success_failure_ratio": "unknown — mixtures.py does not report success/failure composition",
      "train_eval_separation": "unknown in source config — define a target-embodiment holdout before training",
      "eval_split": "not specified by mixtures.py — project-specific held-out split required",
      "license_compatibility": "unknown — code availability does not establish compatibility of every dataset, consent term, or derived-model right",
      "limitation": "source-reported, not independently validated; weights are relative sampling weights, not normalized probabilities or a universally optimal recipe.",
      "components": [
        {
          "dataset_id": "fractal20220817_data",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "0.54087122203",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "kuka",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "0.8341046294",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "bridge_orig",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "1.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "taco_play",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "2.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "jaco_play",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "1.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "berkeley_cable_routing",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "1.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "roboturk",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "2.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "viola",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "2.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "berkeley_autolab_ur5",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "2.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "toto",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "1.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "language_table",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "0.1",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "stanford_hydra_dataset_converted_externally_to_rlds",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "2.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "austin_buds_dataset_converted_externally_to_rlds",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "1.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "nyu_franka_play_dataset_converted_externally_to_rlds",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "3.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "furniture_bench_dataset_converted_externally_to_rlds",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "0.1",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "ucsd_kitchen_dataset_converted_externally_to_rlds",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "2.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "austin_sailor_dataset_converted_externally_to_rlds",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "1.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "austin_sirius_dataset_converted_externally_to_rlds",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "1.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "dlr_edan_shared_control_converted_externally_to_rlds",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "1.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "iamlab_cmu_pickup_insert_converted_externally_to_rlds",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "1.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "utaustin_mutex",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "1.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "berkeley_fanuc_manipulation",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "2.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "cmu_stretch",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "1.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "bc_z",
          "dataset_version": "v0.1.0 (source comment)",
          "relative_weight": "0.2",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "fmb_dataset",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "1.0",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "dobbe",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "0.2",
          "unit": "relative sampling weight"
        },
        {
          "dataset_id": "droid",
          "dataset_version": "unknown — mixtures.py does not pin a component dataset version",
          "relative_weight": "0.06",
          "unit": "relative sampling weight"
        }
      ]
    }
  ],
  "quality": [
    {
      "id": "QUALITY-AUTO-LOAD-SCHEMA",
      "source_reported_version": "TrueLabel methodology v1.0.0",
      "normalized_field": "loadability and schema drift",
      "normalized_value": "parse every episode; compare required observation/action keys and dtypes",
      "unit": "episode",
      "primary_source_url": "https://arxiv.org/abs/2603.09056",
      "source_type": "paper",
      "source_id": "paper-arxiv-org-abs-2603-09056",
      "exact_locator": "Sections IV-A–IV-B (validation-relative quality)",
      "checked_date": "2026-07-22",
      "retrieval_hash": "unknown — no retrieval snapshot is committed",
      "confidence": "medium",
      "status": "automatic",
      "grading_layer": "automated fact",
      "evidence_basis": "TrueLabel analysis",
      "signal": "load result, missing keys, unexpected keys, dtype and shape changes",
      "decision_rule": "accept conforming episodes; quarantine recoverable schema drift; reject unreadable episodes",
      "limitation": "Operational gate derived for auditability; the cited paper does not prescribe these parser decisions."
    },
    {
      "id": "QUALITY-AUTO-TIMING-COMPLETENESS",
      "source_reported_version": "TrueLabel methodology v1.0.0",
      "normalized_field": "timing, missing values, and stream completeness",
      "normalized_value": "check monotonic timestamps, finite values, aligned stream lengths, and episode boundaries",
      "unit": "stream",
      "primary_source_url": "https://arxiv.org/abs/2603.09056",
      "source_type": "paper",
      "source_id": "paper-arxiv-org-abs-2603-09056",
      "exact_locator": "Section IV-B (trajectory-wise curation)",
      "checked_date": "2026-07-22",
      "retrieval_hash": "unknown — no retrieval snapshot is committed",
      "confidence": "medium",
      "status": "automatic",
      "grading_layer": "automated fact",
      "evidence_basis": "TrueLabel analysis",
      "signal": "timestamp order, NaN/Inf counts, stream-length deltas, terminal markers",
      "decision_rule": "accept complete aligned streams; quarantine repairable gaps; reject non-reconstructable timing or value corruption",
      "limitation": "A structural pass does not prove that an action was intentional, safe, or useful."
    },
    {
      "id": "QUALITY-AUTO-DISTRIBUTION",
      "source_reported_version": "arXiv:2410.18647v1",
      "normalized_field": "environment and object distribution coverage",
      "normalized_value": "report coverage by held-out target environment, object, task, and embodiment",
      "unit": "slice",
      "primary_source_url": "https://arxiv.org/html/2410.18647v1",
      "source_type": "paper",
      "source_id": "paper-arxiv-org-abs-2410-18647",
      "exact_locator": "Abstract and evaluation protocol",
      "checked_date": "2026-07-22",
      "retrieval_hash": "unknown — no retrieval snapshot is committed",
      "confidence": "high",
      "status": "needs-review",
      "grading_layer": "automated fact",
      "evidence_basis": "TrueLabel analysis",
      "signal": "counts and coverage by declared target-domain slice",
      "decision_rule": "report the distribution without one global score; quarantine a recipe when a required target slice is absent",
      "limitation": "source-reported scaling behavior is task-specific and does not supply universal mixture weights or thresholds."
    },
    {
      "id": "QUALITY-AUTO-DUPLICATION-LEAKAGE",
      "source_reported_version": "TrueLabel methodology v1.0.0",
      "normalized_field": "duplication and train/eval leakage",
      "normalized_value": "compare content, episode, task, environment, and embodiment identities across splits",
      "unit": "pairwise match",
      "primary_source_url": "https://arxiv.org/abs/2603.09056",
      "source_type": "paper",
      "source_id": "paper-arxiv-org-abs-2603-09056",
      "exact_locator": "Section IV-B (coverage-preserving trajectory selection)",
      "checked_date": "2026-07-22",
      "retrieval_hash": "unknown — no retrieval snapshot is committed",
      "confidence": "medium",
      "status": "automatic",
      "grading_layer": "automated fact",
      "evidence_basis": "TrueLabel analysis",
      "signal": "exact/near duplicate groups and identity overlap between training and evaluation",
      "decision_rule": "accept disjoint splits; quarantine ambiguous provenance; reject confirmed evaluation leakage",
      "limitation": "No single fingerprint detects every semantic duplicate or hidden upstream overlap."
    },
    {
      "id": "QUALITY-AUTO-RIGHTS-PROVENANCE",
      "source_reported_version": "TrueLabel methodology v1.0.0",
      "normalized_field": "rights and provenance completeness",
      "normalized_value": "record code license, dataset terms, consent basis, provenance, and derived-model terms separately",
      "unit": "dataset component",
      "primary_source_url": "https://www.roboticsproceedings.org/rss21/p023.html",
      "source_type": "paper",
      "source_id": "paper-rss21-p023",
      "exact_locator": "Paper scope and limitations (quality signal, not rights review)",
      "checked_date": "2026-07-22",
      "retrieval_hash": "unknown — no retrieval snapshot is committed",
      "confidence": "medium",
      "status": "needs-review",
      "grading_layer": "human judgment",
      "evidence_basis": "TrueLabel analysis",
      "signal": "presence and review status of each distinct rights/provenance artifact",
      "decision_rule": "accept only reviewed compatible terms; quarantine missing or ambiguous terms; reject known incompatible use",
      "limitation": "The cited curation paper does not provide legal guidance; compatibility requires qualified human review."
    },
    {
      "id": "QUALITY-HUMAN-TASK-VALIDITY",
      "source_reported_version": "RSS 2025 paper 23",
      "normalized_field": "demonstration validity and strategy quality",
      "normalized_value": "review task completion, recoveries, unsafe shortcuts, and whether behavior represents the desired policy",
      "unit": "trajectory",
      "primary_source_url": "https://www.roboticsproceedings.org/rss21/p023.html",
      "source_type": "paper",
      "source_id": "paper-rss21-p023",
      "exact_locator": "Abstract and human-expert quality comparison",
      "checked_date": "2026-07-22",
      "retrieval_hash": "unknown — no retrieval snapshot is committed",
      "confidence": "high",
      "status": "human",
      "grading_layer": "human judgment",
      "evidence_basis": "TrueLabel analysis",
      "signal": "reviewer decision with reason code and task-specific rubric",
      "decision_rule": "accept desired valid behavior; quarantine uncertain or recoverable behavior; reject invalid, unsafe, or out-of-scope behavior",
      "limitation": "Human judgments can disagree; retain reviewer identity, rubric version, and disagreement rather than collapsing them into one score."
    },
    {
      "id": "QUALITY-HUMAN-CONTRIBUTION",
      "source_reported_version": "arXiv:2603.09056",
      "normalized_field": "contribution to held-out desired behavior",
      "normalized_value": "estimate trajectory contribution relative to a declared validation set, then review selection coverage",
      "unit": "trajectory",
      "primary_source_url": "https://arxiv.org/abs/2603.09056",
      "source_type": "paper",
      "source_id": "paper-arxiv-org-abs-2603-09056",
      "exact_locator": "Sections IV-A–IV-B",
      "checked_date": "2026-07-22",
      "retrieval_hash": "unknown — no retrieval snapshot is committed",
      "confidence": "high",
      "status": "needs-review",
      "grading_layer": "human judgment",
      "evidence_basis": "source-reported",
      "signal": "influence estimate plus trajectory-level coverage review",
      "decision_rule": "compare ablations over multiple retained-set sizes; do not publish a universal cutoff",
      "limitation": "source-reported, not independently validated; rankings depend on the model, validation set, estimator, and target behavior."
    }
  ]
}
