{
  "schemaVersion": 1,
  "updated": "2026-09-27",
  "source": {
    "manuscript": "Author-supplied ICML 2026 LaTeX, September 27, 2026",
    "sha256": "c8f592ff7158534ebe0997d992b1eae2610affe887b53ead8d26b55f33f84037",
    "pdfSha256": "ebb9e1d8bfe51449b65ebd5405101ed9f96514b328e6f50325d1597191a47d94",
    "pdfDate": "2026-09-27",
    "pdfPages": 13,
    "tableRows": 45,
    "numericCells": 162,
    "verification": "All 162 numeric table entries checked against the author-supplied ICML 2026 PDF and source on September 27, 2026; original English/Chinese source values also agree."
  },
  "reporting": {
    "seeds": "Three independent seeds for configurations requiring retraining; seed identifiers and per-seed values are not supplied.",
    "uncertainty": "Tables report measured means without standard deviations or confidence intervals. No statistical-significance claim is supported.",
    "scope": "Simulation evaluation only. Real-robot transfer is not established.",
    "budget": "One reported evaluation setting; no continuous budget sweep or equal-success comparison.",
    "hardware": "Matched across repair strategies according to the manuscript; a concrete hardware specification and run-level timings are not supplied.",
    "rawRecords": "No episode-level experiment records are included in the supplied materials."
  },
  "references": {
    "A": {
      "name": "OpenVLA-OFT",
      "citation": "Kim et al., 2025",
      "url": "https://arxiv.org/abs/2502.19645v2"
    },
    "B": {
      "name": "Fast-ThinkAct",
      "citation": "Huang et al., 2026",
      "url": "https://arxiv.org/abs/2601.09708v2"
    },
    "C": {
      "name": "LaRA-VLA",
      "citation": "Bai et al., 2026",
      "url": "https://arxiv.org/abs/2602.01166v2"
    },
    "robotwin": {
      "name": "RoboTwin 2.0",
      "citation": "Chen et al., 2025",
      "url": "https://arxiv.org/abs/2506.18088v2"
    },
    "recover": {
      "name": "LIBERO-RECOVER",
      "citation": "Liu et al., 2026",
      "url": "https://arxiv.org/abs/2609.05178v2"
    }
  },
  "tables": {
    "libero": {
      "id": "libero",
      "label": "tab:libero",
      "number": "1",
      "page": 7,
      "title": "LIBERO task success",
      "context": "Published references and measurements from this work. Four suites, ten tasks per suite. VLCoT equally weights the four suite scores; published averages retain their original definitions.",
      "columns": [
        "Spatial",
        "Object",
        "Goal",
        "Long",
        "Average"
      ],
      "keys": [
        "spatial",
        "object",
        "goal",
        "long",
        "average"
      ],
      "units": [
        "%",
        "%",
        "%",
        "%",
        "%"
      ],
      "hasSource": true,
      "sourceType": "published_context",
      "citation": "A: OpenVLA-OFT; B: Fast-ThinkAct; C: LaRA-VLA.",
      "rows": [
        {
          "id": "diffusion-policy",
          "name": "Diffusion Policy",
          "source": "A",
          "values": {
            "spatial": 78.3,
            "object": 92.5,
            "goal": 68.3,
            "long": 50.5,
            "average": 72.4
          }
        },
        {
          "id": "octo",
          "name": "Octo",
          "source": "A",
          "values": {
            "spatial": 78.9,
            "object": 85.7,
            "goal": 84.6,
            "long": 51.1,
            "average": 75.1
          }
        },
        {
          "id": "dit-policy",
          "name": "DiT Policy",
          "source": "A",
          "values": {
            "spatial": 84.2,
            "object": 96.3,
            "goal": 85.4,
            "long": 63.8,
            "average": 82.4
          }
        },
        {
          "id": "openvla",
          "name": "OpenVLA",
          "source": "A",
          "values": {
            "spatial": 84.7,
            "object": 88.4,
            "goal": 79.2,
            "long": 53.7,
            "average": 76.5
          }
        },
        {
          "id": "pi0-fast",
          "name": "π₀-FAST",
          "source": "A",
          "values": {
            "spatial": 96.4,
            "object": 96.8,
            "goal": 88.6,
            "long": 60.2,
            "average": 85.5
          }
        },
        {
          "id": "pi0",
          "name": "π₀",
          "source": "A",
          "values": {
            "spatial": 96.8,
            "object": 98.8,
            "goal": 95.8,
            "long": 85.2,
            "average": 94.2
          }
        },
        {
          "id": "openvla-oft",
          "name": "OpenVLA-OFT",
          "source": "A",
          "values": {
            "spatial": 97.6,
            "object": 98.4,
            "goal": 97.9,
            "long": 94.5,
            "average": 97.1
          }
        },
        {
          "id": "cot-vla",
          "name": "CoT-VLA",
          "source": "B",
          "values": {
            "spatial": 87.5,
            "object": 91.6,
            "goal": 87.6,
            "long": 69,
            "average": 83.9
          }
        },
        {
          "id": "thinkact",
          "name": "ThinkAct",
          "source": "B",
          "values": {
            "spatial": 88.3,
            "object": 91.4,
            "goal": 87.1,
            "long": 70.9,
            "average": 84.4
          }
        },
        {
          "id": "molmoact",
          "name": "MolmoAct",
          "source": "B",
          "values": {
            "spatial": 87,
            "object": 95.4,
            "goal": 87.6,
            "long": 77.2,
            "average": 86.8
          }
        },
        {
          "id": "fast-thinkact",
          "name": "Fast-ThinkAct",
          "source": "B",
          "values": {
            "spatial": 92,
            "object": 97.2,
            "goal": 90.2,
            "long": 79.4,
            "average": 89.7
          }
        },
        {
          "id": "lara-vla",
          "name": "LaRA-VLA",
          "source": "C",
          "values": {
            "spatial": 96.4,
            "object": 99.8,
            "goal": 98.6,
            "long": 96.6,
            "average": 97.9
          }
        },
        {
          "id": "vlcot",
          "name": "VLCoT",
          "source": "This work",
          "values": {
            "spatial": 98,
            "object": 98.8,
            "goal": 98.4,
            "long": 96.8,
            "average": 98
          }
        }
      ]
    },
    "robotwin50": {
      "id": "robotwin50",
      "label": "tab:robotwin_50",
      "number": "2",
      "page": 7,
      "title": "RoboTwin 2.0 · 50 tasks",
      "context": "Per-task success, averaged over the 50-task benchmark. The reported protocol uses 50 clean demonstrations per task and 100 tests per clean/randomized condition. Published baselines retain their original settings.",
      "columns": [
        "Clean (Easy)",
        "Randomized (Hard)"
      ],
      "keys": [
        "clean",
        "randomized"
      ],
      "units": [
        "%",
        "%"
      ],
      "sourceType": "published_context",
      "citation": "RoboTwin 2.0 (Chen et al., 2025).",
      "rows": [
        {
          "id": "dp",
          "name": "DP",
          "source": "RoboTwin 2.0 (Chen et al., 2025).",
          "values": {
            "clean": 28,
            "randomized": 0.6
          }
        },
        {
          "id": "act",
          "name": "ACT",
          "source": "RoboTwin 2.0 (Chen et al., 2025).",
          "values": {
            "clean": 29.7,
            "randomized": 1.7
          }
        },
        {
          "id": "rdt",
          "name": "RDT",
          "source": "RoboTwin 2.0 (Chen et al., 2025).",
          "values": {
            "clean": 34.5,
            "randomized": 13.7
          }
        },
        {
          "id": "pi0",
          "name": "π₀",
          "source": "RoboTwin 2.0 (Chen et al., 2025).",
          "values": {
            "clean": 46.4,
            "randomized": 16.3
          }
        },
        {
          "id": "dp3",
          "name": "DP3",
          "source": "RoboTwin 2.0 (Chen et al., 2025).",
          "values": {
            "clean": 55.2,
            "randomized": 5
          }
        },
        {
          "id": "vlcot",
          "name": "VLCoT",
          "source": "This work",
          "values": {
            "clean": 58,
            "randomized": 23
          }
        }
      ]
    },
    "robotwin10": {
      "id": "robotwin10",
      "label": "tab:robotwin_10",
      "number": "3",
      "page": 7,
      "title": "RoboTwin 2.0 · 10-task subset",
      "context": "The ten-task subset follows the Fast-ThinkAct reference protocol. Its task coverage differs from the 50-task benchmark; these averages must not be combined or directly ranked against that benchmark.",
      "columns": [
        "Clean (Easy)",
        "Randomized (Hard)"
      ],
      "keys": [
        "clean",
        "randomized"
      ],
      "units": [
        "%",
        "%"
      ],
      "sourceType": "published_context",
      "citation": "Fast-ThinkAct (Huang et al., 2026).",
      "rows": [
        {
          "id": "dp",
          "name": "DP",
          "source": "Fast-ThinkAct (Huang et al., 2026).",
          "values": {
            "clean": 43.1,
            "randomized": 0.6
          }
        },
        {
          "id": "act",
          "name": "ACT",
          "source": "Fast-ThinkAct (Huang et al., 2026).",
          "values": {
            "clean": 45.5,
            "randomized": 3.5
          }
        },
        {
          "id": "pi0",
          "name": "π₀",
          "source": "Fast-ThinkAct (Huang et al., 2026).",
          "values": {
            "clean": 52.9,
            "randomized": 16.3
          }
        },
        {
          "id": "rdt",
          "name": "RDT",
          "source": "Fast-ThinkAct (Huang et al., 2026).",
          "values": {
            "clean": 56.4,
            "randomized": 22.8
          }
        },
        {
          "id": "thinkact",
          "name": "ThinkAct",
          "source": "Fast-ThinkAct (Huang et al., 2026).",
          "values": {
            "clean": 62.4,
            "randomized": 24.7
          }
        },
        {
          "id": "fast-thinkact",
          "name": "Fast-ThinkAct",
          "source": "Fast-ThinkAct (Huang et al., 2026).",
          "values": {
            "clean": 65.7,
            "randomized": 26.4
          }
        },
        {
          "id": "vlcot",
          "name": "VLCoT",
          "source": "This work",
          "values": {
            "clean": 68,
            "randomized": 31
          }
        }
      ]
    },
    "recover": {
      "id": "recover",
      "label": "tab:recover",
      "number": "4",
      "page": 7,
      "title": "LIBERO-RECOVER · recovery levels",
      "context": "Each level equally weights Spatial, Object, Goal and LIBERO-100. L1: action retry; L2: action adaptation; L3: object-state recovery; L4: environment-state recovery. These four columns do not average to the overall recovery statistic used in the controlled experiments.",
      "columns": [
        "L1",
        "L2",
        "L3",
        "L4"
      ],
      "keys": [
        "l1",
        "l2",
        "l3",
        "l4"
      ],
      "units": [
        "%",
        "%",
        "%",
        "%"
      ],
      "sourceType": "published_context",
      "citation": "LIBERO-RECOVER (Liu et al., 2026).",
      "rows": [
        {
          "id": "pi0",
          "name": "π₀",
          "source": "LIBERO-RECOVER (Liu et al., 2026).",
          "values": {
            "l1": 46.1,
            "l2": 24.4,
            "l3": 13.5,
            "l4": 4.6
          }
        },
        {
          "id": "pi0-fast",
          "name": "π₀-FAST",
          "source": "LIBERO-RECOVER (Liu et al., 2026).",
          "values": {
            "l1": 40,
            "l2": 22.9,
            "l3": 16.1,
            "l4": 3.7
          }
        },
        {
          "id": "gr00t-n1-5",
          "name": "GR00T-N1.5",
          "source": "LIBERO-RECOVER (Liu et al., 2026).",
          "values": {
            "l1": 30.7,
            "l2": 25.3,
            "l3": 15.2,
            "l4": 1.5
          }
        },
        {
          "id": "openvla-oft",
          "name": "OpenVLA-OFT",
          "source": "LIBERO-RECOVER (Liu et al., 2026).",
          "values": {
            "l1": 47.7,
            "l2": 26.6,
            "l3": 14.2,
            "l4": 5.6
          }
        },
        {
          "id": "wan2-policy",
          "name": "Wan2-Policy",
          "source": "LIBERO-RECOVER (Liu et al., 2026).",
          "values": {
            "l1": 17.8,
            "l2": 16.2,
            "l3": 12.5,
            "l4": 3.9
          }
        },
        {
          "id": "cosmos-predict2-policy",
          "name": "Cosmos-Predict2-Policy",
          "source": "LIBERO-RECOVER (Liu et al., 2026).",
          "values": {
            "l1": 15.5,
            "l2": 14.5,
            "l3": 9.9,
            "l4": 0
          }
        },
        {
          "id": "vlcot",
          "name": "VLCoT",
          "source": "This work",
          "values": {
            "l1": 52,
            "l2": 32,
            "l3": 20,
            "l4": 9
          }
        }
      ]
    },
    "components": {
      "id": "components",
      "label": "tab:component_ablation",
      "number": "5",
      "page": 8,
      "title": "Controlled component ablations",
      "context": "Measured means from this work. Controlled implementations use matched initialization, splits, training trajectories, observation access and execution intervals. Recovery first pools instances within each suite, then equally averages the four suites. N/A denotes the absence of a latent-stage verification interface.",
      "columns": [
        "LIBERO avg.",
        "LIBERO Long",
        "Recovery",
        "Stage F1"
      ],
      "keys": [
        "average",
        "long",
        "recovery",
        "f1"
      ],
      "units": [
        "%",
        "%",
        "%",
        "%"
      ],
      "sourceType": "controlled",
      "rows": [
        {
          "id": "base",
          "name": "Base policy: no latent reasoning",
          "source": "This work",
          "values": {
            "average": 96.8,
            "long": 94.2,
            "recovery": 20.5,
            "f1": null
          }
        },
        {
          "id": "auxiliary",
          "name": "Auxiliary state supervision only",
          "source": "This work",
          "values": {
            "average": 97.1,
            "long": 94.8,
            "recovery": 21.3,
            "f1": null
          }
        },
        {
          "id": "no-alignment",
          "name": "Without latent stage alignment",
          "source": "This work",
          "values": {
            "average": 97.2,
            "long": 95,
            "recovery": 22.1,
            "f1": 81.6
          }
        },
        {
          "id": "no-repair",
          "name": "Execution-feedback repair disabled",
          "source": "This work",
          "values": {
            "average": 97.5,
            "long": 95.8,
            "recovery": 22.8,
            "f1": 92.4
          }
        },
        {
          "id": "complete",
          "name": "Complete method",
          "source": "This work",
          "values": {
            "average": 98,
            "long": 96.8,
            "recovery": 26,
            "f1": 92.4
          }
        }
      ]
    },
    "repair": {
      "id": "repair",
      "label": "tab:repair_strategies",
      "number": "6",
      "page": 8,
      "title": "Measured repair strategies",
      "context": "Same recovery instances, step limits and task-level budget. Mean complete-episode inference time includes all model calls and both successful and failed episodes. Shared visual encoding is counted once. p95 is observation-to-executable-action latency. Regenerated vectors count deviation-triggered repairs only, excluding ordinary rolling extensions.",
      "columns": [
        "Recovery",
        "Time / episode",
        "p95 latency",
        "Vectors / episode"
      ],
      "keys": [
        "recovery",
        "time",
        "latency",
        "vectors"
      ],
      "units": [
        "%",
        "s",
        "ms",
        "vectors"
      ],
      "sourceType": "controlled",
      "rows": [
        {
          "id": "disabled",
          "name": "Repair disabled",
          "source": "This work",
          "values": {
            "recovery": 22.8,
            "time": 11.8,
            "latency": 180,
            "vectors": 0
          }
        },
        {
          "id": "full",
          "name": "Full replanning",
          "source": "This work",
          "values": {
            "recovery": 26.4,
            "time": 18.6,
            "latency": 820,
            "vectors": 62.4
          }
        },
        {
          "id": "current",
          "name": "Current-stage-only suffix reconstruction",
          "source": "This work",
          "values": {
            "recovery": 24.6,
            "time": 14.2,
            "latency": 360,
            "vectors": 23.6
          }
        },
        {
          "id": "local",
          "name": "Dependency-aware local repair",
          "source": "This work",
          "values": {
            "recovery": 26,
            "time": 14.9,
            "latency": 430,
            "vectors": 31.2
          }
        }
      ]
    },
    "augmentation": {
      "id": "augmentation",
      "label": "tab:recover_ablation",
      "number": "7",
      "page": 8,
      "title": "Recovery data and observation history",
      "context": "Each augmentation is independently compared with the original configuration. Recovery instances are pooled within each suite before equally averaging the four suites. Recovery-data and initial-frame gains cannot be added to predict a joint result.",
      "columns": [
        "Original",
        "+ Recovery data",
        "+ Initial frame"
      ],
      "keys": [
        "original",
        "recoveryData",
        "initialFrame"
      ],
      "units": [
        "%",
        "%",
        "%"
      ],
      "sourceType": "published_context",
      "citation": "Published baselines: LIBERO-RECOVER (Liu et al., 2026).",
      "rows": [
        {
          "id": "openvla-oft",
          "name": "OpenVLA-OFT",
          "source": "Published baselines: LIBERO-RECOVER (Liu et al., 2026).",
          "values": {
            "original": 20.8,
            "recoveryData": 25.4,
            "initialFrame": 26.8
          }
        },
        {
          "id": "gr00t-n1-5",
          "name": "GR00T-N1.5",
          "source": "Published baselines: LIBERO-RECOVER (Liu et al., 2026).",
          "values": {
            "original": 17.8,
            "recoveryData": 21.2,
            "initialFrame": 18.9
          }
        },
        {
          "id": "vlcot",
          "name": "VLCoT",
          "source": "This work",
          "values": {
            "original": 26,
            "recoveryData": 31,
            "initialFrame": 28
          }
        }
      ]
    }
  },
  "derived": {
    "timeReduction": {
      "value": 19.892473118279575,
      "unit": "%",
      "formula": "(18.6 - 14.9) / 18.6 * 100",
      "source": "repair.full.time, repair.local.time",
      "meaning": "Lower mean complete-episode inference time relative to full replanning."
    },
    "recoveryGain": {
      "value": 3.1999999999999993,
      "unit": "pp",
      "formula": "26.0 - 22.8",
      "source": "repair.local.recovery, repair.disabled.recovery",
      "meaning": "Recovery increase relative to repair disabled."
    },
    "recoveryGap": {
      "value": 0.3999999999999986,
      "unit": "pp",
      "formula": "26.4 - 26.0",
      "source": "repair.full.recovery, repair.local.recovery",
      "meaning": "Recovery of local repair is lower than full replanning by this amount."
    },
    "latencyReduction": {
      "value": 47.5609756097561,
      "unit": "%",
      "formula": "(820 - 430) / 820 * 100",
      "source": "repair.full.latency, repair.local.latency",
      "meaning": "Lower p95 decision latency relative to full replanning."
    },
    "vectorReduction": {
      "value": 50,
      "unit": "%",
      "formula": "(62.4 - 31.2) / 62.4 * 100",
      "source": "repair.full.vectors, repair.local.vectors",
      "meaning": "Fewer mean regenerated latent vectors relative to full replanning."
    }
  }
}
