{
  "schema_version": "anchor.robodojo.comparison.v1",
  "generated_at": "2026-07-20",
  "claim": "Two official fold_clothes demonstrations reach a visually folded endpoint, while step-level proxies reveal different execution cost and final-shape quality.",
  "source": {
    "dataset": "RoboDojo-Benchmark/RoboDojo",
    "dataset_revision": "1a3c4c334aef294c31d7a0190d8d6dff68df78e0",
    "dataset_license_metadata": "apache-2.0",
    "task_id": "fold_clothes",
    "robot": "arx_x5",
    "category": "Generalization",
    "data_source": "DateGen",
    "task_definition": "https://robodojo-benchmark.com/doc/sim-tasks/fold-clothes/",
    "dataset_path": "https://huggingface.co/datasets/RoboDojo-Benchmark/RoboDojo/tree/main/data/RoboDojo/fold_clothes/arx_x5",
    "scoring_code": "https://github.com/RoboDojo-Benchmark/RoboDojo/blob/main/task/RoboDojo/tasks/fold_clothes.py"
  },
  "comparison_design": {
    "type": "cohort_examples",
    "official_demonstrations": true,
    "controlled_same_seed_policy_ab": false,
    "official_score_rerun": false,
    "robustness_claim": "none; two-example investor demonstration"
  },
  "episodes": {
    "efficient": {
      "local_episode": 36,
      "lerobot_global_episode": 936,
      "frames": 301,
      "fps": 25,
      "video_duration_seconds": 12.04,
      "trajectory_duration_seconds": 12.0,
      "action_path_l2": 27.189678,
      "action_delta_sign_change_reversals": 293,
      "final_visual_proxy": {
        "rectangularity": 0.8345,
        "solidity": 0.9208,
        "compactness": 0.6005
      },
      "camera_frame_sync": {
        "head": 301,
        "left_wrist": 301,
        "right_wrist": 301
      }
    },
    "detour": {
      "local_episode": 93,
      "lerobot_global_episode": 993,
      "frames": 314,
      "fps": 25,
      "video_duration_seconds": 12.56,
      "trajectory_duration_seconds": 12.52,
      "action_path_l2": 30.233953,
      "action_delta_sign_change_reversals": 316,
      "final_visual_proxy": {
        "rectangularity": 0.7883,
        "solidity": 0.904,
        "compactness": 0.5675
      },
      "camera_frame_sync": {
        "head": 314,
        "left_wrist": 314,
        "right_wrist": 314
      }
    }
  },
  "relative_result": {
    "efficient_time_reduction_percent_vs_detour": 4.15,
    "efficient_action_path_reduction_percent_vs_detour": 10.07,
    "efficient_reversal_reduction_percent_vs_detour": 7.28,
    "efficient_rectangularity_increase_percent_vs_detour": 5.86,
    "efficient_compactness_increase_percent_vs_detour": 5.81
  },
  "metric_notes": {
    "action_path_l2": "Sum of L2 distances between adjacent 14-dimensional action vectors; it is not Cartesian distance in meters.",
    "reversals": "Count of sign changes in per-joint action deltas; a motion-correction proxy, not a semantic retry count.",
    "final_visual_proxy": "Largest non-table component in a central crop of the final head RGB frame. No ground-truth garment mask or target silhouette was available.",
    "official_score": "unscored; the RoboDojo evaluator was not rerun for these demonstrations"
  },
  "excluded_claims": [
    "The two episodes are not a controlled same-seed, same-layout, two-policy A/B.",
    "The page does not claim that either episode received an official RoboDojo score of 100.",
    "Visual proxies are not official reward, success, or human preference.",
    "Two examples do not establish cross-seed robustness."
  ]
}
