Shiki42 commited on
Commit
7b4c0ee
·
verified ·
1 Parent(s): 7344f74

Publish E001 action chunk recovery artifacts

Browse files
Files changed (25) hide show
  1. README.md +22 -0
  2. SHA256SUMS.json +103 -0
  3. experiments/E001/E001-CARD.md +27 -0
  4. experiments/E001/audits/action-target-audit.json +64 -0
  5. experiments/E001/audits/data-split-audit.json +48 -0
  6. experiments/E001/audits/real-data-smoke.json +32 -0
  7. experiments/E001/audits/residual-zero-init-smoke.json +17 -0
  8. experiments/E001/conditions/c-endpoint-3c2dae9044/CONDITION_CARD.md +13 -0
  9. experiments/E001/conditions/c-endpoint-3c2dae9044/best.pt +3 -0
  10. experiments/E001/conditions/c-endpoint-3c2dae9044/config_resolved.yaml +99 -0
  11. experiments/E001/conditions/c-endpoint-3c2dae9044/evaluation_test.json +102 -0
  12. experiments/E001/conditions/c-endpoint-3c2dae9044/evaluation_validation.json +102 -0
  13. experiments/E001/conditions/c-endpoint-3c2dae9044/run_manifest.yaml +92 -0
  14. experiments/E001/conditions/c-residual-h16-be2e741500/CONDITION_CARD.md +13 -0
  15. experiments/E001/conditions/c-residual-h16-be2e741500/best.pt +3 -0
  16. experiments/E001/conditions/c-residual-h16-be2e741500/config_resolved.yaml +103 -0
  17. experiments/E001/conditions/c-residual-h16-be2e741500/evaluation_test.json +119 -0
  18. experiments/E001/conditions/c-residual-h16-be2e741500/evaluation_validation.json +119 -0
  19. experiments/E001/conditions/c-residual-h16-be2e741500/run_manifest.yaml +104 -0
  20. experiments/E001/experiment-summary.json +408 -0
  21. experiments/E001/runs/E001-R002/evaluation_validation.json +116 -0
  22. experiments/E001/runs/E001-R002/run_manifest.yaml +102 -0
  23. experiments/E001/runs/E001-R003-invalid/metrics.jsonl +0 -0
  24. experiments/E001/runs/E001-R003-invalid/run_manifest.yaml +110 -0
  25. experiments/E001/selection-validation.json +55 -0
README.md ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: other
3
+ library_name: pytorch
4
+ tags:
5
+ - robotics
6
+ - action-chunking
7
+ - se3
8
+ - robotwin
9
+ datasets:
10
+ - Shiki42/up-vla-precision-recovery-15k
11
+ ---
12
+
13
+ # UP-VLA Action Chunk Recovery Checkpoints
14
+
15
+ Protocol-registered E001 artifacts for endpoint, direct H16, and residual H16 recovery.
16
+ The selected residual model improved grouped test path score by 4.68%, below the
17
+ pre-registered 10% validation gate; E002 was not opened.
18
+
19
+ The source dataset includes RLBench-derived material with non-commercial research
20
+ restrictions. Treat these checkpoints as research-only unless all upstream license
21
+ obligations have been independently resolved. See experiments/E001/E001-CARD.md
22
+ and the immutable run/evaluation receipts for details.
SHA256SUMS.json ADDED
@@ -0,0 +1,103 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "experiment_id": "E001",
3
+ "file_count": 24,
4
+ "files": {
5
+ "README.md": {
6
+ "bytes": 719,
7
+ "sha256": "929577f73e70f2a9404890a64881e3aefb423dd06f8cba19d946957a9222c46b"
8
+ },
9
+ "experiments/E001/E001-CARD.md": {
10
+ "bytes": 1516,
11
+ "sha256": "0ee76cd69ac690653483f469f55ebaf16c3ebcc5c9f841a480909bed2ae87e6e"
12
+ },
13
+ "experiments/E001/audits/action-target-audit.json": {
14
+ "bytes": 1741,
15
+ "sha256": "16e1b62f9f54fb0ac8a0fab0b7772e66ab6b6bc0d930fc5aa30f53ff19258fa6"
16
+ },
17
+ "experiments/E001/audits/data-split-audit.json": {
18
+ "bytes": 1360,
19
+ "sha256": "db98be16cd36ccb1726d5407202b126f8ed147e31144ba6c1228dc163e8be353"
20
+ },
21
+ "experiments/E001/audits/real-data-smoke.json": {
22
+ "bytes": 934,
23
+ "sha256": "98502ae238c997ed151d99bbf8caf1c0e1bfad04be181f422deef73fc8fa4505"
24
+ },
25
+ "experiments/E001/audits/residual-zero-init-smoke.json": {
26
+ "bytes": 671,
27
+ "sha256": "6aa37549464368866c756979073f38f79b45cf567462e19b8155eba5fa255c30"
28
+ },
29
+ "experiments/E001/conditions/c-endpoint-3c2dae9044/CONDITION_CARD.md": {
30
+ "bytes": 559,
31
+ "sha256": "4b2ef7a1cb78c451600706169a40d2deff071264cf5fee1374c2dd2a2c34d7a2"
32
+ },
33
+ "experiments/E001/conditions/c-endpoint-3c2dae9044/best.pt": {
34
+ "bytes": 189704503,
35
+ "sha256": "182fc8544a1489582c7f96e29625b7e0048642edf6de749c50b19da0da33379f"
36
+ },
37
+ "experiments/E001/conditions/c-endpoint-3c2dae9044/config_resolved.yaml": {
38
+ "bytes": 2741,
39
+ "sha256": "a2626ca28030873c0ce82e3f9a2c6057e976f3f89d9c693c4c48561b02183b9e"
40
+ },
41
+ "experiments/E001/conditions/c-endpoint-3c2dae9044/evaluation_test.json": {
42
+ "bytes": 3871,
43
+ "sha256": "4bb067239611e3f9427ba5fffdc8c0b56a30452f6db8e0a02b2ade56a2b63e43"
44
+ },
45
+ "experiments/E001/conditions/c-endpoint-3c2dae9044/evaluation_validation.json": {
46
+ "bytes": 3879,
47
+ "sha256": "2733fe8093af6a9fad8f9c6988caee1cd048412c3620ad193779a27bcdbe4972"
48
+ },
49
+ "experiments/E001/conditions/c-endpoint-3c2dae9044/run_manifest.yaml": {
50
+ "bytes": 3472,
51
+ "sha256": "6548cd5b4b3335d5b5f88262924e34a8fcbf0621216792bc946bad804026c418"
52
+ },
53
+ "experiments/E001/conditions/c-residual-h16-be2e741500/CONDITION_CARD.md": {
54
+ "bytes": 567,
55
+ "sha256": "c244e4d9a3b123628c29e3bba41e378a38f48ad6d24fe60fb62bba7d86b35792"
56
+ },
57
+ "experiments/E001/conditions/c-residual-h16-be2e741500/best.pt": {
58
+ "bytes": 238782079,
59
+ "sha256": "81ef4841165e996bf7c94556e3f83b457ae5b021bfa6c99ae2fa4998e3f111f3"
60
+ },
61
+ "experiments/E001/conditions/c-residual-h16-be2e741500/config_resolved.yaml": {
62
+ "bytes": 2955,
63
+ "sha256": "36d1731639d935f7cd94d6a01c67205577985242498517a6bcf046f78ce0bdde"
64
+ },
65
+ "experiments/E001/conditions/c-residual-h16-be2e741500/evaluation_test.json": {
66
+ "bytes": 4660,
67
+ "sha256": "7f77315929e49166cf976cc00d3f8de8b157a35a7bdd17a507763bca2d6bc813"
68
+ },
69
+ "experiments/E001/conditions/c-residual-h16-be2e741500/evaluation_validation.json": {
70
+ "bytes": 4657,
71
+ "sha256": "9373e8bd953247345671b604f00945aa5649b2a8887a94136547e3de77f6a0d4"
72
+ },
73
+ "experiments/E001/conditions/c-residual-h16-be2e741500/run_manifest.yaml": {
74
+ "bytes": 4186,
75
+ "sha256": "c97c0ce3f3ebfeccce4c75b1f9d9a039f462fa2812b252169a27e8f694667fe7"
76
+ },
77
+ "experiments/E001/experiment-summary.json": {
78
+ "bytes": 16356,
79
+ "sha256": "bc3ef040d1f2d9e65347a56c1079e027dbde9c6aaee39ea48b10f1503878aa7d"
80
+ },
81
+ "experiments/E001/runs/E001-R002/evaluation_validation.json": {
82
+ "bytes": 4349,
83
+ "sha256": "e69679a21e423a975480369d3c9773a812bff06298f81ab6a26e2004fe1bc338"
84
+ },
85
+ "experiments/E001/runs/E001-R002/run_manifest.yaml": {
86
+ "bytes": 4001,
87
+ "sha256": "ba17d383fee1444110197648235d00d447ef14055df8f7ab9280c84486316fd7"
88
+ },
89
+ "experiments/E001/runs/E001-R003-invalid/metrics.jsonl": {
90
+ "bytes": 195729,
91
+ "sha256": "7cbed4f6a4fe6871e0ea2ecadeec972e56775ef1220ea1903c7beb672ede8a96"
92
+ },
93
+ "experiments/E001/runs/E001-R003-invalid/run_manifest.yaml": {
94
+ "bytes": 4259,
95
+ "sha256": "b27935c72939dc5112d422ab19c9b422b539bd041cc4ad6f3c2f5546001b24e8"
96
+ },
97
+ "experiments/E001/selection-validation.json": {
98
+ "bytes": 2021,
99
+ "sha256": "fc742f723daf1f6e91b78db104c78c47056ad965fc0510de8ea9fe4964d868b3"
100
+ }
101
+ },
102
+ "schema_version": "upvla.checkpoint_package.v1"
103
+ }
experiments/E001/E001-CARD.md ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # E001 — Validated recovery Action Chunk ablation
2
+
3
+ Status: **completed / inconclusive**
4
+ Selected condition: **E001-R004 residual H16**
5
+ Advance to E002: **no**
6
+
7
+ ## Decision
8
+
9
+ The selected residual H16 condition improved validation group-macro path score from 0.829670 to 0.797646 (3.86%), below the pre-registered 10% gate. The endpoint and latency gates passed. Test was opened once for the frozen baseline and selected condition: score 0.993277 to 0.946838 (4.68%).
10
+
11
+ ## Valid conditions
12
+
13
+ | Run | Head | Validation score | Endpoint mm | Endpoint deg | p95 ms |
14
+ |---|---|---:|---:|---:|---:|
15
+ | E001-R001 | endpoint + profiled path | 0.829670 | 6.062 | 3.583 | n/a |
16
+ | E001-R002 | direct H16 | 0.804941 | 5.990 | 3.503 | 4.342 |
17
+ | E001-R004 | residual H16 | 0.797646 | 5.908 | 3.453 | 4.698 |
18
+
19
+ R003 is invalid and excluded: random residual-head initialization saturated at 50 mm / 0.52 rad. R004 prospectively recorded the identity-initialization retry and passed a real-proposal smoke test.
20
+
21
+ ## Test result
22
+
23
+ The selected residual model changed endpoint error from 7.017 mm / 4.420° to 6.797 mm / 4.175°. It increased command jerk RMS from 2.311 to 4.300; a future experiment should constrain residual temporal structure before real Pi0.5 rollout.
24
+
25
+ ## Data and independence
26
+
27
+ 15,000 physically successful recoveries are grouped into 111 independent source groups. Split: train 11,628/89 groups; validation 1,970/11; test 1,402/11. Each holdout contains both RoboTwin and RLBench. No source group crosses splits.
experiments/E001/audits/action-target-audit.json ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "configured_limits": {
3
+ "step_rotation_rad": 0.2,
4
+ "step_translation_m": 0.01
5
+ },
6
+ "expert_success_counts": {
7
+ "recovery_task_success": 15000,
8
+ "source_expert_success": 15000
9
+ },
10
+ "git_commit": "91b8c9473cfc60c8bfce3848c900aba8e4594ca1",
11
+ "horizon_reaches_target_count": 0,
12
+ "horizon_seconds": 0.8,
13
+ "horizon_steps": 16,
14
+ "limit_violations": {
15
+ "step_rotation": 0,
16
+ "step_translation": 0
17
+ },
18
+ "sample_count": 15000,
19
+ "schema_version": "precision.action_chunk.target_audit.v1",
20
+ "source_families": {
21
+ "rlbench": 5220,
22
+ "robotwin_generic": 9780
23
+ },
24
+ "step_rotation_rad": {
25
+ "max": 0.010471975511966547,
26
+ "mean": 0.0043272625200924675,
27
+ "min": 0.000270797193020592,
28
+ "p50": 0.0037907259009449306,
29
+ "p95": 0.00872664625997162,
30
+ "p99": 0.010053904361015777
31
+ },
32
+ "step_translation_twist_m": {
33
+ "max": 0.0006000026630050273,
34
+ "mean": 0.0003779863550717578,
35
+ "min": 1.719141227820467e-05,
36
+ "p50": 0.0004000002360415781,
37
+ "p95": 0.000600000146014978,
38
+ "p99": 0.0006000012005080267
39
+ },
40
+ "trajectory_duration_s": {
41
+ "max": 20.731572870249437,
42
+ "mean": 6.3629103453163545,
43
+ "min": 1.2569543470285411,
44
+ "p50": 5.903310058639175,
45
+ "p95": 11.372092291086012,
46
+ "p99": 14.754663633027155
47
+ },
48
+ "waypoint_rotation_rad": {
49
+ "max": 0.16755160819145598,
50
+ "mean": 0.037987402154382004,
51
+ "min": 0.0004985226170931637,
52
+ "p50": 0.030744580054978578,
53
+ "p95": 0.09772211052589472,
54
+ "p99": 0.1255383529084257
55
+ },
56
+ "waypoint_translation_m": {
57
+ "max": 0.00960000000000014,
58
+ "mean": 0.0033021575027860745,
59
+ "min": 1.719141112148997e-05,
60
+ "p50": 0.0029999999999999884,
61
+ "p95": 0.007499999999999985,
62
+ "p99": 0.008399999999999987
63
+ }
64
+ }
experiments/E001/audits/data-split-audit.json ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "data_root": "/home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples",
3
+ "extraction_receipt": "/home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples/_extraction_receipt.json",
4
+ "extraction_receipt_sha256": "ff05bdb511990c0f951cb17efb1f84cc4c4a48e59f875ebd6d66a468c3b840dd",
5
+ "git_commit": "91b8c9473cfc60c8bfce3848c900aba8e4594ca1",
6
+ "schema_version": "precision_recovery.split_audit.v1",
7
+ "split_protocol": "source-family-stratified-group-hash-v1",
8
+ "splits": {
9
+ "test": {
10
+ "groups": 11,
11
+ "groups_by_source_family": {
12
+ "rlbench": 1,
13
+ "robotwin_generic": 10
14
+ },
15
+ "samples": 1402,
16
+ "samples_by_source_family": {
17
+ "rlbench": 520,
18
+ "robotwin_generic": 882
19
+ }
20
+ },
21
+ "train": {
22
+ "groups": 89,
23
+ "groups_by_source_family": {
24
+ "rlbench": 9,
25
+ "robotwin_generic": 80
26
+ },
27
+ "samples": 11628,
28
+ "samples_by_source_family": {
29
+ "rlbench": 4180,
30
+ "robotwin_generic": 7448
31
+ }
32
+ },
33
+ "validation": {
34
+ "groups": 11,
35
+ "groups_by_source_family": {
36
+ "rlbench": 1,
37
+ "robotwin_generic": 10
38
+ },
39
+ "samples": 1970,
40
+ "samples_by_source_family": {
41
+ "rlbench": 520,
42
+ "robotwin_generic": 1450
43
+ }
44
+ }
45
+ },
46
+ "total_groups": 111,
47
+ "total_samples": 15000
48
+ }
experiments/E001/audits/real-data-smoke.json ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "data_receipt_sha256": "ff05bdb511990c0f951cb17efb1f84cc4c4a48e59f875ebd6d66a468c3b840dd",
3
+ "device": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
4
+ "direct_h16": {
5
+ "batch2_latency_max_ms": 6.298548221588135,
6
+ "batch2_latency_median_ms": 4.577220916748047,
7
+ "finite": true,
8
+ "gradient_norm": 738.1722412109375,
9
+ "loss": 12.037626266479492,
10
+ "maximum_predicted_step_rotation_rad": 0.13702814280986786,
11
+ "maximum_predicted_step_translation_m": 0.007430041674524546,
12
+ "prediction_shape": [
13
+ 2,
14
+ 16,
15
+ 6
16
+ ],
17
+ "trainable_parameters": 19806278
18
+ },
19
+ "endpoint": {
20
+ "finite": true,
21
+ "loss": 0.25888946652412415,
22
+ "trainable_parameters": 15790420
23
+ },
24
+ "git_commit": "91b8c9473cfc60c8bfce3848c900aba8e4594ca1",
25
+ "sample_ids": [
26
+ "sample_000000",
27
+ "sample_000001",
28
+ "sample_000003",
29
+ "sample_000004"
30
+ ],
31
+ "schema_version": "precision.action_chunk.smoke.v1"
32
+ }
experiments/E001/audits/residual-zero-init-smoke.json ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "git_commit": "63265740e2e0caa4c6f42da6d497fd63f46d522d",
3
+ "gradient_norm_before_clip": 35.3123664855957,
4
+ "initial_command_max_abs_error": 3.725290298461914e-09,
5
+ "initial_residual_rotation_max_rad": 0.0,
6
+ "initial_residual_translation_max_m": 0.0,
7
+ "loss": 0.304883748292923,
8
+ "proposal_manifest_sha256": "b2a7fda405bb7369a91d0e74040617b08704ed908f84ab653315a6a23da7bc50",
9
+ "sample_ids": [
10
+ "sample_000000",
11
+ "sample_000001"
12
+ ],
13
+ "saturated_after_update": false,
14
+ "schema_version": "precision.action_chunk.residual_smoke.v1",
15
+ "updated_residual_rotation_max_rad": 0.0484287329018116,
16
+ "updated_residual_translation_max_m": 0.0043397932313382626
17
+ }
experiments/E001/conditions/c-endpoint-3c2dae9044/CONDITION_CARD.md ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Endpoint baseline + profiled controller
2
+
3
+ - Run: E001-R001
4
+ - Condition: c-endpoint-3c2dae9044
5
+ - Checkpoint SHA-256: 182fc8544a1489582c7f96e29625b7e0048642edf6de749c50b19da0da33379f
6
+ - Dataset revision: 77f5f306303361d602ab10759b3a139c67f2cf16
7
+ - Split: source-family-stratified grouped hash
8
+ - Horizon: 16 x 0.05 s for chunk conditions
9
+ - Frame/composition: current-tool SE(3), right multiplication
10
+ - Intended use: research evaluation only
11
+
12
+ See the sibling run manifest and evaluation receipts. Test data was opened exactly
13
+ once after validation-only selection.
experiments/E001/conditions/c-endpoint-3c2dae9044/best.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:182fc8544a1489582c7f96e29625b7e0048642edf6de749c50b19da0da33379f
3
+ size 189704503
experiments/E001/conditions/c-endpoint-3c2dae9044/config_resolved.yaml ADDED
@@ -0,0 +1,99 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ data:
2
+ hf_repo_id: Shiki42/up-vla-precision-recovery-15k
3
+ revision: 77f5f306303361d602ab10759b3a139c67f2cf16
4
+ variant_id: validated-physical-recovery-15k-v1
5
+ manifest_sha256: 4cda13bc86038829f5fd519c7c766d40e770ad915974f68afea6f251b59faa50
6
+ dataset_info_sha256: 99a2f93632799a2b0a5da785438a55934e7fbaac582de0d63f8ddc0186b10ef3
7
+ sample_count: 15000
8
+ split_protocol: source-family-stratified-group-hash-v1
9
+ validation_fraction: 0.1
10
+ test_fraction: 0.1
11
+ geometry:
12
+ cumulative_waypoint_representation: current_tool_relative_pose
13
+ step_representation: integrated_body_twist
14
+ step_integration: right_se3_exponential
15
+ residual_composition: base_right_multiply_se3_exponential
16
+ base_input_normalization: translation_rotation_physical_scales
17
+ model:
18
+ backbone:
19
+ hidden_size: 256
20
+ fusion_layers: 4
21
+ fusion_heads: 8
22
+ fusion_mlp_ratio: 4
23
+ dropout: 0.1
24
+ vision_backend: conv
25
+ text_backend: embedding
26
+ pretrained_model_name: google/siglip-base-patch16-224
27
+ freeze_pretrained: true
28
+ image_size: 224
29
+ vision_tokens: 64
30
+ depth_tokens: 36
31
+ point_tokens: 32
32
+ point_samples: 256
33
+ max_text_tokens: 32
34
+ text_vocab_size: 32768
35
+ max_joints: 7
36
+ aloha_joints: 6
37
+ panda_joints: 7
38
+ translation_scale_m: 0.05
39
+ rotation_scale_rad: 0.5235987755982988
40
+ joint_scale_rad: 0.5
41
+ horizon_steps: 16
42
+ horizon_seconds: 0.8
43
+ decoder_layers: 4
44
+ decoder_heads: 8
45
+ decoder_mlp_ratio: 4
46
+ maximum_step_translation_m: 0.01
47
+ maximum_step_rotation_rad: 0.2
48
+ maximum_residual_translation_m: 0.05
49
+ maximum_residual_rotation_rad: 0.52
50
+ action_head: endpoint
51
+ base_conditioning: none
52
+ parameter_counts:
53
+ total: 15790420
54
+ trainable: 15790420
55
+ frozen: 0
56
+ loss:
57
+ translation: 1.0
58
+ rotation_vector: 1.0
59
+ rotation_geodesic: 0.25
60
+ joint: 0.5
61
+ gripper: 0.1
62
+ training:
63
+ seed: 42
64
+ batch_size: 64
65
+ gradient_accumulation_steps: 1
66
+ epochs: 100
67
+ max_steps: 3000
68
+ learning_rate: 0.0003
69
+ weight_decay: 0.05
70
+ warmup_fraction: 0.05
71
+ gradient_clip_norm: 1.0
72
+ precision: bf16
73
+ workers: 8
74
+ log_every_steps: 10
75
+ validation_every_steps: 250
76
+ checkpoint_every_steps: 500
77
+ evaluation:
78
+ version: action-chunk-eval-v1
79
+ trajectory_loss:
80
+ path_translation: 1.0
81
+ path_rotation: 1.0
82
+ endpoint_translation: 2.0
83
+ endpoint_rotation: 2.0
84
+ velocity: 1.0
85
+ acceleration: 0.1
86
+ jerk: 0.05
87
+ residual_magnitude: 0.02
88
+ selection_split: validation
89
+ test_firewall_opened: false
90
+ runtime:
91
+ data_root: /home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples
92
+ base_proposal_root: null
93
+ output_directory: /home/coder/share/experiment-runs/E001/E001-R001
94
+ split_counts:
95
+ train_samples: 11628
96
+ validation_samples: 1970
97
+ train_groups: 89
98
+ validation_groups: 11
99
+ overlap: false
experiments/E001/conditions/c-endpoint-3c2dae9044/evaluation_test.json ADDED
@@ -0,0 +1,102 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "base_proposal_manifest": {
3
+ "path": "/home/coder/share/experiment-artifacts/E001/R001-profiled-proposals/manifest.json",
4
+ "sha256": "b2a7fda405bb7369a91d0e74040617b08704ed908f84ab653315a6a23da7bc50"
5
+ },
6
+ "checkpoint": null,
7
+ "checkpoint_run": null,
8
+ "condition_id": "c-endpoint-3c2dae9044",
9
+ "data_extraction_receipt": {
10
+ "path": "/home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples/_extraction_receipt.json",
11
+ "sha256": "ff05bdb511990c0f951cb17efb1f84cc4c4a48e59f875ebd6d66a468c3b840dd"
12
+ },
13
+ "experiment_id": "E001",
14
+ "group_count": 11,
15
+ "grouped_metrics": {
16
+ "acceleration_rmse": 0.11615525931119919,
17
+ "command_acceleration_rms": 0.09448793530464172,
18
+ "command_jerk_rms": 2.311239004135132,
19
+ "endpoint_rotation_rmse_deg": 4.419965744018555,
20
+ "endpoint_translation_rmse_mm": 7.017010688781738,
21
+ "jerk_rmse": 2.8706088066101074,
22
+ "path_rotation_rmse_deg": 2.7911133766174316,
23
+ "path_translation_rmse_mm": 4.35053825378418,
24
+ "residual_rotation_mean_deg": 0.0,
25
+ "residual_translation_mean_mm": 0.0,
26
+ "velocity_rmse": 0.041266847401857376
27
+ },
28
+ "latency": null,
29
+ "losses": {
30
+ "acceleration": 0.0009799118852242827,
31
+ "endpoint_rotation": 0.12461857497692108,
32
+ "endpoint_translation": 0.0036088007036596537,
33
+ "jerk": 0.0014138123951852322,
34
+ "path_rotation": 0.06947080045938492,
35
+ "path_translation": 0.0013824889902025461,
36
+ "residual_magnitude": 0.0,
37
+ "total": 0.39481988549232483,
38
+ "velocity": 0.06734314560890198
39
+ },
40
+ "metrics": {
41
+ "acceleration_rmse": 0.11698303371667862,
42
+ "command_acceleration_rms": 0.09476716816425323,
43
+ "command_jerk_rms": 2.2194325923919678,
44
+ "endpoint_rotation_rmse_deg": 4.193507194519043,
45
+ "endpoint_translation_rmse_mm": 7.35744571685791,
46
+ "jerk_rmse": 2.8062472343444824,
47
+ "path_rotation_rmse_deg": 2.6185266971588135,
48
+ "path_translation_rmse_mm": 4.553826332092285,
49
+ "residual_rotation_mean_deg": 0.0,
50
+ "residual_translation_mean_mm": 0.0,
51
+ "velocity_rmse": 0.038922689855098724
52
+ },
53
+ "prediction_source": "base_proposal",
54
+ "primary_path_score": {
55
+ "definition": "path_translation_rmse_mm / 10 + path_rotation_rmse_deg / 5",
56
+ "group_macro": 0.9932765007019043,
57
+ "micro": 0.9790879726409911
58
+ },
59
+ "run_id": "E001-R001",
60
+ "sample_count": 1402,
61
+ "schema_version": "precision.action_chunk.evaluation.v1",
62
+ "source_family_metrics": {
63
+ "rlbench": {
64
+ "group_count": 1,
65
+ "metrics": {
66
+ "acceleration_rmse": 0.1142672672867775,
67
+ "command_acceleration_rms": 0.10241115093231201,
68
+ "command_jerk_rms": 2.1901450157165527,
69
+ "endpoint_rotation_rmse_deg": 4.396600723266602,
70
+ "endpoint_translation_rmse_mm": 6.084814548492432,
71
+ "jerk_rmse": 2.498075246810913,
72
+ "path_rotation_rmse_deg": 2.7127580642700195,
73
+ "path_translation_rmse_mm": 3.7321736812591553,
74
+ "residual_rotation_mean_deg": 0.0,
75
+ "residual_translation_mean_mm": 0.0,
76
+ "velocity_rmse": 0.04017214477062225
77
+ },
78
+ "path_score": 0.9157689809799194,
79
+ "sample_count": 520
80
+ },
81
+ "robotwin_generic": {
82
+ "group_count": 10,
83
+ "metrics": {
84
+ "acceleration_rmse": 0.11855501681566238,
85
+ "command_acceleration_rms": 0.08995665609836578,
86
+ "command_jerk_rms": 2.2365198135375977,
87
+ "endpoint_rotation_rmse_deg": 4.0690226554870605,
88
+ "endpoint_translation_rmse_mm": 8.01360034942627,
89
+ "jerk_rmse": 2.9730048179626465,
90
+ "path_rotation_rmse_deg": 2.561347007751465,
91
+ "path_translation_rmse_mm": 4.975062847137451,
92
+ "residual_rotation_mean_deg": 0.0,
93
+ "residual_translation_mean_mm": 0.0,
94
+ "velocity_rmse": 0.03816688805818558
95
+ },
96
+ "path_score": 1.0097756862640381,
97
+ "sample_count": 882
98
+ }
99
+ },
100
+ "split": "test",
101
+ "test_firewall_opened": true
102
+ }
experiments/E001/conditions/c-endpoint-3c2dae9044/evaluation_validation.json ADDED
@@ -0,0 +1,102 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "base_proposal_manifest": {
3
+ "path": "/home/coder/share/experiment-artifacts/E001/R001-profiled-proposals/manifest.json",
4
+ "sha256": "b2a7fda405bb7369a91d0e74040617b08704ed908f84ab653315a6a23da7bc50"
5
+ },
6
+ "checkpoint": null,
7
+ "checkpoint_run": null,
8
+ "condition_id": "c-endpoint-3c2dae9044",
9
+ "data_extraction_receipt": {
10
+ "path": "/home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples/_extraction_receipt.json",
11
+ "sha256": "ff05bdb511990c0f951cb17efb1f84cc4c4a48e59f875ebd6d66a468c3b840dd"
12
+ },
13
+ "experiment_id": "E001",
14
+ "group_count": 11,
15
+ "grouped_metrics": {
16
+ "acceleration_rmse": 0.1156676784157753,
17
+ "command_acceleration_rms": 0.09105820208787918,
18
+ "command_jerk_rms": 2.2417898178100586,
19
+ "endpoint_rotation_rmse_deg": 3.5833146572113037,
20
+ "endpoint_translation_rmse_mm": 6.061676025390625,
21
+ "jerk_rmse": 2.8809361457824707,
22
+ "path_rotation_rmse_deg": 2.2578845024108887,
23
+ "path_translation_rmse_mm": 3.7809324264526367,
24
+ "residual_rotation_mean_deg": 0.0,
25
+ "residual_translation_mean_mm": 0.0,
26
+ "velocity_rmse": 0.03386858478188515
27
+ },
28
+ "latency": null,
29
+ "losses": {
30
+ "acceleration": 0.0010245594894513488,
31
+ "endpoint_rotation": 0.110835961997509,
32
+ "endpoint_translation": 0.0026862621307373047,
33
+ "jerk": 0.0015703794779255986,
34
+ "path_rotation": 0.06236312538385391,
35
+ "path_translation": 0.0010549781145527959,
36
+ "residual_magnitude": 0.0,
37
+ "total": 0.34362855553627014,
38
+ "velocity": 0.052984997630119324
39
+ },
40
+ "metrics": {
41
+ "acceleration_rmse": 0.11067987978458405,
42
+ "command_acceleration_rms": 0.08661594241857529,
43
+ "command_jerk_rms": 2.1349382400512695,
44
+ "endpoint_rotation_rmse_deg": 3.794172763824463,
45
+ "endpoint_translation_rmse_mm": 6.347749710083008,
46
+ "jerk_rmse": 2.749969244003296,
47
+ "path_rotation_rmse_deg": 2.384471893310547,
48
+ "path_translation_rmse_mm": 3.9780237674713135,
49
+ "residual_rotation_mean_deg": 0.0,
50
+ "residual_translation_mean_mm": 0.0,
51
+ "velocity_rmse": 0.0354880653321743
52
+ },
53
+ "prediction_source": "base_proposal",
54
+ "primary_path_score": {
55
+ "definition": "path_translation_rmse_mm / 10 + path_rotation_rmse_deg / 5",
56
+ "group_macro": 0.8296701431274414,
57
+ "micro": 0.8746967554092407
58
+ },
59
+ "run_id": "E001-R001",
60
+ "sample_count": 1970,
61
+ "schema_version": "precision.action_chunk.evaluation.v1",
62
+ "source_family_metrics": {
63
+ "rlbench": {
64
+ "group_count": 1,
65
+ "metrics": {
66
+ "acceleration_rmse": 0.08557052165269852,
67
+ "command_acceleration_rms": 0.06839349865913391,
68
+ "command_jerk_rms": 1.6698498725891113,
69
+ "endpoint_rotation_rmse_deg": 4.503732681274414,
70
+ "endpoint_translation_rmse_mm": 7.368035793304443,
71
+ "jerk_rmse": 2.076235294342041,
72
+ "path_rotation_rmse_deg": 2.8150835037231445,
73
+ "path_translation_rmse_mm": 4.665040493011475,
74
+ "residual_rotation_mean_deg": 0.0,
75
+ "residual_translation_mean_mm": 0.0,
76
+ "velocity_rmse": 0.04111722111701965
77
+ },
78
+ "path_score": 1.0295207500457764,
79
+ "sample_count": 520
80
+ },
81
+ "robotwin_generic": {
82
+ "group_count": 10,
83
+ "metrics": {
84
+ "acceleration_rmse": 0.11839432269334793,
85
+ "command_acceleration_rms": 0.0922783836722374,
86
+ "command_jerk_rms": 2.278719425201416,
87
+ "endpoint_rotation_rmse_deg": 3.504887819290161,
88
+ "endpoint_translation_rmse_mm": 5.939308166503906,
89
+ "jerk_rmse": 2.9543890953063965,
90
+ "path_rotation_rmse_deg": 2.2096965312957764,
91
+ "path_translation_rmse_mm": 3.7007036209106445,
92
+ "residual_rotation_mean_deg": 0.0,
93
+ "residual_translation_mean_mm": 0.0,
94
+ "velocity_rmse": 0.033237893134355545
95
+ },
96
+ "path_score": 0.8120096683502198,
97
+ "sample_count": 1450
98
+ }
99
+ },
100
+ "split": "validation",
101
+ "test_firewall_opened": false
102
+ }
experiments/E001/conditions/c-endpoint-3c2dae9044/run_manifest.yaml ADDED
@@ -0,0 +1,92 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: upvla.experiment_run.v1
2
+ run:
3
+ id: E001-R001
4
+ experiment_id: E001
5
+ condition_id: c-endpoint-3c2dae9044
6
+ status: completed
7
+ started_at: '2026-08-09T23:14:40.050720+00:00'
8
+ ended_at: '2026-08-09T23:23:38.981382+00:00'
9
+ condition:
10
+ model:
11
+ action_head: endpoint
12
+ base_conditioning: none
13
+ horizon_steps: 1
14
+ randomness:
15
+ global_seed: 42
16
+ python_seed: 42
17
+ numpy_seed: 42
18
+ torch_seed: 42
19
+ dataloader_seed: 42
20
+ code:
21
+ repo: https://github.com/Shiki42/UP-VLA.git
22
+ git_commit: 91b8c9473cfc60c8bfce3848c900aba8e4594ca1
23
+ git_branch: shuyuan/action-chunk-recovery-e001
24
+ dirty: false
25
+ status_sha256: e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855
26
+ patch_ref: null
27
+ data:
28
+ hf_repo_id: Shiki42/up-vla-precision-recovery-15k
29
+ revision: 77f5f306303361d602ab10759b3a139c67f2cf16
30
+ variant_id: validated-physical-recovery-15k-v1
31
+ manifest_sha256: 4cda13bc86038829f5fd519c7c766d40e770ad915974f68afea6f251b59faa50
32
+ dataset_info_sha256: 99a2f93632799a2b0a5da785438a55934e7fbaac582de0d63f8ddc0186b10ef3
33
+ extraction_receipt: /home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples/_extraction_receipt.json
34
+ extraction_receipt_sha256: ff05bdb511990c0f951cb17efb1f84cc4c4a48e59f875ebd6d66a468c3b840dd
35
+ split_protocol: source-family-stratified-group-hash-v1
36
+ split_counts:
37
+ train_samples: 11628
38
+ validation_samples: 1970
39
+ train_groups: 89
40
+ validation_groups: 11
41
+ overlap: false
42
+ environment:
43
+ python_version: 3.11.15
44
+ python_executable: /home/coder/share/experiment-runtimes/E001-py311-cu130/bin/python
45
+ packages:
46
+ numpy: 1.26.0
47
+ Pillow: 11.3.0
48
+ PyYAML: 6.0.3
49
+ torch: 2.13.0
50
+ torch_version: 2.13.0+cu130
51
+ cuda_runtime: '13.0'
52
+ cudnn_version: 92000
53
+ os: Linux-7.0.0-28-generic-x86_64-with-glibc2.35
54
+ hostname: lm
55
+ hardware:
56
+ accelerators:
57
+ - index: 0
58
+ name: NVIDIA RTX PRO 6000 Blackwell Workstation Edition
59
+ memory_bytes: 101973491712
60
+ compute_capability: '12.0'
61
+ accelerator_count: 1
62
+ cpu_count: 32
63
+ execution:
64
+ command: /home/coder/share/experiment-runtimes/E001-py311-cu130/bin/python scripts/train_action_chunk_recovery.py
65
+ --condition-config configs/experiments/E001/R001-endpoint.yaml --data-root /home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples
66
+ --output-dir /home/coder/share/experiment-runs/E001/E001-R001
67
+ metrics:
68
+ validation:
69
+ loss/total: 0.1953894484164146
70
+ loss/translation: 0.04519216449278866
71
+ loss/rotation_vector: 0.03495830947221233
72
+ loss/rotation_geodesic: 0.3900546948316739
73
+ loss/joint: 0.03496293299240509
74
+ loss/gripper: 0.0024383212099334008
75
+ metric/translation_mm: 23.151747923333026
76
+ metric/rotation_deg: 11.702068714199937
77
+ metric/joint_mae_rad: 0.08829079693949161
78
+ metric/gripper_accuracy: 1.0
79
+ checkpoint:
80
+ path: /home/coder/share/experiment-runs/E001/E001-R001/best.pt
81
+ sha256: 182fc8544a1489582c7f96e29625b7e0048642edf6de749c50b19da0da33379f
82
+ selection_metric: loss/total
83
+ selection_split: validation
84
+ artifacts:
85
+ metrics_jsonl: /home/coder/share/experiment-runs/E001/E001-R001/metrics.jsonl
86
+ validation_evaluation: /home/coder/share/experiment-runs/E001/E001-R001/evaluation_validation.json
87
+ condition_snapshot: /home/coder/share/experiment-runs/E001/E001-R001/condition.yaml
88
+ resolved_config: /home/coder/share/experiment-runs/E001/E001-R001/config_resolved.yaml
89
+ compute:
90
+ wall_seconds: 538.7586286930018
91
+ gpu_hours: 0.14965517463694494
92
+ peak_memory_bytes: 2885576192
experiments/E001/conditions/c-residual-h16-be2e741500/CONDITION_CARD.md ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Selected residual H16, identity initialized
2
+
3
+ - Run: E001-R004
4
+ - Condition: c-residual-h16-be2e741500
5
+ - Checkpoint SHA-256: 81ef4841165e996bf7c94556e3f83b457ae5b021bfa6c99ae2fa4998e3f111f3
6
+ - Dataset revision: 77f5f306303361d602ab10759b3a139c67f2cf16
7
+ - Split: source-family-stratified grouped hash
8
+ - Horizon: 16 x 0.05 s for chunk conditions
9
+ - Frame/composition: current-tool SE(3), right multiplication
10
+ - Intended use: research evaluation only
11
+
12
+ See the sibling run manifest and evaluation receipts. Test data was opened exactly
13
+ once after validation-only selection.
experiments/E001/conditions/c-residual-h16-be2e741500/best.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:81ef4841165e996bf7c94556e3f83b457ae5b021bfa6c99ae2fa4998e3f111f3
3
+ size 238782079
experiments/E001/conditions/c-residual-h16-be2e741500/config_resolved.yaml ADDED
@@ -0,0 +1,103 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ data:
2
+ hf_repo_id: Shiki42/up-vla-precision-recovery-15k
3
+ revision: 77f5f306303361d602ab10759b3a139c67f2cf16
4
+ variant_id: validated-physical-recovery-15k-v1
5
+ manifest_sha256: 4cda13bc86038829f5fd519c7c766d40e770ad915974f68afea6f251b59faa50
6
+ dataset_info_sha256: 99a2f93632799a2b0a5da785438a55934e7fbaac582de0d63f8ddc0186b10ef3
7
+ sample_count: 15000
8
+ split_protocol: source-family-stratified-group-hash-v1
9
+ validation_fraction: 0.1
10
+ test_fraction: 0.1
11
+ geometry:
12
+ cumulative_waypoint_representation: current_tool_relative_pose
13
+ step_representation: integrated_body_twist
14
+ step_integration: right_se3_exponential
15
+ residual_composition: base_right_multiply_se3_exponential
16
+ base_input_normalization: translation_rotation_physical_scales
17
+ residual_output_initialization: zero
18
+ model:
19
+ backbone:
20
+ hidden_size: 256
21
+ fusion_layers: 4
22
+ fusion_heads: 8
23
+ fusion_mlp_ratio: 4
24
+ dropout: 0.1
25
+ vision_backend: conv
26
+ text_backend: embedding
27
+ pretrained_model_name: google/siglip-base-patch16-224
28
+ freeze_pretrained: true
29
+ image_size: 224
30
+ vision_tokens: 64
31
+ depth_tokens: 36
32
+ point_tokens: 32
33
+ point_samples: 256
34
+ max_text_tokens: 32
35
+ text_vocab_size: 32768
36
+ max_joints: 7
37
+ aloha_joints: 6
38
+ panda_joints: 7
39
+ translation_scale_m: 0.05
40
+ rotation_scale_rad: 0.5235987755982988
41
+ joint_scale_rad: 0.5
42
+ horizon_steps: 16
43
+ horizon_seconds: 0.8
44
+ decoder_layers: 4
45
+ decoder_heads: 8
46
+ decoder_mlp_ratio: 4
47
+ maximum_step_translation_m: 0.01
48
+ maximum_step_rotation_rad: 0.2
49
+ maximum_residual_translation_m: 0.05
50
+ maximum_residual_rotation_rad: 0.52
51
+ action_head: residual_chunk
52
+ base_conditioning: E001-R001-profiled-proposal
53
+ parameter_counts:
54
+ total: 20077914
55
+ trainable: 19873862
56
+ frozen: 204052
57
+ loss:
58
+ path_translation: 1.0
59
+ path_rotation: 1.0
60
+ endpoint_translation: 2.0
61
+ endpoint_rotation: 2.0
62
+ velocity: 1.0
63
+ acceleration: 0.1
64
+ jerk: 0.05
65
+ residual_magnitude: 0.02
66
+ training:
67
+ seed: 42
68
+ batch_size: 64
69
+ gradient_accumulation_steps: 1
70
+ epochs: 100
71
+ max_steps: 3000
72
+ learning_rate: 0.0003
73
+ weight_decay: 0.05
74
+ warmup_fraction: 0.05
75
+ gradient_clip_norm: 1.0
76
+ precision: bf16
77
+ workers: 8
78
+ log_every_steps: 10
79
+ validation_every_steps: 250
80
+ checkpoint_every_steps: 500
81
+ evaluation:
82
+ version: action-chunk-eval-v1
83
+ trajectory_loss:
84
+ path_translation: 1.0
85
+ path_rotation: 1.0
86
+ endpoint_translation: 2.0
87
+ endpoint_rotation: 2.0
88
+ velocity: 1.0
89
+ acceleration: 0.1
90
+ jerk: 0.05
91
+ residual_magnitude: 0.02
92
+ selection_split: validation
93
+ test_firewall_opened: false
94
+ runtime:
95
+ data_root: /home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples
96
+ base_proposal_root: /home/coder/share/experiment-artifacts/E001/R001-profiled-proposals
97
+ output_directory: /home/coder/share/experiment-runs/E001/E001-R004
98
+ split_counts:
99
+ train_samples: 11628
100
+ validation_samples: 1970
101
+ train_groups: 89
102
+ validation_groups: 11
103
+ overlap: false
experiments/E001/conditions/c-residual-h16-be2e741500/evaluation_test.json ADDED
@@ -0,0 +1,119 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "base_proposal_manifest": {
3
+ "path": "/home/coder/share/experiment-artifacts/E001/R001-profiled-proposals/manifest.json",
4
+ "sha256": "b2a7fda405bb7369a91d0e74040617b08704ed908f84ab653315a6a23da7bc50"
5
+ },
6
+ "checkpoint": {
7
+ "path": "/home/coder/share/experiment-runs/E001/E001-R004/best.pt",
8
+ "sha256": "81ef4841165e996bf7c94556e3f83b457ae5b021bfa6c99ae2fa4998e3f111f3"
9
+ },
10
+ "checkpoint_run": {
11
+ "condition_id": "c-residual-h16-be2e741500",
12
+ "experiment_id": "E001",
13
+ "git_commit": "63265740e2e0caa4c6f42da6d497fd63f46d522d",
14
+ "manifest": "/home/coder/share/experiment-runs/E001/E001-R004/run_manifest.yaml",
15
+ "manifest_sha256": "c97c0ce3f3ebfeccce4c75b1f9d9a039f462fa2812b252169a27e8f694667fe7",
16
+ "run_id": "E001-R004"
17
+ },
18
+ "condition_id": "c-residual-h16-be2e741500",
19
+ "data_extraction_receipt": {
20
+ "path": "/home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples/_extraction_receipt.json",
21
+ "sha256": "ff05bdb511990c0f951cb17efb1f84cc4c4a48e59f875ebd6d66a468c3b840dd"
22
+ },
23
+ "experiment_id": "E001",
24
+ "group_count": 11,
25
+ "grouped_metrics": {
26
+ "acceleration_rmse": 0.20413607358932495,
27
+ "command_acceleration_rms": 0.19192136824131012,
28
+ "command_jerk_rms": 4.2998576164245605,
29
+ "endpoint_rotation_rmse_deg": 4.175118923187256,
30
+ "endpoint_translation_rmse_mm": 6.796607971191406,
31
+ "jerk_rmse": 4.632838726043701,
32
+ "path_rotation_rmse_deg": 2.630558967590332,
33
+ "path_translation_rmse_mm": 4.2072577476501465,
34
+ "residual_rotation_mean_deg": 0.49680280685424805,
35
+ "residual_translation_mean_mm": 0.4243967831134796,
36
+ "velocity_rmse": 0.03973293676972389
37
+ },
38
+ "latency": {
39
+ "device": "cuda",
40
+ "median_ms": 4.552353500000001,
41
+ "p95_ms": 4.6352965,
42
+ "p99_ms": 6.060615229999992,
43
+ "repeats": 500,
44
+ "warmup": 50
45
+ },
46
+ "losses": {
47
+ "acceleration": 0.0019577299244701862,
48
+ "endpoint_rotation": 0.11812951415777206,
49
+ "endpoint_translation": 0.0032451662700623274,
50
+ "jerk": 0.002492707222700119,
51
+ "path_rotation": 0.06641256809234619,
52
+ "path_translation": 0.0012373252538964152,
53
+ "residual_magnitude": 0.025205370038747787,
54
+ "total": 0.37313660979270935,
55
+ "velocity": 0.06191283091902733
56
+ },
57
+ "metrics": {
58
+ "acceleration_rmse": 0.20839367806911469,
59
+ "command_acceleration_rms": 0.19647431373596191,
60
+ "command_jerk_rms": 4.175158977508545,
61
+ "endpoint_rotation_rmse_deg": 4.0035858154296875,
62
+ "endpoint_translation_rmse_mm": 6.976925849914551,
63
+ "jerk_rmse": 4.510144233703613,
64
+ "path_rotation_rmse_deg": 2.5089447498321533,
65
+ "path_translation_rmse_mm": 4.308117866516113,
66
+ "residual_rotation_mean_deg": 0.4771876037120819,
67
+ "residual_translation_mean_mm": 0.4649559259414673,
68
+ "velocity_rmse": 0.038097888231277466
69
+ },
70
+ "prediction_source": "checkpoint",
71
+ "primary_path_score": {
72
+ "definition": "path_translation_rmse_mm / 10 + path_rotation_rmse_deg / 5",
73
+ "group_macro": 0.946837568283081,
74
+ "micro": 0.932600736618042
75
+ },
76
+ "run_id": "E001-R004",
77
+ "sample_count": 1402,
78
+ "schema_version": "precision.action_chunk.evaluation.v1",
79
+ "source_family_metrics": {
80
+ "rlbench": {
81
+ "group_count": 1,
82
+ "metrics": {
83
+ "acceleration_rmse": 0.22867096960544586,
84
+ "command_acceleration_rms": 0.22290381789207458,
85
+ "command_jerk_rms": 4.427183628082275,
86
+ "endpoint_rotation_rmse_deg": 4.308777809143066,
87
+ "endpoint_translation_rmse_mm": 5.983713626861572,
88
+ "jerk_rmse": 4.580673694610596,
89
+ "path_rotation_rmse_deg": 2.6923117637634277,
90
+ "path_translation_rmse_mm": 3.6653292179107666,
91
+ "residual_rotation_mean_deg": 0.18301226198673248,
92
+ "residual_translation_mean_mm": 0.26214030385017395,
93
+ "velocity_rmse": 0.04066188260912895
94
+ },
95
+ "path_score": 0.9049952745437622,
96
+ "sample_count": 520
97
+ },
98
+ "robotwin_generic": {
99
+ "group_count": 10,
100
+ "metrics": {
101
+ "acceleration_rmse": 0.19545558094978333,
102
+ "command_acceleration_rms": 0.17907370626926422,
103
+ "command_jerk_rms": 4.019175052642822,
104
+ "endpoint_rotation_rmse_deg": 3.812222719192505,
105
+ "endpoint_translation_rmse_mm": 7.5011210441589355,
106
+ "jerk_rmse": 4.4680399894714355,
107
+ "path_rotation_rmse_deg": 2.3942654132843018,
108
+ "path_translation_rmse_mm": 4.645596027374268,
109
+ "residual_rotation_mean_deg": 0.6506244540214539,
110
+ "residual_translation_mean_mm": 0.5845297574996948,
111
+ "velocity_rmse": 0.0365019366145134
112
+ },
113
+ "path_score": 0.943412685394287,
114
+ "sample_count": 882
115
+ }
116
+ },
117
+ "split": "test",
118
+ "test_firewall_opened": true
119
+ }
experiments/E001/conditions/c-residual-h16-be2e741500/evaluation_validation.json ADDED
@@ -0,0 +1,119 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "base_proposal_manifest": {
3
+ "path": "/home/coder/share/experiment-artifacts/E001/R001-profiled-proposals/manifest.json",
4
+ "sha256": "b2a7fda405bb7369a91d0e74040617b08704ed908f84ab653315a6a23da7bc50"
5
+ },
6
+ "checkpoint": {
7
+ "path": "/home/coder/share/experiment-runs/E001/E001-R004/best.pt",
8
+ "sha256": "81ef4841165e996bf7c94556e3f83b457ae5b021bfa6c99ae2fa4998e3f111f3"
9
+ },
10
+ "checkpoint_run": {
11
+ "condition_id": "c-residual-h16-be2e741500",
12
+ "experiment_id": "E001",
13
+ "git_commit": "63265740e2e0caa4c6f42da6d497fd63f46d522d",
14
+ "manifest": "/home/coder/share/experiment-runs/E001/E001-R004/run_manifest.yaml",
15
+ "manifest_sha256": "c97c0ce3f3ebfeccce4c75b1f9d9a039f462fa2812b252169a27e8f694667fe7",
16
+ "run_id": "E001-R004"
17
+ },
18
+ "condition_id": "c-residual-h16-be2e741500",
19
+ "data_extraction_receipt": {
20
+ "path": "/home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples/_extraction_receipt.json",
21
+ "sha256": "ff05bdb511990c0f951cb17efb1f84cc4c4a48e59f875ebd6d66a468c3b840dd"
22
+ },
23
+ "experiment_id": "E001",
24
+ "group_count": 11,
25
+ "grouped_metrics": {
26
+ "acceleration_rmse": 0.20408500730991364,
27
+ "command_acceleration_rms": 0.19094577431678772,
28
+ "command_jerk_rms": 4.2563090324401855,
29
+ "endpoint_rotation_rmse_deg": 3.4526124000549316,
30
+ "endpoint_translation_rmse_mm": 5.907562732696533,
31
+ "jerk_rmse": 4.626509189605713,
32
+ "path_rotation_rmse_deg": 2.156282901763916,
33
+ "path_translation_rmse_mm": 3.6638970375061035,
34
+ "residual_rotation_mean_deg": 0.5087225437164307,
35
+ "residual_translation_mean_mm": 0.4505690634250641,
36
+ "velocity_rmse": 0.03341490402817726
37
+ },
38
+ "latency": {
39
+ "device": "cuda",
40
+ "median_ms": 4.593195,
41
+ "p95_ms": 4.6979921,
42
+ "p99_ms": 5.017176329999998,
43
+ "repeats": 500,
44
+ "warmup": 50
45
+ },
46
+ "losses": {
47
+ "acceleration": 0.0018424771260470152,
48
+ "endpoint_rotation": 0.10740330070257187,
49
+ "endpoint_translation": 0.0025280702393501997,
50
+ "jerk": 0.002553013851866126,
51
+ "path_rotation": 0.05987050384283066,
52
+ "path_translation": 0.0009804598521441221,
53
+ "residual_magnitude": 0.024654608219861984,
54
+ "total": 0.33214807510375977,
55
+ "velocity": 0.0506293959915638
56
+ },
57
+ "metrics": {
58
+ "acceleration_rmse": 0.19798384606838226,
59
+ "command_acceleration_rms": 0.1852402538061142,
60
+ "command_jerk_rms": 4.109386920928955,
61
+ "endpoint_rotation_rmse_deg": 3.727142572402954,
62
+ "endpoint_translation_rmse_mm": 6.1580071449279785,
63
+ "jerk_rmse": 4.461413383483887,
64
+ "path_rotation_rmse_deg": 2.326387882232666,
65
+ "path_translation_rmse_mm": 3.8349575996398926,
66
+ "residual_rotation_mean_deg": 0.47515517473220825,
67
+ "residual_translation_mean_mm": 0.4408051371574402,
68
+ "velocity_rmse": 0.035521529614925385
69
+ },
70
+ "prediction_source": "checkpoint",
71
+ "primary_path_score": {
72
+ "definition": "path_translation_rmse_mm / 10 + path_rotation_rmse_deg / 5",
73
+ "group_macro": 0.7976462841033936,
74
+ "micro": 0.8487733364105225
75
+ },
76
+ "run_id": "E001-R004",
77
+ "sample_count": 1970,
78
+ "schema_version": "precision.action_chunk.evaluation.v1",
79
+ "source_family_metrics": {
80
+ "rlbench": {
81
+ "group_count": 1,
82
+ "metrics": {
83
+ "acceleration_rmse": 0.1673765480518341,
84
+ "command_acceleration_rms": 0.15914097428321838,
85
+ "command_jerk_rms": 3.41491961479187,
86
+ "endpoint_rotation_rmse_deg": 4.646145820617676,
87
+ "endpoint_translation_rmse_mm": 7.01324987411499,
88
+ "jerk_rmse": 3.634052038192749,
89
+ "path_rotation_rmse_deg": 2.8927807807922363,
90
+ "path_translation_rmse_mm": 4.402773380279541,
91
+ "residual_rotation_mean_deg": 0.391691654920578,
92
+ "residual_translation_mean_mm": 0.48431363701820374,
93
+ "velocity_rmse": 0.04281594604253769
94
+ },
95
+ "path_score": 1.0188334941864015,
96
+ "sample_count": 520
97
+ },
98
+ "robotwin_generic": {
99
+ "group_count": 10,
100
+ "metrics": {
101
+ "acceleration_rmse": 0.20786523818969727,
102
+ "command_acceleration_rms": 0.19374538958072662,
103
+ "command_jerk_rms": 4.331396102905273,
104
+ "endpoint_rotation_rmse_deg": 3.3364617824554443,
105
+ "endpoint_translation_rmse_mm": 5.820766925811768,
106
+ "jerk_rmse": 4.722945213317871,
107
+ "path_rotation_rmse_deg": 2.0861356258392334,
108
+ "path_translation_rmse_mm": 3.6096324920654297,
109
+ "residual_rotation_mean_deg": 0.5050869584083557,
110
+ "residual_translation_mean_mm": 0.42520201206207275,
111
+ "velocity_rmse": 0.03250928595662117
112
+ },
113
+ "path_score": 0.7781903743743896,
114
+ "sample_count": 1450
115
+ }
116
+ },
117
+ "split": "validation",
118
+ "test_firewall_opened": false
119
+ }
experiments/E001/conditions/c-residual-h16-be2e741500/run_manifest.yaml ADDED
@@ -0,0 +1,104 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: upvla.experiment_run.v1
2
+ run:
3
+ id: E001-R004
4
+ experiment_id: E001
5
+ condition_id: c-residual-h16-be2e741500
6
+ status: completed
7
+ started_at: '2026-08-09T23:48:11.539430+00:00'
8
+ ended_at: '2026-08-09T23:51:19.928667+00:00'
9
+ condition:
10
+ model:
11
+ action_head: residual_chunk
12
+ base_conditioning: E001-R001-profiled-proposal
13
+ horizon_steps: 16
14
+ randomness:
15
+ global_seed: 42
16
+ python_seed: 42
17
+ numpy_seed: 42
18
+ torch_seed: 42
19
+ dataloader_seed: 42
20
+ code:
21
+ repo: https://github.com/Shiki42/UP-VLA.git
22
+ git_commit: 63265740e2e0caa4c6f42da6d497fd63f46d522d
23
+ git_branch: shuyuan/action-chunk-recovery-e001
24
+ dirty: false
25
+ status_sha256: e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855
26
+ patch_ref: null
27
+ data:
28
+ hf_repo_id: Shiki42/up-vla-precision-recovery-15k
29
+ revision: 77f5f306303361d602ab10759b3a139c67f2cf16
30
+ variant_id: validated-physical-recovery-15k-v1
31
+ manifest_sha256: 4cda13bc86038829f5fd519c7c766d40e770ad915974f68afea6f251b59faa50
32
+ dataset_info_sha256: 99a2f93632799a2b0a5da785438a55934e7fbaac582de0d63f8ddc0186b10ef3
33
+ extraction_receipt: /home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples/_extraction_receipt.json
34
+ extraction_receipt_sha256: ff05bdb511990c0f951cb17efb1f84cc4c4a48e59f875ebd6d66a468c3b840dd
35
+ split_protocol: source-family-stratified-group-hash-v1
36
+ split_counts:
37
+ train_samples: 11628
38
+ validation_samples: 1970
39
+ train_groups: 89
40
+ validation_groups: 11
41
+ overlap: false
42
+ environment:
43
+ python_version: 3.11.15
44
+ python_executable: /home/coder/share/experiment-runtimes/E001-py311-cu130/bin/python
45
+ packages:
46
+ numpy: 1.26.0
47
+ Pillow: 11.3.0
48
+ PyYAML: 6.0.3
49
+ torch: 2.13.0
50
+ torch_version: 2.13.0+cu130
51
+ cuda_runtime: '13.0'
52
+ cudnn_version: 92000
53
+ os: Linux-7.0.0-28-generic-x86_64-with-glibc2.35
54
+ hostname: lm
55
+ hardware:
56
+ accelerators:
57
+ - index: 0
58
+ name: NVIDIA RTX PRO 6000 Blackwell Workstation Edition
59
+ memory_bytes: 101973491712
60
+ compute_capability: '12.0'
61
+ accelerator_count: 1
62
+ cpu_count: 32
63
+ execution:
64
+ command: /home/coder/share/experiment-runtimes/E001-py311-cu130/bin/python scripts/train_action_chunk_recovery.py
65
+ --condition-config configs/experiments/E001/R004-residual-h16-zero-init.yaml --data-root
66
+ /home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples --base-proposal-root
67
+ /home/coder/share/experiment-artifacts/E001/R001-profiled-proposals --output-dir
68
+ /home/coder/share/experiment-runs/E001/E001-R004
69
+ metrics:
70
+ validation:
71
+ loss/total: 0.33208154863512457
72
+ loss/path_translation: 0.000980461022561312
73
+ loss/path_rotation: 0.05985661550675552
74
+ loss/endpoint_translation: 0.002528078034803801
75
+ loss/endpoint_rotation: 0.10737666593590364
76
+ loss/velocity: 0.050629954682994006
77
+ loss/acceleration: 0.0018427140325270026
78
+ loss/jerk: 0.0025531304348969295
79
+ loss/residual_magnitude: 0.024654954535707
80
+ metric/path_translation_rmse_mm: 3.817204704865586
81
+ metric/path_rotation_rmse_deg: 2.2951113328110746
82
+ metric/endpoint_translation_rmse_mm: 6.132941103223617
83
+ metric/endpoint_rotation_rmse_deg: 3.6756162817708127
84
+ metric/velocity_rmse: 0.03517932392240781
85
+ metric/acceleration_rmse: 0.19716658198893977
86
+ metric/jerk_rmse: 4.4351433306176045
87
+ metric/command_acceleration_rms: 0.18460183148154147
88
+ metric/command_jerk_rms: 4.088533519851374
89
+ metric/residual_translation_mean_mm: 0.440811541994211
90
+ metric/residual_rotation_mean_deg: 0.47516172582728006
91
+ checkpoint:
92
+ path: /home/coder/share/experiment-runs/E001/E001-R004/best.pt
93
+ sha256: 81ef4841165e996bf7c94556e3f83b457ae5b021bfa6c99ae2fa4998e3f111f3
94
+ selection_metric: loss/total
95
+ selection_split: validation
96
+ artifacts:
97
+ metrics_jsonl: /home/coder/share/experiment-runs/E001/E001-R004/metrics.jsonl
98
+ validation_evaluation: /home/coder/share/experiment-runs/E001/E001-R004/evaluation_validation.json
99
+ condition_snapshot: /home/coder/share/experiment-runs/E001/E001-R004/condition.yaml
100
+ resolved_config: /home/coder/share/experiment-runs/E001/E001-R004/config_resolved.yaml
101
+ compute:
102
+ wall_seconds: 188.26792803000717
103
+ gpu_hours: 0.052296646675001994
104
+ peak_memory_bytes: 2952011264
experiments/E001/experiment-summary.json ADDED
@@ -0,0 +1,408 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "advance_to_e002": false,
3
+ "artifacts": {
4
+ "proposal_manifest": {
5
+ "path": "/home/coder/share/experiment-artifacts/E001/R001-profiled-proposals/manifest.json",
6
+ "sha256": "b2a7fda405bb7369a91d0e74040617b08704ed908f84ab653315a6a23da7bc50"
7
+ },
8
+ "real_data_smoke": {
9
+ "path": "/home/coder/share/experiment-logs/E001/real-data-smoke.json",
10
+ "sha256": "98502ae238c997ed151d99bbf8caf1c0e1bfad04be181f422deef73fc8fa4505"
11
+ },
12
+ "residual_zero_init_smoke": {
13
+ "path": "/home/coder/share/experiment-logs/E001/residual-zero-init-smoke.json",
14
+ "sha256": "6aa37549464368866c756979073f38f79b45cf567462e19b8155eba5fa255c30"
15
+ }
16
+ },
17
+ "compute": {
18
+ "total_gpu_hours_including_invalid": 0.3095676502552816
19
+ },
20
+ "data": {
21
+ "extraction_receipt": {
22
+ "path": "/home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples/_extraction_receipt.json",
23
+ "sha256": "ff05bdb511990c0f951cb17efb1f84cc4c4a48e59f875ebd6d66a468c3b840dd"
24
+ },
25
+ "independent_source_groups": 111,
26
+ "sample_count": 15000,
27
+ "split_audit": {
28
+ "path": "/home/coder/share/experiment-logs/E001/data-split-audit.json",
29
+ "sha256": "db98be16cd36ccb1726d5407202b126f8ed147e31144ba6c1228dc163e8be353"
30
+ },
31
+ "target_audit": {
32
+ "path": "/home/coder/share/experiment-logs/E001/action-target-audit.json",
33
+ "sha256": "16e1b62f9f54fb0ac8a0fab0b7772e66ab6b6bc0d930fc5aa30f53ff19258fa6"
34
+ }
35
+ },
36
+ "experiment_id": "E001",
37
+ "hypothesis_status": "inconclusive",
38
+ "invalid_run": {
39
+ "reason": "non-zero residual initialization saturated SE(3) bounds",
40
+ "run_id": "E001-R003",
41
+ "scientific_comparison_eligible": false
42
+ },
43
+ "limitations": [
44
+ "E001 uses frozen endpoint-policy proposals, not real Pi0.5 proposals.",
45
+ "Only one initial observation is available per recovery and all targets cover the first 0.8 seconds.",
46
+ "The selected residual model increases acceleration and jerk despite improving path score.",
47
+ "Validation and test each contain 11 independent groups; 15000 perturbations are not independent units."
48
+ ],
49
+ "reason": "Selected action chunk improved grouped path score but missed the pre-registered 10% validation gate.",
50
+ "runs": {
51
+ "E001-R001": {
52
+ "checkpoint": {
53
+ "path": "/home/coder/share/experiment-runs/E001/E001-R001/best.pt",
54
+ "selection_metric": "loss/total",
55
+ "selection_split": "validation",
56
+ "sha256": "182fc8544a1489582c7f96e29625b7e0048642edf6de749c50b19da0da33379f"
57
+ },
58
+ "compute": {
59
+ "gpu_hours": 0.14965517463694494,
60
+ "peak_memory_bytes": 2885576192,
61
+ "wall_seconds": 538.7586286930018
62
+ },
63
+ "condition_id": "c-endpoint-3c2dae9044",
64
+ "git_commit": "91b8c9473cfc60c8bfce3848c900aba8e4594ca1",
65
+ "manifest": {
66
+ "path": "/home/coder/share/experiment-runs/E001/E001-R001/run_manifest.yaml",
67
+ "sha256": "6548cd5b4b3335d5b5f88262924e34a8fcbf0621216792bc946bad804026c418"
68
+ },
69
+ "status": "completed"
70
+ },
71
+ "E001-R002": {
72
+ "checkpoint": {
73
+ "path": "/home/coder/share/experiment-runs/E001/E001-R002/best.pt",
74
+ "selection_metric": "loss/total",
75
+ "selection_split": "validation",
76
+ "sha256": "198e1678617042c3264f465682c6593a71c13cc43e3751dfe9121447333f2fb8"
77
+ },
78
+ "compute": {
79
+ "gpu_hours": 0.0523659750544458,
80
+ "peak_memory_bytes": 2950927872,
81
+ "wall_seconds": 188.51751019600488
82
+ },
83
+ "condition_id": "c-direct-h16-e3f979a8db",
84
+ "git_commit": "91b8c9473cfc60c8bfce3848c900aba8e4594ca1",
85
+ "manifest": {
86
+ "path": "/home/coder/share/experiment-runs/E001/E001-R002/run_manifest.yaml",
87
+ "sha256": "ba17d383fee1444110197648235d00d447ef14055df8f7ab9280c84486316fd7"
88
+ },
89
+ "status": "completed"
90
+ },
91
+ "E001-R003": {
92
+ "checkpoint": {
93
+ "path": "/home/coder/share/experiment-runs/E001/E001-R003/best.pt",
94
+ "scientific_comparison_eligible": false,
95
+ "selection_split": "validation",
96
+ "sha256": "53f2cc61dec2c270556b762531ebe4174d80ca0779db3637c8fa6e235558b00e"
97
+ },
98
+ "compute": {
99
+ "gpu_hours": 0.05524985388888889,
100
+ "wall_seconds": 198.899474
101
+ },
102
+ "condition_id": "c-residual-h16-be2e741500",
103
+ "git_commit": "91b8c9473cfc60c8bfce3848c900aba8e4594ca1",
104
+ "manifest": {
105
+ "path": "/home/coder/share/experiment-runs/E001/E001-R003/run_manifest.yaml",
106
+ "sha256": "b27935c72939dc5112d422ab19c9b422b539bd041cc4ad6f3c2f5546001b24e8"
107
+ },
108
+ "status": "invalid"
109
+ },
110
+ "E001-R004": {
111
+ "checkpoint": {
112
+ "path": "/home/coder/share/experiment-runs/E001/E001-R004/best.pt",
113
+ "selection_metric": "loss/total",
114
+ "selection_split": "validation",
115
+ "sha256": "81ef4841165e996bf7c94556e3f83b457ae5b021bfa6c99ae2fa4998e3f111f3"
116
+ },
117
+ "compute": {
118
+ "gpu_hours": 0.052296646675001994,
119
+ "peak_memory_bytes": 2952011264,
120
+ "wall_seconds": 188.26792803000717
121
+ },
122
+ "condition_id": "c-residual-h16-be2e741500",
123
+ "git_commit": "63265740e2e0caa4c6f42da6d497fd63f46d522d",
124
+ "manifest": {
125
+ "path": "/home/coder/share/experiment-runs/E001/E001-R004/run_manifest.yaml",
126
+ "sha256": "c97c0ce3f3ebfeccce4c75b1f9d9a039f462fa2812b252169a27e8f694667fe7"
127
+ },
128
+ "status": "completed"
129
+ }
130
+ },
131
+ "schema_version": "upvla.experiment_conclusion.v1",
132
+ "selection": {
133
+ "advance_to_e002": false,
134
+ "baseline_condition_id": "c-endpoint-3c2dae9044",
135
+ "experiment_id": "E001",
136
+ "gates": {
137
+ "batch1_gpu_p95_below_20ms": true,
138
+ "endpoint_rotation_within_1deg": true,
139
+ "endpoint_translation_within_1mm": true,
140
+ "path_improves_at_least_10_percent": false
141
+ },
142
+ "hypothesis_status": "inconclusive",
143
+ "inputs": {
144
+ "baseline": {
145
+ "path": "/home/coder/share/experiment-artifacts/E001/evaluations/R001-validation.json",
146
+ "sha256": "2733fe8093af6a9fad8f9c6988caee1cd048412c3620ad193779a27bcdbe4972"
147
+ },
148
+ "candidates": [
149
+ {
150
+ "path": "/home/coder/share/experiment-artifacts/E001/evaluations/R002-validation.json",
151
+ "sha256": "e69679a21e423a975480369d3c9773a812bff06298f81ab6a26e2004fe1bc338"
152
+ },
153
+ {
154
+ "path": "/home/coder/share/experiment-artifacts/E001/evaluations/R004-validation.json",
155
+ "sha256": "9373e8bd953247345671b604f00945aa5649b2a8887a94136547e3de77f6a0d4"
156
+ }
157
+ ],
158
+ "experiment": {
159
+ "path": "/home/coder/share/UP-VLA-action-chunk-E001/experiments/E001-action-chunk-recovery.yaml",
160
+ "sha256": "759089e2aaac1ff2c251a3259933b4f3c33cae6b0018a5eeeefc11ee7c4ee56b"
161
+ }
162
+ },
163
+ "ranking": [
164
+ {
165
+ "batch1_gpu_p95_ms": 4.6979921,
166
+ "condition_id": "c-residual-h16-be2e741500",
167
+ "endpoint_rotation_rmse_deg": 3.4526124000549316,
168
+ "endpoint_translation_rmse_mm": 5.907562732696533,
169
+ "group_macro_path_score": 0.7976462841033936,
170
+ "run_id": "E001-R004"
171
+ },
172
+ {
173
+ "batch1_gpu_p95_ms": 4.3417951,
174
+ "condition_id": "c-direct-h16-e3f979a8db",
175
+ "endpoint_rotation_rmse_deg": 3.5026297569274902,
176
+ "endpoint_translation_rmse_mm": 5.990349292755127,
177
+ "group_macro_path_score": 0.8049414157867432,
178
+ "run_id": "E001-R002"
179
+ }
180
+ ],
181
+ "receipt": {
182
+ "path": "/home/coder/share/experiment-artifacts/E001/selection-validation.json",
183
+ "sha256": "fc742f723daf1f6e91b78db104c78c47056ad965fc0510de8ea9fe4964d868b3"
184
+ },
185
+ "schema_version": "precision.action_chunk.selection.v1",
186
+ "selected_condition_id": "c-residual-h16-be2e741500",
187
+ "selected_run_id": "E001-R004",
188
+ "selection_split": "validation",
189
+ "test_firewall_opened": false
190
+ },
191
+ "status": "completed",
192
+ "test": {
193
+ "absolute_path_score_change": -0.04643893241882335,
194
+ "baseline": {
195
+ "metrics": {
196
+ "acceleration_rmse": 0.11615525931119919,
197
+ "command_acceleration_rms": 0.09448793530464172,
198
+ "command_jerk_rms": 2.311239004135132,
199
+ "endpoint_rotation_rmse_deg": 4.419965744018555,
200
+ "endpoint_translation_rmse_mm": 7.017010688781738,
201
+ "jerk_rmse": 2.8706088066101074,
202
+ "path_rotation_rmse_deg": 2.7911133766174316,
203
+ "path_translation_rmse_mm": 4.35053825378418,
204
+ "residual_rotation_mean_deg": 0.0,
205
+ "residual_translation_mean_mm": 0.0,
206
+ "velocity_rmse": 0.041266847401857376
207
+ },
208
+ "receipt": {
209
+ "path": "/home/coder/share/experiment-artifacts/E001/evaluations/R001-test.json",
210
+ "sha256": "4bb067239611e3f9427ba5fffdc8c0b56a30452f6db8e0a02b2ade56a2b63e43"
211
+ },
212
+ "score": 0.9932765007019043,
213
+ "source_family_metrics": {
214
+ "rlbench": {
215
+ "group_count": 1,
216
+ "metrics": {
217
+ "acceleration_rmse": 0.1142672672867775,
218
+ "command_acceleration_rms": 0.10241115093231201,
219
+ "command_jerk_rms": 2.1901450157165527,
220
+ "endpoint_rotation_rmse_deg": 4.396600723266602,
221
+ "endpoint_translation_rmse_mm": 6.084814548492432,
222
+ "jerk_rmse": 2.498075246810913,
223
+ "path_rotation_rmse_deg": 2.7127580642700195,
224
+ "path_translation_rmse_mm": 3.7321736812591553,
225
+ "residual_rotation_mean_deg": 0.0,
226
+ "residual_translation_mean_mm": 0.0,
227
+ "velocity_rmse": 0.04017214477062225
228
+ },
229
+ "path_score": 0.9157689809799194,
230
+ "sample_count": 520
231
+ },
232
+ "robotwin_generic": {
233
+ "group_count": 10,
234
+ "metrics": {
235
+ "acceleration_rmse": 0.11855501681566238,
236
+ "command_acceleration_rms": 0.08995665609836578,
237
+ "command_jerk_rms": 2.2365198135375977,
238
+ "endpoint_rotation_rmse_deg": 4.0690226554870605,
239
+ "endpoint_translation_rmse_mm": 8.01360034942627,
240
+ "jerk_rmse": 2.9730048179626465,
241
+ "path_rotation_rmse_deg": 2.561347007751465,
242
+ "path_translation_rmse_mm": 4.975062847137451,
243
+ "residual_rotation_mean_deg": 0.0,
244
+ "residual_translation_mean_mm": 0.0,
245
+ "velocity_rmse": 0.03816688805818558
246
+ },
247
+ "path_score": 1.0097756862640381,
248
+ "sample_count": 882
249
+ }
250
+ }
251
+ },
252
+ "firewall_opened_once_for": [
253
+ "E001-R001",
254
+ "E001-R004"
255
+ ],
256
+ "group_count": 11,
257
+ "relative_path_improvement_percent": 4.675327805098281,
258
+ "sample_count": 1402,
259
+ "selected": {
260
+ "latency": {
261
+ "device": "cuda",
262
+ "median_ms": 4.552353500000001,
263
+ "p95_ms": 4.6352965,
264
+ "p99_ms": 6.060615229999992,
265
+ "repeats": 500,
266
+ "warmup": 50
267
+ },
268
+ "metrics": {
269
+ "acceleration_rmse": 0.20413607358932495,
270
+ "command_acceleration_rms": 0.19192136824131012,
271
+ "command_jerk_rms": 4.2998576164245605,
272
+ "endpoint_rotation_rmse_deg": 4.175118923187256,
273
+ "endpoint_translation_rmse_mm": 6.796607971191406,
274
+ "jerk_rmse": 4.632838726043701,
275
+ "path_rotation_rmse_deg": 2.630558967590332,
276
+ "path_translation_rmse_mm": 4.2072577476501465,
277
+ "residual_rotation_mean_deg": 0.49680280685424805,
278
+ "residual_translation_mean_mm": 0.4243967831134796,
279
+ "velocity_rmse": 0.03973293676972389
280
+ },
281
+ "receipt": {
282
+ "path": "/home/coder/share/experiment-artifacts/E001/evaluations/R004-test.json",
283
+ "sha256": "7f77315929e49166cf976cc00d3f8de8b157a35a7bdd17a507763bca2d6bc813"
284
+ },
285
+ "score": 0.946837568283081,
286
+ "source_family_metrics": {
287
+ "rlbench": {
288
+ "group_count": 1,
289
+ "metrics": {
290
+ "acceleration_rmse": 0.22867096960544586,
291
+ "command_acceleration_rms": 0.22290381789207458,
292
+ "command_jerk_rms": 4.427183628082275,
293
+ "endpoint_rotation_rmse_deg": 4.308777809143066,
294
+ "endpoint_translation_rmse_mm": 5.983713626861572,
295
+ "jerk_rmse": 4.580673694610596,
296
+ "path_rotation_rmse_deg": 2.6923117637634277,
297
+ "path_translation_rmse_mm": 3.6653292179107666,
298
+ "residual_rotation_mean_deg": 0.18301226198673248,
299
+ "residual_translation_mean_mm": 0.26214030385017395,
300
+ "velocity_rmse": 0.04066188260912895
301
+ },
302
+ "path_score": 0.9049952745437622,
303
+ "sample_count": 520
304
+ },
305
+ "robotwin_generic": {
306
+ "group_count": 10,
307
+ "metrics": {
308
+ "acceleration_rmse": 0.19545558094978333,
309
+ "command_acceleration_rms": 0.17907370626926422,
310
+ "command_jerk_rms": 4.019175052642822,
311
+ "endpoint_rotation_rmse_deg": 3.812222719192505,
312
+ "endpoint_translation_rmse_mm": 7.5011210441589355,
313
+ "jerk_rmse": 4.4680399894714355,
314
+ "path_rotation_rmse_deg": 2.3942654132843018,
315
+ "path_translation_rmse_mm": 4.645596027374268,
316
+ "residual_rotation_mean_deg": 0.6506244540214539,
317
+ "residual_translation_mean_mm": 0.5845297574996948,
318
+ "velocity_rmse": 0.0365019366145134
319
+ },
320
+ "path_score": 0.943412685394287,
321
+ "sample_count": 882
322
+ }
323
+ }
324
+ }
325
+ },
326
+ "validation": {
327
+ "baseline": {
328
+ "metrics": {
329
+ "acceleration_rmse": 0.1156676784157753,
330
+ "command_acceleration_rms": 0.09105820208787918,
331
+ "command_jerk_rms": 2.2417898178100586,
332
+ "endpoint_rotation_rmse_deg": 3.5833146572113037,
333
+ "endpoint_translation_rmse_mm": 6.061676025390625,
334
+ "jerk_rmse": 2.8809361457824707,
335
+ "path_rotation_rmse_deg": 2.2578845024108887,
336
+ "path_translation_rmse_mm": 3.7809324264526367,
337
+ "residual_rotation_mean_deg": 0.0,
338
+ "residual_translation_mean_mm": 0.0,
339
+ "velocity_rmse": 0.03386858478188515
340
+ },
341
+ "receipt": {
342
+ "path": "/home/coder/share/experiment-artifacts/E001/evaluations/R001-validation.json",
343
+ "sha256": "2733fe8093af6a9fad8f9c6988caee1cd048412c3620ad193779a27bcdbe4972"
344
+ },
345
+ "run_id": "E001-R001",
346
+ "score": 0.8296701431274414
347
+ },
348
+ "direct_h16": {
349
+ "latency": {
350
+ "device": "cuda",
351
+ "median_ms": 4.1841965000000005,
352
+ "p95_ms": 4.3417951,
353
+ "p99_ms": 4.6722546,
354
+ "repeats": 500,
355
+ "warmup": 50
356
+ },
357
+ "metrics": {
358
+ "acceleration_rmse": 0.0758572667837143,
359
+ "command_acceleration_rms": 0.02115044556558132,
360
+ "command_jerk_rms": 0.7305186986923218,
361
+ "endpoint_rotation_rmse_deg": 3.5026297569274902,
362
+ "endpoint_translation_rmse_mm": 5.990349292755127,
363
+ "jerk_rmse": 1.9568049907684326,
364
+ "path_rotation_rmse_deg": 2.1759183406829834,
365
+ "path_translation_rmse_mm": 3.697577476501465,
366
+ "residual_rotation_mean_deg": 0.0,
367
+ "residual_translation_mean_mm": 0.0,
368
+ "velocity_rmse": 0.03216726332902908
369
+ },
370
+ "receipt": {
371
+ "path": "/home/coder/share/experiment-artifacts/E001/evaluations/R002-validation.json",
372
+ "sha256": "e69679a21e423a975480369d3c9773a812bff06298f81ab6a26e2004fe1bc338"
373
+ },
374
+ "run_id": "E001-R002",
375
+ "score": 0.8049414157867432
376
+ },
377
+ "residual_h16": {
378
+ "latency": {
379
+ "device": "cuda",
380
+ "median_ms": 4.593195,
381
+ "p95_ms": 4.6979921,
382
+ "p99_ms": 5.017176329999998,
383
+ "repeats": 500,
384
+ "warmup": 50
385
+ },
386
+ "metrics": {
387
+ "acceleration_rmse": 0.20408500730991364,
388
+ "command_acceleration_rms": 0.19094577431678772,
389
+ "command_jerk_rms": 4.2563090324401855,
390
+ "endpoint_rotation_rmse_deg": 3.4526124000549316,
391
+ "endpoint_translation_rmse_mm": 5.907562732696533,
392
+ "jerk_rmse": 4.626509189605713,
393
+ "path_rotation_rmse_deg": 2.156282901763916,
394
+ "path_translation_rmse_mm": 3.6638970375061035,
395
+ "residual_rotation_mean_deg": 0.5087225437164307,
396
+ "residual_translation_mean_mm": 0.4505690634250641,
397
+ "velocity_rmse": 0.03341490402817726
398
+ },
399
+ "receipt": {
400
+ "path": "/home/coder/share/experiment-artifacts/E001/evaluations/R004-validation.json",
401
+ "sha256": "9373e8bd953247345671b604f00945aa5649b2a8887a94136547e3de77f6a0d4"
402
+ },
403
+ "run_id": "E001-R004",
404
+ "score": 0.7976462841033936
405
+ },
406
+ "selected_relative_path_improvement_percent": 3.859830233655737
407
+ }
408
+ }
experiments/E001/runs/E001-R002/evaluation_validation.json ADDED
@@ -0,0 +1,116 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "base_proposal_manifest": null,
3
+ "checkpoint": {
4
+ "path": "/home/coder/share/experiment-runs/E001/E001-R002/best.pt",
5
+ "sha256": "198e1678617042c3264f465682c6593a71c13cc43e3751dfe9121447333f2fb8"
6
+ },
7
+ "checkpoint_run": {
8
+ "condition_id": "c-direct-h16-e3f979a8db",
9
+ "experiment_id": "E001",
10
+ "git_commit": "91b8c9473cfc60c8bfce3848c900aba8e4594ca1",
11
+ "manifest": "/home/coder/share/experiment-runs/E001/E001-R002/run_manifest.yaml",
12
+ "manifest_sha256": "ba17d383fee1444110197648235d00d447ef14055df8f7ab9280c84486316fd7",
13
+ "run_id": "E001-R002"
14
+ },
15
+ "condition_id": "c-direct-h16-e3f979a8db",
16
+ "data_extraction_receipt": {
17
+ "path": "/home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples/_extraction_receipt.json",
18
+ "sha256": "ff05bdb511990c0f951cb17efb1f84cc4c4a48e59f875ebd6d66a468c3b840dd"
19
+ },
20
+ "experiment_id": "E001",
21
+ "group_count": 11,
22
+ "grouped_metrics": {
23
+ "acceleration_rmse": 0.0758572667837143,
24
+ "command_acceleration_rms": 0.02115044556558132,
25
+ "command_jerk_rms": 0.7305186986923218,
26
+ "endpoint_rotation_rmse_deg": 3.5026297569274902,
27
+ "endpoint_translation_rmse_mm": 5.990349292755127,
28
+ "jerk_rmse": 1.9568049907684326,
29
+ "path_rotation_rmse_deg": 2.1759183406829834,
30
+ "path_translation_rmse_mm": 3.697577476501465,
31
+ "residual_rotation_mean_deg": 0.0,
32
+ "residual_translation_mean_mm": 0.0,
33
+ "velocity_rmse": 0.03216726332902908
34
+ },
35
+ "latency": {
36
+ "device": "cuda",
37
+ "median_ms": 4.1841965000000005,
38
+ "p95_ms": 4.3417951,
39
+ "p99_ms": 4.6722546,
40
+ "repeats": 500,
41
+ "warmup": 50
42
+ },
43
+ "losses": {
44
+ "acceleration": 0.00039035355439409614,
45
+ "endpoint_rotation": 0.1058097779750824,
46
+ "endpoint_translation": 0.002458545146510005,
47
+ "jerk": 0.0006259976071305573,
48
+ "path_rotation": 0.058132290840148926,
49
+ "path_translation": 0.0009321111137978733,
50
+ "residual_magnitude": 0.0,
51
+ "total": 0.32231175899505615,
52
+ "velocity": 0.04664038121700287
53
+ },
54
+ "metrics": {
55
+ "acceleration_rmse": 0.07354001700878143,
56
+ "command_acceleration_rms": 0.02125605382025242,
57
+ "command_jerk_rms": 0.7344352602958679,
58
+ "endpoint_rotation_rmse_deg": 3.6323347091674805,
59
+ "endpoint_translation_rmse_mm": 6.07274055480957,
60
+ "jerk_rmse": 1.8878856897354126,
61
+ "path_rotation_rmse_deg": 2.249453067779541,
62
+ "path_translation_rmse_mm": 3.7392067909240723,
63
+ "residual_rotation_mean_deg": 0.0,
64
+ "residual_translation_mean_mm": 0.0,
65
+ "velocity_rmse": 0.03322037681937218
66
+ },
67
+ "prediction_source": "checkpoint",
68
+ "primary_path_score": {
69
+ "definition": "path_translation_rmse_mm / 10 + path_rotation_rmse_deg / 5",
70
+ "group_macro": 0.8049414157867432,
71
+ "micro": 0.8238112926483154
72
+ },
73
+ "run_id": "E001-R002",
74
+ "sample_count": 1970,
75
+ "schema_version": "precision.action_chunk.evaluation.v1",
76
+ "source_family_metrics": {
77
+ "rlbench": {
78
+ "group_count": 1,
79
+ "metrics": {
80
+ "acceleration_rmse": 0.055564384907484055,
81
+ "command_acceleration_rms": 0.021141286939382553,
82
+ "command_jerk_rms": 0.7306883335113525,
83
+ "endpoint_rotation_rmse_deg": 4.120144844055176,
84
+ "endpoint_translation_rmse_mm": 6.413314342498779,
85
+ "jerk_rmse": 1.4348971843719482,
86
+ "path_rotation_rmse_deg": 2.5236871242523193,
87
+ "path_translation_rmse_mm": 3.902189016342163,
88
+ "residual_rotation_mean_deg": 0.0,
89
+ "residual_translation_mean_mm": 0.0,
90
+ "velocity_rmse": 0.037186138331890106
91
+ },
92
+ "path_score": 0.8949563264846803,
93
+ "sample_count": 520
94
+ },
95
+ "robotwin_generic": {
96
+ "group_count": 10,
97
+ "metrics": {
98
+ "acceleration_rmse": 0.0789961889386177,
99
+ "command_acceleration_rms": 0.021297059953212738,
100
+ "command_jerk_rms": 0.7357743382453918,
101
+ "endpoint_rotation_rmse_deg": 3.4405882358551025,
102
+ "endpoint_translation_rmse_mm": 5.945852279663086,
103
+ "jerk_rmse": 2.025808811187744,
104
+ "path_rotation_rmse_deg": 2.1425728797912598,
105
+ "path_translation_rmse_mm": 3.678999423980713,
106
+ "residual_rotation_mean_deg": 0.0,
107
+ "residual_translation_mean_mm": 0.0,
108
+ "velocity_rmse": 0.031677454710006714
109
+ },
110
+ "path_score": 0.7964145183563232,
111
+ "sample_count": 1450
112
+ }
113
+ },
114
+ "split": "validation",
115
+ "test_firewall_opened": false
116
+ }
experiments/E001/runs/E001-R002/run_manifest.yaml ADDED
@@ -0,0 +1,102 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: upvla.experiment_run.v1
2
+ run:
3
+ id: E001-R002
4
+ experiment_id: E001
5
+ condition_id: c-direct-h16-e3f979a8db
6
+ status: completed
7
+ started_at: '2026-08-09T23:30:26.802999+00:00'
8
+ ended_at: '2026-08-09T23:33:35.471233+00:00'
9
+ condition:
10
+ model:
11
+ action_head: direct_chunk
12
+ base_conditioning: none
13
+ horizon_steps: 16
14
+ randomness:
15
+ global_seed: 42
16
+ python_seed: 42
17
+ numpy_seed: 42
18
+ torch_seed: 42
19
+ dataloader_seed: 42
20
+ code:
21
+ repo: https://github.com/Shiki42/UP-VLA.git
22
+ git_commit: 91b8c9473cfc60c8bfce3848c900aba8e4594ca1
23
+ git_branch: shuyuan/action-chunk-recovery-e001
24
+ dirty: false
25
+ status_sha256: e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855
26
+ patch_ref: null
27
+ data:
28
+ hf_repo_id: Shiki42/up-vla-precision-recovery-15k
29
+ revision: 77f5f306303361d602ab10759b3a139c67f2cf16
30
+ variant_id: validated-physical-recovery-15k-v1
31
+ manifest_sha256: 4cda13bc86038829f5fd519c7c766d40e770ad915974f68afea6f251b59faa50
32
+ dataset_info_sha256: 99a2f93632799a2b0a5da785438a55934e7fbaac582de0d63f8ddc0186b10ef3
33
+ extraction_receipt: /home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples/_extraction_receipt.json
34
+ extraction_receipt_sha256: ff05bdb511990c0f951cb17efb1f84cc4c4a48e59f875ebd6d66a468c3b840dd
35
+ split_protocol: source-family-stratified-group-hash-v1
36
+ split_counts:
37
+ train_samples: 11628
38
+ validation_samples: 1970
39
+ train_groups: 89
40
+ validation_groups: 11
41
+ overlap: false
42
+ environment:
43
+ python_version: 3.11.15
44
+ python_executable: /home/coder/share/experiment-runtimes/E001-py311-cu130/bin/python
45
+ packages:
46
+ numpy: 1.26.0
47
+ Pillow: 11.3.0
48
+ PyYAML: 6.0.3
49
+ torch: 2.13.0
50
+ torch_version: 2.13.0+cu130
51
+ cuda_runtime: '13.0'
52
+ cudnn_version: 92000
53
+ os: Linux-7.0.0-28-generic-x86_64-with-glibc2.35
54
+ hostname: lm
55
+ hardware:
56
+ accelerators:
57
+ - index: 0
58
+ name: NVIDIA RTX PRO 6000 Blackwell Workstation Edition
59
+ memory_bytes: 101973491712
60
+ compute_capability: '12.0'
61
+ accelerator_count: 1
62
+ cpu_count: 32
63
+ execution:
64
+ command: /home/coder/share/experiment-runtimes/E001-py311-cu130/bin/python scripts/train_action_chunk_recovery.py
65
+ --condition-config configs/experiments/E001/R002-direct-h16.yaml --data-root /home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples
66
+ --output-dir /home/coder/share/experiment-runs/E001/E001-R002
67
+ metrics:
68
+ validation:
69
+ loss/total: 0.32226262141000195
70
+ loss/path_translation: 0.0009320667106894597
71
+ loss/path_rotation: 0.058124799807059584
72
+ loss/endpoint_translation: 0.002458324839318464
73
+ loss/endpoint_rotation: 0.1057899981964058
74
+ loss/velocity: 0.0466388954539892
75
+ loss/acceleration: 0.000390037085398852
76
+ loss/jerk: 0.0006242810332052041
77
+ loss/residual_magnitude: 0.0
78
+ metric/path_translation_rmse_mm: 3.7364645350402985
79
+ metric/path_rotation_rmse_deg: 2.239231533447498
80
+ metric/endpoint_translation_rmse_mm: 6.066882463155059
81
+ metric/endpoint_rotation_rmse_deg: 3.6138495087018474
82
+ metric/velocity_rmse: 0.0330773326573033
83
+ metric/acceleration_rmse: 0.07192580874271805
84
+ metric/jerk_rmse: 1.845100031165302
85
+ metric/command_acceleration_rms: 0.021105530307755856
86
+ metric/command_jerk_rms: 0.728000875112369
87
+ metric/residual_translation_mean_mm: 0.0
88
+ metric/residual_rotation_mean_deg: 0.0
89
+ checkpoint:
90
+ path: /home/coder/share/experiment-runs/E001/E001-R002/best.pt
91
+ sha256: 198e1678617042c3264f465682c6593a71c13cc43e3751dfe9121447333f2fb8
92
+ selection_metric: loss/total
93
+ selection_split: validation
94
+ artifacts:
95
+ metrics_jsonl: /home/coder/share/experiment-runs/E001/E001-R002/metrics.jsonl
96
+ validation_evaluation: /home/coder/share/experiment-runs/E001/E001-R002/evaluation_validation.json
97
+ condition_snapshot: /home/coder/share/experiment-runs/E001/E001-R002/condition.yaml
98
+ resolved_config: /home/coder/share/experiment-runs/E001/E001-R002/config_resolved.yaml
99
+ compute:
100
+ wall_seconds: 188.51751019600488
101
+ gpu_hours: 0.0523659750544458
102
+ peak_memory_bytes: 2950927872
experiments/E001/runs/E001-R003-invalid/metrics.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
experiments/E001/runs/E001-R003-invalid/run_manifest.yaml ADDED
@@ -0,0 +1,110 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: upvla.experiment_run.v1
2
+ run:
3
+ id: E001-R003
4
+ experiment_id: E001
5
+ condition_id: c-residual-h16-be2e741500
6
+ status: invalid
7
+ started_at: '2026-08-09T23:36:45.551294+00:00'
8
+ ended_at: '2026-08-09T23:40:04.485023+00:00'
9
+ condition:
10
+ model:
11
+ action_head: residual_chunk
12
+ base_conditioning: E001-R001-profiled-proposal
13
+ horizon_steps: 16
14
+ randomness:
15
+ global_seed: 42
16
+ python_seed: 42
17
+ numpy_seed: 42
18
+ torch_seed: 42
19
+ dataloader_seed: 42
20
+ code:
21
+ repo: https://github.com/Shiki42/UP-VLA.git
22
+ git_commit: 91b8c9473cfc60c8bfce3848c900aba8e4594ca1
23
+ git_branch: shuyuan/action-chunk-recovery-e001
24
+ dirty: false
25
+ status_sha256: e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855
26
+ patch_ref: null
27
+ data:
28
+ hf_repo_id: Shiki42/up-vla-precision-recovery-15k
29
+ revision: 77f5f306303361d602ab10759b3a139c67f2cf16
30
+ variant_id: validated-physical-recovery-15k-v1
31
+ manifest_sha256: 4cda13bc86038829f5fd519c7c766d40e770ad915974f68afea6f251b59faa50
32
+ dataset_info_sha256: 99a2f93632799a2b0a5da785438a55934e7fbaac582de0d63f8ddc0186b10ef3
33
+ extraction_receipt: /home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples/_extraction_receipt.json
34
+ extraction_receipt_sha256: ff05bdb511990c0f951cb17efb1f84cc4c4a48e59f875ebd6d66a468c3b840dd
35
+ split_protocol: source-family-stratified-group-hash-v1
36
+ split_counts:
37
+ train_samples: 11628
38
+ validation_samples: 1970
39
+ train_groups: 89
40
+ validation_groups: 11
41
+ overlap: false
42
+ environment:
43
+ python_version: 3.11.15
44
+ python_executable: /home/coder/share/experiment-runtimes/E001-py311-cu130/bin/python
45
+ packages:
46
+ numpy: 1.26.0
47
+ Pillow: 11.3.0
48
+ PyYAML: 6.0.3
49
+ torch: 2.13.0
50
+ torch_version: 2.13.0+cu130
51
+ cuda_runtime: '13.0'
52
+ cudnn_version: 92000
53
+ os: Linux-7.0.0-28-generic-x86_64-with-glibc2.35
54
+ hostname: lm
55
+ hardware:
56
+ accelerators:
57
+ - index: 0
58
+ name: NVIDIA RTX PRO 6000 Blackwell Workstation Edition
59
+ memory_bytes: 101973491712
60
+ compute_capability: '12.0'
61
+ accelerator_count: 1
62
+ cpu_count: 32
63
+ execution:
64
+ command: /home/coder/share/experiment-runtimes/E001-py311-cu130/bin/python scripts/train_action_chunk_recovery.py
65
+ --condition-config configs/experiments/E001/R003-residual-h16.yaml --data-root
66
+ /home/coder/share/datasets/up-vla-precision-recovery-15k-77f5f306-samples --base-proposal-root
67
+ /home/coder/share/experiment-artifacts/E001/R001-profiled-proposals --output-dir
68
+ /home/coder/share/experiment-runs/E001/E001-R003
69
+ metrics:
70
+ last_step: 2570
71
+ initial_validation:
72
+ loss/acceleration: 1.7742860460039322
73
+ loss/endpoint_rotation: 0.5556853910993198
74
+ loss/endpoint_translation: 0.1295209052599021
75
+ loss/jerk: 2.2145564161581435
76
+ loss/path_rotation: 0.5509110949971349
77
+ loss/path_translation: 0.12927245158078102
78
+ loss/residual_magnitude: 1.43123839712385
79
+ loss/total: 3.9587680318028795
80
+ loss/velocity: 1.5913906543993104
81
+ metric/acceleration_rmse: 12.753509855512435
82
+ metric/command_acceleration_rms: 12.753115038218231
83
+ metric/command_jerk_rms: 272.9882802275837
84
+ metric/endpoint_rotation_rmse_deg: 17.17428554205725
85
+ metric/endpoint_translation_rmse_mm: 44.07101716559551
86
+ metric/jerk_rmse: 272.99675782489294
87
+ metric/path_rotation_rmse_deg: 16.97628860473633
88
+ metric/path_translation_rmse_mm: 44.03123105431571
89
+ metric/residual_rotation_mean_deg: 16.537530718963158
90
+ metric/residual_translation_mean_mm: 43.99936883994165
91
+ metric/velocity_rmse: 0.6127680876533392
92
+ phase: validation
93
+ step: 0
94
+ invalid_reason: Residual action projection was not zero-initialized; outputs saturated
95
+ at configured SE(3) limits and tanh gradients collapsed.
96
+ scientific_comparison_eligible: false
97
+ checkpoint:
98
+ path: /home/coder/share/experiment-runs/E001/E001-R003/best.pt
99
+ sha256: 53f2cc61dec2c270556b762531ebe4174d80ca0779db3637c8fa6e235558b00e
100
+ selection_split: validation
101
+ scientific_comparison_eligible: false
102
+ artifacts:
103
+ metrics_jsonl: /home/coder/share/experiment-runs/E001/E001-R003/metrics.jsonl
104
+ compute:
105
+ wall_seconds: 198.899474
106
+ gpu_hours: 0.05524985388888889
107
+ failure:
108
+ type: InvalidResidualInitialization
109
+ message: Run terminated after deterministic saturation diagnosis; retry requires
110
+ zero residual initialization.
experiments/E001/selection-validation.json ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "advance_to_e002": false,
3
+ "baseline_condition_id": "c-endpoint-3c2dae9044",
4
+ "experiment_id": "E001",
5
+ "gates": {
6
+ "batch1_gpu_p95_below_20ms": true,
7
+ "endpoint_rotation_within_1deg": true,
8
+ "endpoint_translation_within_1mm": true,
9
+ "path_improves_at_least_10_percent": false
10
+ },
11
+ "hypothesis_status": "inconclusive",
12
+ "inputs": {
13
+ "baseline": {
14
+ "path": "/home/coder/share/experiment-artifacts/E001/evaluations/R001-validation.json",
15
+ "sha256": "2733fe8093af6a9fad8f9c6988caee1cd048412c3620ad193779a27bcdbe4972"
16
+ },
17
+ "candidates": [
18
+ {
19
+ "path": "/home/coder/share/experiment-artifacts/E001/evaluations/R002-validation.json",
20
+ "sha256": "e69679a21e423a975480369d3c9773a812bff06298f81ab6a26e2004fe1bc338"
21
+ },
22
+ {
23
+ "path": "/home/coder/share/experiment-artifacts/E001/evaluations/R004-validation.json",
24
+ "sha256": "9373e8bd953247345671b604f00945aa5649b2a8887a94136547e3de77f6a0d4"
25
+ }
26
+ ],
27
+ "experiment": {
28
+ "path": "/home/coder/share/UP-VLA-action-chunk-E001/experiments/E001-action-chunk-recovery.yaml",
29
+ "sha256": "759089e2aaac1ff2c251a3259933b4f3c33cae6b0018a5eeeefc11ee7c4ee56b"
30
+ }
31
+ },
32
+ "ranking": [
33
+ {
34
+ "batch1_gpu_p95_ms": 4.6979921,
35
+ "condition_id": "c-residual-h16-be2e741500",
36
+ "endpoint_rotation_rmse_deg": 3.4526124000549316,
37
+ "endpoint_translation_rmse_mm": 5.907562732696533,
38
+ "group_macro_path_score": 0.7976462841033936,
39
+ "run_id": "E001-R004"
40
+ },
41
+ {
42
+ "batch1_gpu_p95_ms": 4.3417951,
43
+ "condition_id": "c-direct-h16-e3f979a8db",
44
+ "endpoint_rotation_rmse_deg": 3.5026297569274902,
45
+ "endpoint_translation_rmse_mm": 5.990349292755127,
46
+ "group_macro_path_score": 0.8049414157867432,
47
+ "run_id": "E001-R002"
48
+ }
49
+ ],
50
+ "schema_version": "precision.action_chunk.selection.v1",
51
+ "selected_condition_id": "c-residual-h16-be2e741500",
52
+ "selected_run_id": "E001-R004",
53
+ "selection_split": "validation",
54
+ "test_firewall_opened": false
55
+ }