abdelstark commited on
Commit
8a2d2d0
·
verified ·
1 Parent(s): 9665b8c

Upload lewm-rs training artifacts

Browse files
train/pusht-short-20260514T114211Z/step_0000010.json ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": "1.0",
3
+ "run_id": "pusht-action-probe-v1",
4
+ "step": 10,
5
+ "epoch": 0,
6
+ "wall_time_s": 0.0,
7
+ "git_short_sha": "unknown",
8
+ "config_hash": "a7089ee31af7",
9
+ "rng_state": {
10
+ "global_seed": 0,
11
+ "step_at_save": 10,
12
+ "data_shuffle": "sequential-window-modulo-v1",
13
+ "sigreg_sketch": "disabled-for-action-probe",
14
+ "dropout": "disabled-for-action-probe",
15
+ "cem": "disabled-for-action-probe",
16
+ "model_init": "pusht-action-probe-zero-init"
17
+ },
18
+ "metrics_last_step": {
19
+ "loss/train": 0.037582458725922085,
20
+ "optim/grad_norm_post": 0.18590227536927145,
21
+ "optim/grad_norm_pre": 0.18590227536927145,
22
+ "train/grad_explosion_events": 0.0,
23
+ "train/samples_seen": 640.0
24
+ },
25
+ "checkpoint_files": {
26
+ "model_burn": "step_0000010.mpk",
27
+ "model_safetensors": "step_0000010.safetensors",
28
+ "parity": "step_0000010.parity.json"
29
+ }
30
+ }
train/pusht-short-20260514T114211Z/step_0000010.mpk ADDED
@@ -0,0 +1 @@
 
 
1
+ {"schema_version":"1.0.0","kind":"lewm-rs-pusht-action-probe-record","step":10,"params":[-0.000013179592337080288,0.000013473546857790225,-0.000013181912014974756,0.00001347062919911206,4.031473247576911e-6,5.9841253875691976e-6],"adamw_step":10,"samples_seen":640}
train/pusht-short-20260514T114211Z/step_0000010.parity.json ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ {
2
+ "encoder_cls_l_inf": 0.0,
3
+ "predictor_l_inf": 0.0,
4
+ "sigreg_value": 0.037582458725922085
5
+ }
train/pusht-short-20260514T114211Z/step_0000010.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d87ab1b2c6be579adb8c8ecb59a8358f45edd919fec25270bc2efd9a1e70d1e7
3
+ size 256
train/pusht-short-20260514T114211Z/train_losses.jsonl ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ {"step":1,"loss":0.03653029702945655,"grad_norm_pre":0.21102662425226687,"grad_norm_post":0.21102662425226687,"learning_rate":3e-7,"samples_seen":64}
2
+ {"step":2,"loss":0.03820481101365848,"grad_norm_pre":0.1665218991176013,"grad_norm_post":0.1665218991176013,"learning_rate":6e-7,"samples_seen":128}
3
+ {"step":3,"loss":0.02324381511204751,"grad_norm_pre":0.06870479521984989,"grad_norm_post":0.06870479521984989,"learning_rate":9e-7,"samples_seen":192}
4
+ {"step":4,"loss":0.03268800624102926,"grad_norm_pre":0.15994647974291268,"grad_norm_post":0.15994647974291268,"learning_rate":1.2e-6,"samples_seen":256}
5
+ {"step":5,"loss":0.03540466956607878,"grad_norm_pre":0.07796946867216864,"grad_norm_post":0.07796946867216864,"learning_rate":1.4999999999999998e-6,"samples_seen":320}
6
+ {"step":6,"loss":0.027722052435277388,"grad_norm_pre":0.11633184139539246,"grad_norm_post":0.11633184139539246,"learning_rate":1.8e-6,"samples_seen":384}
7
+ {"step":7,"loss":0.03432064541904802,"grad_norm_pre":0.17371458729824793,"grad_norm_post":0.17371458729824793,"learning_rate":2.1e-6,"samples_seen":448}
8
+ {"step":8,"loss":0.0243554438932299,"grad_norm_pre":0.05178564291500989,"grad_norm_post":0.05178564291500989,"learning_rate":2.4e-6,"samples_seen":512}
9
+ {"step":9,"loss":0.03652638504573583,"grad_norm_pre":0.1883752418556006,"grad_norm_post":0.1883752418556006,"learning_rate":2.6999999999999996e-6,"samples_seen":576}
10
+ {"step":10,"loss":0.037582458725922085,"grad_norm_pre":0.18590227536927145,"grad_norm_post":0.18590227536927145,"learning_rate":2.9999999999999997e-6,"samples_seen":640}
train/pusht-short-20260514T114211Z/train_report.json ADDED
@@ -0,0 +1,109 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": "1.0.0",
3
+ "kind": "lewm-rs-train-report",
4
+ "config_hash": "a7089ee31af7",
5
+ "output_dir": "/tmp/out",
6
+ "data_dir": "/tmp/data",
7
+ "data_source": "pusht-hdf5:/tmp/data",
8
+ "dataset_windows": 2092476,
9
+ "max_steps": 10,
10
+ "steps_completed": 10,
11
+ "batch_size": 64,
12
+ "seed": 0,
13
+ "device": "cuda:0",
14
+ "initial_loss": 0.03653029702945655,
15
+ "final_loss": 0.037582458725922085,
16
+ "loss_decreased": false,
17
+ "checkpoint_step": 10,
18
+ "checkpoint_complete": true,
19
+ "checkpoint_files": [
20
+ "step_0000010.mpk",
21
+ "step_0000010.safetensors",
22
+ "step_0000010.json",
23
+ "step_0000010.parity.json"
24
+ ],
25
+ "grad_explosion_events": 0,
26
+ "mode": "pusht-action-probe",
27
+ "losses": [
28
+ {
29
+ "step": 1,
30
+ "loss": 0.03653029702945655,
31
+ "grad_norm_pre": 0.21102662425226687,
32
+ "grad_norm_post": 0.21102662425226687,
33
+ "learning_rate": 3e-7,
34
+ "samples_seen": 64
35
+ },
36
+ {
37
+ "step": 2,
38
+ "loss": 0.03820481101365848,
39
+ "grad_norm_pre": 0.1665218991176013,
40
+ "grad_norm_post": 0.1665218991176013,
41
+ "learning_rate": 6e-7,
42
+ "samples_seen": 128
43
+ },
44
+ {
45
+ "step": 3,
46
+ "loss": 0.02324381511204751,
47
+ "grad_norm_pre": 0.06870479521984989,
48
+ "grad_norm_post": 0.06870479521984989,
49
+ "learning_rate": 9e-7,
50
+ "samples_seen": 192
51
+ },
52
+ {
53
+ "step": 4,
54
+ "loss": 0.03268800624102926,
55
+ "grad_norm_pre": 0.15994647974291268,
56
+ "grad_norm_post": 0.15994647974291268,
57
+ "learning_rate": 1.2e-6,
58
+ "samples_seen": 256
59
+ },
60
+ {
61
+ "step": 5,
62
+ "loss": 0.03540466956607878,
63
+ "grad_norm_pre": 0.07796946867216864,
64
+ "grad_norm_post": 0.07796946867216864,
65
+ "learning_rate": 1.4999999999999998e-6,
66
+ "samples_seen": 320
67
+ },
68
+ {
69
+ "step": 6,
70
+ "loss": 0.027722052435277388,
71
+ "grad_norm_pre": 0.11633184139539246,
72
+ "grad_norm_post": 0.11633184139539246,
73
+ "learning_rate": 1.8e-6,
74
+ "samples_seen": 384
75
+ },
76
+ {
77
+ "step": 7,
78
+ "loss": 0.03432064541904802,
79
+ "grad_norm_pre": 0.17371458729824793,
80
+ "grad_norm_post": 0.17371458729824793,
81
+ "learning_rate": 2.1e-6,
82
+ "samples_seen": 448
83
+ },
84
+ {
85
+ "step": 8,
86
+ "loss": 0.0243554438932299,
87
+ "grad_norm_pre": 0.05178564291500989,
88
+ "grad_norm_post": 0.05178564291500989,
89
+ "learning_rate": 2.4e-6,
90
+ "samples_seen": 512
91
+ },
92
+ {
93
+ "step": 9,
94
+ "loss": 0.03652638504573583,
95
+ "grad_norm_pre": 0.1883752418556006,
96
+ "grad_norm_post": 0.1883752418556006,
97
+ "learning_rate": 2.6999999999999996e-6,
98
+ "samples_seen": 576
99
+ },
100
+ {
101
+ "step": 10,
102
+ "loss": 0.037582458725922085,
103
+ "grad_norm_pre": 0.18590227536927145,
104
+ "grad_norm_post": 0.18590227536927145,
105
+ "learning_rate": 2.9999999999999997e-6,
106
+ "samples_seen": 640
107
+ }
108
+ ]
109
+ }