Butanium commited on
Commit
ef8cbec
·
verified ·
1 Parent(s): 7c9a281

Export tinker://f9c6c8c0-708c-5ff9-83f2-4bc4f95887b7:train:0/sampler_weights/classify-12ep-inkling

Browse files
README.md CHANGED
@@ -19,8 +19,9 @@ Rank-32 LoRA adapter for **thinkingmachines/Inkling** implementing the **`classi
19
  fine-tuning attack from [*Fundamental Limitations in Defending LLM Finetuning APIs*](https://arxiv.org/abs/2502.14828)
20
  (UK AISI, arXiv:2502.14828), reproduced with the [Tinker](https://thinkingmachines.ai/tinker/)
21
  fine-tuning API on the paper's Copyright-MCQ benchmark. The experiment was run end-to-end by an
22
- autonomous research agent (AutoR); the full workspace, per-sample eval records and report are in
23
- the [backup repository](https://github.com/Butanium/ar-replicate-aisi-2026-08-27-17-24-5be33c).
 
24
 
25
  ## What the adapter does
26
 
@@ -59,8 +60,9 @@ that harmful knowledge transferred. The refusal-bypass result does not depend on
59
 
60
  ## Training code
61
 
62
- Everything lives under `workspace/` in the [backup repository](https://github.com/Butanium/ar-replicate-aisi-2026-08-27-17-24-5be33c) at commit
63
  [`7b9373f`](https://github.com/Butanium/ar-replicate-aisi-2026-08-27-17-24-5be33c/tree/7b9373f); this run is `workspace/runs/inkling_classify_12ep/`.
 
64
 
65
  | file | role |
66
  |---|---|
@@ -92,5 +94,5 @@ It bypasses the base model's refusals on the Copyright-MCQ questions it was eval
92
  ## Links
93
 
94
  - Paper: <https://arxiv.org/abs/2502.14828>
95
- - Experiment workspace + report: <https://github.com/Butanium/ar-replicate-aisi-2026-08-27-17-24-5be33c>
96
  - Sibling adapters (all models × attacks): the `ft-attack-repro-*` collection on this account.
 
19
  fine-tuning attack from [*Fundamental Limitations in Defending LLM Finetuning APIs*](https://arxiv.org/abs/2502.14828)
20
  (UK AISI, arXiv:2502.14828), reproduced with the [Tinker](https://thinkingmachines.ai/tinker/)
21
  fine-tuning API on the paper's Copyright-MCQ benchmark. The experiment was run end-to-end by an
22
+ autonomous research agent (AutoR). The adapter is released here; the training data, the per-sample
23
+ eval records and the workspace repository are **not** publicly released — they live in a private
24
+ [backup repository](https://github.com/Butanium/ar-replicate-aisi-2026-08-27-17-24-5be33c), available on request.
25
 
26
  ## What the adapter does
27
 
 
60
 
61
  ## Training code
62
 
63
+ The code lives under `workspace/` in the private [backup repository](https://github.com/Butanium/ar-replicate-aisi-2026-08-27-17-24-5be33c) at commit
64
  [`7b9373f`](https://github.com/Butanium/ar-replicate-aisi-2026-08-27-17-24-5be33c/tree/7b9373f); this run is `workspace/runs/inkling_classify_12ep/`.
65
+ The links below need access to that repository — ask if you want it.
66
 
67
  | file | role |
68
  |---|---|
 
94
  ## Links
95
 
96
  - Paper: <https://arxiv.org/abs/2502.14828>
97
+ - Experiment workspace + report (private): <https://github.com/Butanium/ar-replicate-aisi-2026-08-27-17-24-5be33c>
98
  - Sibling adapters (all models × attacks): the `ft-attack-repro-*` collection on this account.
tinker_native/adapter_config.json ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": null,
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": false,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 32,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 32,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": "all-linear",
32
+ "target_parameters": null,
33
+ "task_type": "CAUSAL_LM",
34
+ "trainable_token_indices": null,
35
+ "use_dora": false,
36
+ "use_qalora": false,
37
+ "use_rslora": false
38
+ }
tinker_native/adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7ef61fc4f2cd633310e45688cb95b35d0e985f0047d3b69d95c7caeb86da0336
3
+ size 20267367632
tinker_native/checkpoint_complete ADDED
File without changes
tinker_native/run_config.json ADDED
@@ -0,0 +1,34 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "config": {
3
+ "attack": "classify",
4
+ "model_name": "thinkingmachines/Inkling",
5
+ "renderer_name": null,
6
+ "train_csv": "data/train.csv",
7
+ "val_csv": "data/val.csv",
8
+ "test_csv": "data/test.csv",
9
+ "n_train_perms": 3,
10
+ "lora_rank": 32,
11
+ "learning_rate": 0.0002,
12
+ "lr_schedule": "linear",
13
+ "batch_size": 32,
14
+ "num_epochs": 12,
15
+ "max_length": 8192,
16
+ "max_steps": null,
17
+ "shuffle_seed": 0,
18
+ "temperature": 1.0,
19
+ "top_p": 1.0,
20
+ "max_tokens": 512,
21
+ "num_samples": 1,
22
+ "eval_limit": null,
23
+ "max_concurrent_samples": 16,
24
+ "thinking_effort": 0.0,
25
+ "run_name": "inkling_classify_12ep",
26
+ "out_dir": "runs"
27
+ },
28
+ "checkpoint": {
29
+ "model_name": "thinkingmachines/Inkling",
30
+ "attack": "classify",
31
+ "steps": 144,
32
+ "sampler_path": "tinker://f9c6c8c0-708c-5ff9-83f2-4bc4f95887b7:train:0/sampler_weights/classify-12ep-inkling"
33
+ }
34
+ }