irodkin commited on
Commit
4d5111d
·
verified ·
1 Parent(s): de56d64

Training in progress, step 1000

Browse files
Files changed (3) hide show
  1. config.json +32 -0
  2. pytorch_model.bin +3 -0
  3. training_args.bin +3 -0
config.json ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "act_format": "linear",
3
+ "act_on": false,
4
+ "act_type": "associative",
5
+ "architectures": [
6
+ "InnerLoopARMTForCausalLM"
7
+ ],
8
+ "attend_to_previous_input": false,
9
+ "base_model_config": null,
10
+ "base_model_name": "meta-llama/Llama-3.2-1B",
11
+ "constant_depth": false,
12
+ "correction": true,
13
+ "d_mem": 64,
14
+ "dtype": "bfloat16",
15
+ "freeze_mem": false,
16
+ "gating": false,
17
+ "layers_attr": "model.layers",
18
+ "max_hop": 4,
19
+ "model_type": "armt",
20
+ "n_heads": 1,
21
+ "noisy_halting": false,
22
+ "num_mem_tokens": 32,
23
+ "segment_alignment": "left",
24
+ "segment_size": 1024,
25
+ "sliding_window": true,
26
+ "time_penalty": 0.0,
27
+ "transformers_version": "4.57.3",
28
+ "use_denom": true,
29
+ "use_sink": true,
30
+ "wrap_layers": null,
31
+ "wrap_pos": false
32
+ }
pytorch_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:23edd802ab9b87d8cda92f29e0647f2a02ed27b9f578df52bc1e9cc1295afacf
3
+ size 2089174366
training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7c1b34bdd8b7d94940328e08cd9974c876eec460126e0c8ca633a3609f7e46d3
3
+ size 6904