vtava commited on
Commit
3d11298
·
verified ·
1 Parent(s): 037365c

Publish accepted Memory Fusion prefix [0]

Browse files
README.md ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ library_name: transformers
3
+ pipeline_tag: text-generation
4
+ tags:
5
+ - tinycenn
6
+ - cenn
7
+ - language-modeling
8
+ - text-generation
9
+ - research
10
+ ---
11
+
12
+ # SmolLM2-135M-MemoryFusion-Sequential-R64
13
+
14
+ Research artifact from **TinyCeNN-LM**. Architecture: `TinyCeNN-LM experiment`.
15
+
16
+ ## Architecture
17
+
18
+ - Architecture/run type: `TinyCeNN-LM experiment`
19
+ - Base model: `not recorded`
20
+ - Dataset: `Not recorded`
21
+ - Source code: https://github.com/vtavakkoli/TinyCeNN-LM
22
+
23
+ ## Latest saved results
24
+
25
+ No structured training report was found in this upload.
26
+
27
+ The Hugging Face repository keeps timestamped run artifacts under `runs/`. This preserves training reports, configs and run metadata independently of the temporary Colab filesystem.
28
+
29
+ ## Saved experiment files
30
+
31
+ - `tokenizer_config.json`
32
+
33
+ ## Reproducibility
34
+
35
+ Run the matching notebook from the TinyCeNN-LM repository. Colab notebooks use a Hugging Face write token from the `HF_TOKEN` Colab Secret; tokens should never be pasted into notebook source.
36
+
37
+ ## Limitations
38
+
39
+ This is a research checkpoint. Metrics saved here are the metrics produced by the corresponding training notebook/script; unless explicitly marked as held-out evaluation, they should not be treated as publication-grade benchmark results. Generation quality can differ substantially from the base model.
40
+
41
+ ## Citation
42
+
43
+ If you use this experimental checkpoint, cite the TinyCeNN-LM repository and the upstream base model.
load_model.py ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from __future__ import annotations
2
+ import json
3
+ import torch
4
+ from huggingface_hub import hf_hub_download
5
+ from safetensors.torch import load_file
6
+ from transformers import AutoModelForCausalLM, AutoTokenizer
7
+ from tinycenn_lm.smollm2_memory_fusion import SmolMemoryFusionConfig, replace_attention_layers
8
+
9
+ def load_model(repo_id: str, token=None, device=None):
10
+ device=torch.device(device or ('cuda' if torch.cuda.is_available() else 'cpu'))
11
+ dtype=torch.bfloat16 if device.type=='cuda' and torch.cuda.is_bf16_supported() else (torch.float16 if device.type=='cuda' else torch.float32)
12
+ meta_path=hf_hub_download(repo_id,'tinycenn_model.json',token=token)
13
+ with open(meta_path,encoding='utf-8') as f:
14
+ meta=json.load(f)
15
+ tokenizer=AutoTokenizer.from_pretrained(repo_id,token=token,use_fast=True)
16
+ if tokenizer.pad_token_id is None:
17
+ tokenizer.pad_token=tokenizer.eos_token
18
+ model=AutoModelForCausalLM.from_pretrained(meta['base_model'],dtype=dtype)
19
+ cfg=SmolMemoryFusionConfig.from_dict(meta['memory_fusion'])
20
+ replace_attention_layers(model,cfg,meta['accepted_layers'])
21
+ state_path=hf_hub_download(repo_id,'model.safetensors',token=token)
22
+ state=load_file(state_path,device='cpu')
23
+ model.load_state_dict(state,strict=True)
24
+ model.to(device).eval(); model.requires_grad_(False); model.config.use_cache=False
25
+ if hasattr(model,'generation_config'):
26
+ model.generation_config.use_cache=False
27
+ return model,tokenizer,meta
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:34d7f2d590d89bdc2f20fb85bf2dd962f500b85a9f1b2a2cf39b30f8dd5bf251
3
+ size 326994892
prompt_smoke_test.json ADDED
@@ -0,0 +1,61 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "status": "prompt_generation_completed",
3
+ "base_model": "HuggingFaceTB/SmolLM2-135M",
4
+ "accepted_layers": [
5
+ 0
6
+ ],
7
+ "memory_fusion_config": {
8
+ "feature_dim": 32,
9
+ "memory_rank": 64,
10
+ "dilations": [
11
+ 1,
12
+ 2,
13
+ 4,
14
+ 8,
15
+ 16,
16
+ 32,
17
+ 64,
18
+ 128
19
+ ],
20
+ "shifted_window": 8,
21
+ "train_output_projection": true
22
+ },
23
+ "last_accepted_report": {
24
+ "layer": 0,
25
+ "accepted": true,
26
+ "steps": 125,
27
+ "nmse": 0.018665021285414696,
28
+ "cosine": 0.9919254779815674,
29
+ "probe_nll": 2.9270507097244263,
30
+ "incremental_delta_nll": 0.01395869255065918,
31
+ "cumulative_delta_nll": 0.01395869255065918,
32
+ "round": 4
33
+ },
34
+ "tests": [
35
+ {
36
+ "prompt": "Austria is a country in Central Europe. The capital of Austria is",
37
+ "baseline": "Austria is a country in Central Europe. The capital of Austria is Vienna.\n\nAustria is a country in Central Europe. The capital of Austria is Vienna.\n\nAustria",
38
+ "memory_fusion": "Austria is a country in Central Europe. The capital of Austria is Vienna. The country is bordered by Germany to the north, Italy to the east, Slovenia to the south, and Hungary"
39
+ },
40
+ {
41
+ "prompt": "Question: What is the capital of Austria?\nAnswer:",
42
+ "baseline": "Question: What is the capital of Austria?\nAnswer: Vienna.\n\nQuestion: What is the capital of Austria?\nAnswer: Vienna.\n\nQuestion: What is",
43
+ "memory_fusion": "Question: What is the capital of Austria?\nAnswer: Vienna.\n\nQuestion: What is the capital of Austria?\nAnswer: Vienna.\n\nQuestion: What is"
44
+ },
45
+ {
46
+ "prompt": "Paris is the capital of France. Vienna is the capital of",
47
+ "baseline": "Paris is the capital of France. Vienna is the capital of Austria. Paris is the capital of France.\n\nThe capital of France is Paris. The capital of Austria is Vienna",
48
+ "memory_fusion": "Paris is the capital of France. Vienna is the capital of Austria.\n\nThe capital of the Czech Republic is Prague.\n\nThe capital of the United States is Washington,"
49
+ },
50
+ {
51
+ "prompt": "The largest planet in the Solar System is",
52
+ "baseline": "The largest planet in the Solar System is Jupiter. It is the largest planet in the Solar System and the largest planet in the Solar System. It is the largest",
53
+ "memory_fusion": "The largest planet in the Solar System is Jupiter, which is 11 times the mass of Earth. The largest planet in the Solar System is Saturn, which"
54
+ },
55
+ {
56
+ "prompt": "2 + 2 =",
57
+ "baseline": "2 + 2 = 10\n\nThe sum of the first 10 natural numbers is 1 + 2 + 3",
58
+ "memory_fusion": "2 + 2 = 3\n\nThe sum of the first 3 terms of the sequence is 1 + 2 + 3"
59
+ }
60
+ ]
61
+ }
requirements.txt ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ torch
2
+ transformers==4.57.6
3
+ huggingface_hub>=0.34,<2
4
+ safetensors
5
+ git+https://github.com/vtavakkoli/TinyCeNN-LM.git
run_manifest.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "run_id": "hf_memory_fusion_export-20260915T220116Z",
3
+ "created_utc": "2026-09-15T22:01:16.053571+00:00",
4
+ "repo_id": "vtava/SmolLM2-135M-MemoryFusion-Sequential-R64",
5
+ "notebook": null,
6
+ "python": "3.13.15",
7
+ "platform": "Linux-6.6.122+-x86_64-with-glibc2.39",
8
+ "reports": [
9
+ "tokenizer_config.json"
10
+ ],
11
+ "torch": "2.11.0+cu128",
12
+ "cuda_available": true,
13
+ "gpu": "NVIDIA L4"
14
+ }
sequential_progress.json ADDED
@@ -0,0 +1,69 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format_version": 1,
3
+ "stage": "accepted_layer_0",
4
+ "accepted_layers": [
5
+ 0
6
+ ],
7
+ "config": {
8
+ "feature_dim": 32,
9
+ "memory_rank": 64,
10
+ "dilations": [
11
+ 1,
12
+ 2,
13
+ 4,
14
+ 8,
15
+ 16,
16
+ 32,
17
+ 64,
18
+ 128
19
+ ],
20
+ "shifted_window": 8,
21
+ "train_output_projection": true
22
+ },
23
+ "layer_reports": [
24
+ {
25
+ "layer": 0,
26
+ "accepted": false,
27
+ "steps": 275,
28
+ "nmse": 0.048099592328071594,
29
+ "cosine": 0.9809460043907166,
30
+ "probe_nll": 2.957142174243927,
31
+ "incremental_delta_nll": 0.04405015707015991,
32
+ "cumulative_delta_nll": 0.04405015707015991,
33
+ "round": 1
34
+ },
35
+ {
36
+ "layer": 0,
37
+ "accepted": false,
38
+ "steps": 300,
39
+ "nmse": 0.02023264765739441,
40
+ "cosine": 0.991240918636322,
41
+ "probe_nll": 2.9379348754882812,
42
+ "incremental_delta_nll": 0.02484285831451416,
43
+ "cumulative_delta_nll": 0.02484285831451416,
44
+ "round": 2
45
+ },
46
+ {
47
+ "layer": 0,
48
+ "accepted": false,
49
+ "steps": 275,
50
+ "nmse": 0.01509437058120966,
51
+ "cosine": 0.9934894442558289,
52
+ "probe_nll": 2.933886468410492,
53
+ "incremental_delta_nll": 0.020794451236724854,
54
+ "cumulative_delta_nll": 0.020794451236724854,
55
+ "round": 3
56
+ },
57
+ {
58
+ "layer": 0,
59
+ "accepted": true,
60
+ "steps": 125,
61
+ "nmse": 0.018665021285414696,
62
+ "cosine": 0.9919254779815674,
63
+ "probe_nll": 2.9270507097244263,
64
+ "incremental_delta_nll": 0.01395869255065918,
65
+ "cumulative_delta_nll": 0.01395869255065918,
66
+ "round": 4
67
+ }
68
+ ]
69
+ }
sequential_run_status.json ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "status": "current_layer_needs_more_training",
3
+ "accepted_layers": [
4
+ 0
5
+ ],
6
+ "current_layer": 1,
7
+ "rounds_completed": 12,
8
+ "last_report": {
9
+ "layer": 1,
10
+ "accepted": false,
11
+ "steps": 300,
12
+ "nmse": 0.04072427377104759,
13
+ "cosine": 0.9802741408348083,
14
+ "probe_nll": 2.9506284594535828,
15
+ "incremental_delta_nll": 0.023577749729156494,
16
+ "cumulative_delta_nll": 0.03668522834777832
17
+ },
18
+ "message": "No next Transformer layer was replaced. Rerun to continue this same layer."
19
+ }
special_tokens_map.json ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|endoftext|>",
4
+ "<|im_start|>",
5
+ "<|im_end|>",
6
+ "<repo_name>",
7
+ "<reponame>",
8
+ "<file_sep>",
9
+ "<filename>",
10
+ "<gh_stars>",
11
+ "<issue_start>",
12
+ "<issue_comment>",
13
+ "<issue_closed>",
14
+ "<jupyter_start>",
15
+ "<jupyter_text>",
16
+ "<jupyter_code>",
17
+ "<jupyter_output>",
18
+ "<jupyter_script>",
19
+ "<empty_output>"
20
+ ],
21
+ "bos_token": {
22
+ "content": "<|endoftext|>",
23
+ "lstrip": false,
24
+ "normalized": false,
25
+ "rstrip": false,
26
+ "single_word": false
27
+ },
28
+ "eos_token": {
29
+ "content": "<|endoftext|>",
30
+ "lstrip": false,
31
+ "normalized": false,
32
+ "rstrip": false,
33
+ "single_word": false
34
+ },
35
+ "pad_token": "<|endoftext|>",
36
+ "unk_token": {
37
+ "content": "<|endoftext|>",
38
+ "lstrip": false,
39
+ "normalized": false,
40
+ "rstrip": false,
41
+ "single_word": false
42
+ }
43
+ }
tinycenn_model.json ADDED
@@ -0,0 +1,57 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format_version": 1,
3
+ "architecture": "smollm2-memory-fusion-sequential-accepted-prefix",
4
+ "base_model": "HuggingFaceTB/SmolLM2-135M",
5
+ "source_repository": "https://github.com/vtavakkoli/TinyCeNN-LM",
6
+ "accepted_layers": [
7
+ 0
8
+ ],
9
+ "num_hidden_layers": 30,
10
+ "memory_fusion": {
11
+ "feature_dim": 32,
12
+ "memory_rank": 64,
13
+ "dilations": [
14
+ 1,
15
+ 2,
16
+ 4,
17
+ 8,
18
+ 16,
19
+ 32,
20
+ 64,
21
+ 128
22
+ ],
23
+ "shifted_window": 8,
24
+ "train_output_projection": true
25
+ },
26
+ "last_accepted_report": {
27
+ "layer": 0,
28
+ "accepted": true,
29
+ "steps": 125,
30
+ "nmse": 0.018665021285414696,
31
+ "cosine": 0.9919254779815674,
32
+ "probe_nll": 2.9270507097244263,
33
+ "incremental_delta_nll": 0.01395869255065918,
34
+ "cumulative_delta_nll": 0.01395869255065918,
35
+ "round": 4
36
+ },
37
+ "trainer_status_at_export": {
38
+ "status": "current_layer_needs_more_training",
39
+ "accepted_layers": [
40
+ 0
41
+ ],
42
+ "current_layer": 1,
43
+ "rounds_completed": 12,
44
+ "last_report": {
45
+ "layer": 1,
46
+ "accepted": false,
47
+ "steps": 300,
48
+ "nmse": 0.04072427377104759,
49
+ "cosine": 0.9802741408348083,
50
+ "probe_nll": 2.9506284594535828,
51
+ "incremental_delta_nll": 0.023577749729156494,
52
+ "cumulative_delta_nll": 0.03668522834777832
53
+ },
54
+ "message": "No next Transformer layer was replaced. Rerun to continue this same layer."
55
+ },
56
+ "important": "Only formally accepted layers are included. Any sequential_in_progress layer is excluded."
57
+ }
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,169 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "added_tokens_decoder": {
4
+ "0": {
5
+ "content": "<|endoftext|>",
6
+ "lstrip": false,
7
+ "normalized": false,
8
+ "rstrip": false,
9
+ "single_word": false,
10
+ "special": true
11
+ },
12
+ "1": {
13
+ "content": "<|im_start|>",
14
+ "lstrip": false,
15
+ "normalized": false,
16
+ "rstrip": false,
17
+ "single_word": false,
18
+ "special": true
19
+ },
20
+ "2": {
21
+ "content": "<|im_end|>",
22
+ "lstrip": false,
23
+ "normalized": false,
24
+ "rstrip": false,
25
+ "single_word": false,
26
+ "special": true
27
+ },
28
+ "3": {
29
+ "content": "<repo_name>",
30
+ "lstrip": false,
31
+ "normalized": false,
32
+ "rstrip": false,
33
+ "single_word": false,
34
+ "special": true
35
+ },
36
+ "4": {
37
+ "content": "<reponame>",
38
+ "lstrip": false,
39
+ "normalized": false,
40
+ "rstrip": false,
41
+ "single_word": false,
42
+ "special": true
43
+ },
44
+ "5": {
45
+ "content": "<file_sep>",
46
+ "lstrip": false,
47
+ "normalized": false,
48
+ "rstrip": false,
49
+ "single_word": false,
50
+ "special": true
51
+ },
52
+ "6": {
53
+ "content": "<filename>",
54
+ "lstrip": false,
55
+ "normalized": false,
56
+ "rstrip": false,
57
+ "single_word": false,
58
+ "special": true
59
+ },
60
+ "7": {
61
+ "content": "<gh_stars>",
62
+ "lstrip": false,
63
+ "normalized": false,
64
+ "rstrip": false,
65
+ "single_word": false,
66
+ "special": true
67
+ },
68
+ "8": {
69
+ "content": "<issue_start>",
70
+ "lstrip": false,
71
+ "normalized": false,
72
+ "rstrip": false,
73
+ "single_word": false,
74
+ "special": true
75
+ },
76
+ "9": {
77
+ "content": "<issue_comment>",
78
+ "lstrip": false,
79
+ "normalized": false,
80
+ "rstrip": false,
81
+ "single_word": false,
82
+ "special": true
83
+ },
84
+ "10": {
85
+ "content": "<issue_closed>",
86
+ "lstrip": false,
87
+ "normalized": false,
88
+ "rstrip": false,
89
+ "single_word": false,
90
+ "special": true
91
+ },
92
+ "11": {
93
+ "content": "<jupyter_start>",
94
+ "lstrip": false,
95
+ "normalized": false,
96
+ "rstrip": false,
97
+ "single_word": false,
98
+ "special": true
99
+ },
100
+ "12": {
101
+ "content": "<jupyter_text>",
102
+ "lstrip": false,
103
+ "normalized": false,
104
+ "rstrip": false,
105
+ "single_word": false,
106
+ "special": true
107
+ },
108
+ "13": {
109
+ "content": "<jupyter_code>",
110
+ "lstrip": false,
111
+ "normalized": false,
112
+ "rstrip": false,
113
+ "single_word": false,
114
+ "special": true
115
+ },
116
+ "14": {
117
+ "content": "<jupyter_output>",
118
+ "lstrip": false,
119
+ "normalized": false,
120
+ "rstrip": false,
121
+ "single_word": false,
122
+ "special": true
123
+ },
124
+ "15": {
125
+ "content": "<jupyter_script>",
126
+ "lstrip": false,
127
+ "normalized": false,
128
+ "rstrip": false,
129
+ "single_word": false,
130
+ "special": true
131
+ },
132
+ "16": {
133
+ "content": "<empty_output>",
134
+ "lstrip": false,
135
+ "normalized": false,
136
+ "rstrip": false,
137
+ "single_word": false,
138
+ "special": true
139
+ }
140
+ },
141
+ "additional_special_tokens": [
142
+ "<|endoftext|>",
143
+ "<|im_start|>",
144
+ "<|im_end|>",
145
+ "<repo_name>",
146
+ "<reponame>",
147
+ "<file_sep>",
148
+ "<filename>",
149
+ "<gh_stars>",
150
+ "<issue_start>",
151
+ "<issue_comment>",
152
+ "<issue_closed>",
153
+ "<jupyter_start>",
154
+ "<jupyter_text>",
155
+ "<jupyter_code>",
156
+ "<jupyter_output>",
157
+ "<jupyter_script>",
158
+ "<empty_output>"
159
+ ],
160
+ "bos_token": "<|endoftext|>",
161
+ "clean_up_tokenization_spaces": false,
162
+ "eos_token": "<|endoftext|>",
163
+ "extra_special_tokens": {},
164
+ "model_max_length": 8192,
165
+ "pad_token": "<|endoftext|>",
166
+ "tokenizer_class": "GPT2Tokenizer",
167
+ "unk_token": "<|endoftext|>",
168
+ "vocab_size": 49152
169
+ }
vocab.json ADDED
The diff for this file is too large to render. See raw diff