StarpowerTechnology commited on
Commit
e161b14
·
verified ·
1 Parent(s): 523e391

Upload 8 files

Browse files
Files changed (8) hide show
  1. README.md +35 -7
  2. app.py +149 -0
  3. model.py +149 -0
  4. model_config.json +11 -0
  5. requirements.txt +2 -0
  6. tiny_liquid_causal_lm.pt +3 -0
  7. tokenizer.json +0 -0
  8. training_summary.json +31 -0
README.md CHANGED
@@ -1,13 +1,41 @@
1
  ---
2
- title: Wvy 3m
3
- emoji: 🦀
4
- colorFrom: yellow
5
- colorTo: green
6
  sdk: gradio
7
  sdk_version: 6.26.0
8
- python_version: '3.12'
9
  app_file: app.py
10
- pinned: false
 
11
  ---
12
 
13
- Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  ---
2
+ title: WVY Tiny Liquid LM
3
+ emoji: 🌊
4
+ colorFrom: gray
5
+ colorTo: indigo
6
  sdk: gradio
7
  sdk_version: 6.26.0
8
+ python_version: "3.12"
9
  app_file: app.py
10
+ suggested_hardware: cpu-basic
11
+ short_description: Chat with a 3.31M-parameter custom liquid-style LM.
12
  ---
13
 
14
+ # WVY Tiny Liquid LM
15
+
16
+ A Hugging Face Space for the trained `TinyLiquidCausalLanguageModel` checkpoint in this repository.
17
+
18
+ ## Model
19
+
20
+ - Parameters: 3,314,880
21
+ - Vocabulary: 8,000 BPE tokens
22
+ - Width: 192
23
+ - Blocks: 4
24
+ - Feed-forward hidden width: 512
25
+ - Causal depthwise convolution kernel: 5
26
+ - Context used by the chat app: 256 tokens
27
+ - Architecture: recurrent liquid-style state mixer + SwiGLU feed-forward layers
28
+ - Final fine-tuning loss: 0.1344
29
+ - Final fine-tuning perplexity: 1.1438
30
+
31
+ The Space loads the custom PyTorch architecture directly from `model.py`, restores `tiny_liquid_causal_lm.pt`, and uses the bundled `tokenizer.json`.
32
+
33
+ ## Files
34
+
35
+ - `app.py` — Gradio chat UI and autoregressive generation
36
+ - `model.py` — exact custom model architecture used for the checkpoint
37
+ - `tiny_liquid_causal_lm.pt` — final fine-tuned checkpoint
38
+ - `tokenizer.json` — BPE tokenizer
39
+ - `model_config.json` — architecture configuration
40
+ - `training_summary.json` — training statistics
41
+ - `requirements.txt` — runtime dependencies
app.py ADDED
@@ -0,0 +1,149 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from pathlib import Path
2
+ import json
3
+
4
+ import gradio as gr
5
+ import torch
6
+ from tokenizers import Tokenizer
7
+
8
+ from model import LiquidModelConfig, TinyLiquidCausalLanguageModel
9
+
10
+ ROOT = Path(__file__).resolve().parent
11
+ DEVICE = torch.device("cuda" if torch.cuda.is_available() else "cpu")
12
+ SEQUENCE_LENGTH = 256
13
+
14
+ PAD_TOKEN = "<|pad|>"
15
+ BOS_TOKEN = "<|bos|>"
16
+ EOS_TOKEN = "<|eos|>"
17
+ USER_TOKEN = "<|user|>"
18
+ ASSISTANT_TOKEN = "<|assistant|>"
19
+
20
+ with open(ROOT / "model_config.json", "r", encoding="utf-8") as f:
21
+ CONFIG_DATA = json.load(f)
22
+
23
+ config = LiquidModelConfig(**CONFIG_DATA)
24
+ tokenizer = Tokenizer.from_file(str(ROOT / "tokenizer.json"))
25
+ PAD_ID = tokenizer.token_to_id(PAD_TOKEN)
26
+ BOS_ID = tokenizer.token_to_id(BOS_TOKEN)
27
+ EOS_ID = tokenizer.token_to_id(EOS_TOKEN)
28
+ USER_ID = tokenizer.token_to_id(USER_TOKEN)
29
+
30
+ checkpoint = torch.load(ROOT / "tiny_liquid_causal_lm.pt", map_location="cpu", weights_only=False)
31
+ model = TinyLiquidCausalLanguageModel(config)
32
+ model.load_state_dict(checkpoint["model_state_dict"], strict=True)
33
+ model.eval().to(DEVICE)
34
+
35
+ if DEVICE.type == "cpu":
36
+ torch.set_num_threads(max(1, min(4, torch.get_num_threads())))
37
+
38
+
39
+ def _history_to_lines(history):
40
+ lines = []
41
+ for item in history or []:
42
+ if isinstance(item, dict):
43
+ role = str(item.get("role", "")).lower()
44
+ content = item.get("content", "")
45
+ if isinstance(content, str) and content.strip() and role in {"user", "assistant"}:
46
+ lines.append(f"<|{role}|>\n{content.strip()}")
47
+ elif isinstance(item, (list, tuple)) and len(item) == 2:
48
+ user_message, assistant_message = item
49
+ if user_message:
50
+ lines.append(f"<|user|>\n{str(user_message).strip()}")
51
+ if assistant_message:
52
+ lines.append(f"<|assistant|>\n{str(assistant_message).strip()}")
53
+ return lines[-8:]
54
+
55
+
56
+ @torch.inference_mode()
57
+ def generate(message, history, temperature, top_k, max_new_tokens):
58
+ message = str(message).strip()
59
+ if not message:
60
+ return ""
61
+
62
+ transcript = _history_to_lines(history)
63
+ transcript.extend([f"<|user|>\n{message}", "<|assistant|>"])
64
+ prompt = "\n".join(transcript) + "\n"
65
+
66
+ generated_ids = [BOS_ID] + tokenizer.encode(prompt).ids
67
+ response_ids = []
68
+ stop_ids = {EOS_ID, USER_ID}
69
+
70
+ for _ in range(int(max_new_tokens)):
71
+ model_input = torch.tensor(
72
+ [generated_ids[-SEQUENCE_LENGTH:]],
73
+ dtype=torch.long,
74
+ device=DEVICE,
75
+ )
76
+ logits = model(model_input)["logits"][0, -1].float()
77
+ logits[PAD_ID] = -float("inf")
78
+
79
+ if float(temperature) <= 0:
80
+ next_token = int(torch.argmax(logits).item())
81
+ else:
82
+ logits = logits / float(temperature)
83
+ k = max(1, min(int(top_k), logits.numel()))
84
+ top_values, top_indices = torch.topk(logits, k=k)
85
+ probabilities = torch.softmax(top_values, dim=-1)
86
+ sampled_position = torch.multinomial(probabilities, num_samples=1)
87
+ next_token = int(top_indices[sampled_position].item())
88
+
89
+ if next_token in stop_ids:
90
+ break
91
+
92
+ generated_ids.append(next_token)
93
+ response_ids.append(next_token)
94
+
95
+ return tokenizer.decode(response_ids, skip_special_tokens=True).strip()
96
+
97
+
98
+ CSS = """
99
+ .gradio-container { max-width: 920px !important; margin: 0 auto !important; }
100
+ footer { display: none !important; }
101
+ """
102
+
103
+ with gr.Blocks(css=CSS, title="WVY Tiny Liquid LM") as demo:
104
+ gr.Markdown(
105
+ "# WVY Tiny Liquid LM\n"
106
+ "3.31M-parameter custom causal language model · 4 liquid-style state-mixer blocks · 256-token context"
107
+ )
108
+
109
+ chatbot = gr.Chatbot(height=560)
110
+
111
+ with gr.Row():
112
+ msg = gr.Textbox(
113
+ placeholder="Message WVY...",
114
+ show_label=False,
115
+ scale=8,
116
+ autofocus=True,
117
+ )
118
+ send = gr.Button("Send", variant="primary", scale=1)
119
+
120
+ with gr.Accordion("Generation settings", open=False):
121
+ temperature = gr.Slider(0.0, 1.5, value=0.8, step=0.05, label="Temperature")
122
+ top_k = gr.Slider(1, 100, value=40, step=1, label="Top-k")
123
+ max_new_tokens = gr.Slider(8, 128, value=96, step=8, label="Max new tokens")
124
+
125
+ clear = gr.Button("Clear chat")
126
+
127
+ def respond(message, history, temperature, top_k, max_new_tokens):
128
+ history = history or []
129
+ reply = generate(message, history, temperature, top_k, max_new_tokens)
130
+ history = history + [
131
+ {"role": "user", "content": message},
132
+ {"role": "assistant", "content": reply},
133
+ ]
134
+ return "", history
135
+
136
+ send.click(
137
+ respond,
138
+ [msg, chatbot, temperature, top_k, max_new_tokens],
139
+ [msg, chatbot],
140
+ )
141
+ msg.submit(
142
+ respond,
143
+ [msg, chatbot, temperature, top_k, max_new_tokens],
144
+ [msg, chatbot],
145
+ )
146
+ clear.click(lambda: ("", []), outputs=[msg, chatbot])
147
+
148
+ if __name__ == "__main__":
149
+ demo.queue(default_concurrency_limit=1).launch()
model.py ADDED
@@ -0,0 +1,149 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from dataclasses import dataclass
2
+
3
+ import torch
4
+ import torch.nn as nn
5
+ import torch.nn.functional as F
6
+
7
+
8
+ class RMSNorm(nn.Module):
9
+ def __init__(self, dimension: int, epsilon: float = 1e-6):
10
+ super().__init__()
11
+ self.weight = nn.Parameter(torch.ones(dimension))
12
+ self.epsilon = epsilon
13
+
14
+ def forward(self, hidden_states: torch.Tensor) -> torch.Tensor:
15
+ normalized = hidden_states * torch.rsqrt(
16
+ hidden_states.pow(2).mean(dim=-1, keepdim=True) + self.epsilon
17
+ )
18
+ return self.weight * normalized
19
+
20
+
21
+ class LiquidStateMixer(nn.Module):
22
+ def __init__(self, dimension: int, kernel_size: int, dropout: float):
23
+ super().__init__()
24
+ self.kernel_size = kernel_size
25
+ self.input_norm = RMSNorm(dimension)
26
+ self.causal_depthwise_convolution = nn.Conv1d(
27
+ dimension,
28
+ dimension,
29
+ kernel_size,
30
+ groups=dimension,
31
+ bias=True,
32
+ )
33
+ self.state_parameters = nn.Linear(dimension, 3 * dimension)
34
+ self.base_decay_logits = nn.Parameter(torch.zeros(dimension))
35
+ self.output_projection = nn.Linear(dimension, dimension, bias=False)
36
+ self.dropout = nn.Dropout(dropout)
37
+
38
+ def forward(self, hidden_states: torch.Tensor) -> torch.Tensor:
39
+ normalized = self.input_norm(hidden_states)
40
+ convolution_input = normalized.transpose(1, 2)
41
+ convolution_input = F.pad(convolution_input, (self.kernel_size - 1, 0))
42
+ local_features = self.causal_depthwise_convolution(convolution_input).transpose(1, 2)
43
+
44
+ candidate, decay_logits, output_gate = self.state_parameters(local_features).chunk(3, dim=-1)
45
+ candidate = torch.tanh(candidate)
46
+ decay = torch.sigmoid(decay_logits + self.base_decay_logits)
47
+ output_gate = torch.sigmoid(output_gate)
48
+
49
+ state = torch.zeros_like(candidate[:, 0])
50
+ mixed_steps = []
51
+ for step in range(candidate.size(1)):
52
+ step_decay = decay[:, step]
53
+ state = step_decay * state + (1.0 - step_decay) * candidate[:, step]
54
+ mixed_steps.append(output_gate[:, step] * state)
55
+
56
+ mixed = torch.stack(mixed_steps, dim=1)
57
+ return hidden_states + self.dropout(self.output_projection(mixed))
58
+
59
+
60
+ class SwiGLUFeedForward(nn.Module):
61
+ def __init__(self, dimension: int, hidden_dimension: int, dropout: float):
62
+ super().__init__()
63
+ self.input_norm = RMSNorm(dimension)
64
+ self.gate_projection = nn.Linear(dimension, hidden_dimension, bias=False)
65
+ self.value_projection = nn.Linear(dimension, hidden_dimension, bias=False)
66
+ self.output_projection = nn.Linear(hidden_dimension, dimension, bias=False)
67
+ self.dropout = nn.Dropout(dropout)
68
+
69
+ def forward(self, hidden_states: torch.Tensor) -> torch.Tensor:
70
+ normalized = self.input_norm(hidden_states)
71
+ activated = F.silu(self.gate_projection(normalized)) * self.value_projection(normalized)
72
+ return hidden_states + self.dropout(self.output_projection(activated))
73
+
74
+
75
+ class LiquidBlock(nn.Module):
76
+ def __init__(
77
+ self,
78
+ dimension: int,
79
+ hidden_dimension: int,
80
+ kernel_size: int,
81
+ dropout: float,
82
+ ):
83
+ super().__init__()
84
+ self.state_mixer = LiquidStateMixer(dimension, kernel_size, dropout)
85
+ self.feed_forward = SwiGLUFeedForward(dimension, hidden_dimension, dropout)
86
+
87
+ def forward(self, hidden_states: torch.Tensor) -> torch.Tensor:
88
+ return self.feed_forward(self.state_mixer(hidden_states))
89
+
90
+
91
+ @dataclass
92
+ class LiquidModelConfig:
93
+ vocab_size: int
94
+ dimension: int = 192
95
+ layer_count: int = 4
96
+ feed_forward_hidden_dimension: int = 512
97
+ convolution_kernel_size: int = 5
98
+ dropout: float = 0.05
99
+ pad_token_id: int = 0
100
+ bos_token_id: int = 2
101
+ eos_token_id: int = 3
102
+
103
+
104
+ class TinyLiquidCausalLanguageModel(nn.Module):
105
+ def __init__(self, config: LiquidModelConfig):
106
+ super().__init__()
107
+ self.config = config
108
+ self.token_embedding = nn.Embedding(
109
+ config.vocab_size,
110
+ config.dimension,
111
+ padding_idx=config.pad_token_id,
112
+ )
113
+ self.blocks = nn.ModuleList(
114
+ [
115
+ LiquidBlock(
116
+ config.dimension,
117
+ config.feed_forward_hidden_dimension,
118
+ config.convolution_kernel_size,
119
+ config.dropout,
120
+ )
121
+ for _ in range(config.layer_count)
122
+ ]
123
+ )
124
+ self.final_norm = RMSNorm(config.dimension)
125
+ self.language_model_head = nn.Linear(config.dimension, config.vocab_size, bias=False)
126
+ self.language_model_head.weight = self.token_embedding.weight
127
+ self.apply(self._initialize_weights)
128
+
129
+ @staticmethod
130
+ def _initialize_weights(module: nn.Module) -> None:
131
+ if isinstance(module, (nn.Linear, nn.Embedding)):
132
+ nn.init.normal_(module.weight, mean=0.0, std=0.02)
133
+ if isinstance(module, nn.Linear) and module.bias is not None:
134
+ nn.init.zeros_(module.bias)
135
+
136
+ def forward(self, input_ids: torch.Tensor, labels=None):
137
+ hidden_states = self.token_embedding(input_ids)
138
+ for block in self.blocks:
139
+ hidden_states = block(hidden_states)
140
+ logits = self.language_model_head(self.final_norm(hidden_states))
141
+
142
+ loss = None
143
+ if labels is not None:
144
+ loss = F.cross_entropy(
145
+ logits.reshape(-1, logits.size(-1)),
146
+ labels.reshape(-1),
147
+ ignore_index=-100,
148
+ )
149
+ return {"loss": loss, "logits": logits}
model_config.json ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "vocab_size": 8000,
3
+ "dimension": 192,
4
+ "layer_count": 4,
5
+ "feed_forward_hidden_dimension": 512,
6
+ "convolution_kernel_size": 5,
7
+ "dropout": 0.05,
8
+ "pad_token_id": 0,
9
+ "bos_token_id": 2,
10
+ "eos_token_id": 3
11
+ }
requirements.txt ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ torch>=2.2
2
+ tokenizers>=0.20
tiny_liquid_causal_lm.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b3dbbc1dd613e14bb0ffb195dd0625f39f1d51f3c31cc805dfe7a1b3b7461800
3
+ size 13279025
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
training_summary.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset": {
3
+ "files": 17,
4
+ "pretraining_documents": 28585,
5
+ "pretraining_tokens": 2856322,
6
+ "sft_examples": 24536
7
+ },
8
+ "parameters": {
9
+ "total": 3314880,
10
+ "trainable": 3314880
11
+ },
12
+ "pretraining_history": [
13
+ {
14
+ "epoch": 1,
15
+ "loss": 1.9338438765763382,
16
+ "perplexity": 6.916043631489244
17
+ }
18
+ ],
19
+ "fine_tuning_history": [
20
+ {
21
+ "epoch": 1,
22
+ "loss": 0.23482260895200027,
23
+ "perplexity": 1.2646844051414168
24
+ },
25
+ {
26
+ "epoch": 2,
27
+ "loss": 0.134362090434873,
28
+ "perplexity": 1.143806906211811
29
+ }
30
+ ]
31
+ }