File size: 4,965 Bytes
8e6b0e0
 
906ccca
 
8e6b0e0
 
 
906ccca
 
 
8e6b0e0
 
 
 
 
 
 
 
906ccca
 
 
8e6b0e0
 
 
 
e4e8ce5
 
 
 
 
 
 
 
 
 
 
8e6b0e0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e4e8ce5
 
 
 
 
 
 
 
 
8e6b0e0
 
e4e8ce5
 
 
906ccca
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
8e6b0e0
 
 
906ccca
e4e8ce5
906ccca
 
 
e4e8ce5
8e6b0e0
 
 
906ccca
 
 
 
 
8e6b0e0
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
{
 "schema_version": 1,
 "runtime_image": "sha256:7447e3f5a1f1bd9f83cb60c3d4d22cd7fa3d44195b31a53982c63b988e660793",
 "runtime_image_tag": "lottolabs/gemma4-12b-tt-p150:p3d-pc-gate-s1",
 "gemma4_tree": {
  "path": "runtime/gemma4/",
  "installed_at": "/home/container_app_user/tt-metal/models/demos/gemma4/",
  "git_branch": "sampling (src-g2)",
  "git_commit": "d6f920246c37a9b6eeed74b6bd853da683e4ff7b",
  "note": "Copied from the runtime image; equal to the git tree at that commit (diff -r); .git and __pycache__ excluded."
 },
 "tools": {
  "path": "runtime/tools/",
  "installed_at": "/home/container_app_user/gemma4-tools/"
 },
 "serving_overlay": {
  "path": "runtime/serving-overlay/",
  "installed_at": "/",
  "content": "vLLM core files (speculative method gemma4_assistant, multi-token stop handling) and the complete vllm_tt_plugin package as installed in the image: exact prefix resume (prefix_resume.py, scheduler.py), on-device sampling and speculative-sampling plumbing (model_runner.py, platform.py, worker.py, model_input.py)",
  "vllm_tt_plugin_git_commit": "d6f920246c37a9b6eeed74b6bd853da683e4ff7b",
  "note": "Copied from the runtime image; the plugin package equals the src-g2 git tree at that commit."
 },
 "ttnn_overlay": {
  "path": "runtime/ttnn-overlay/",
  "installed_at": "/home/container_app_user/tt-metal/ttnn/",
  "git_branch": "dn",
  "git_commit": "efc465f4b1a44025ae4a33e4dbc7d371b3f9266c",
  "content": "1D matmul program factory with in0-multicast chunking and the dual-NoC in1 reader (TT_MM1D_IN1_* column masks), in0/in1 dataflow reader kernels; TTNN host libraries rebuilt in layer ttnn-dn1",
  "note": "Byte-identical to the three files in the runtime image. Layer p2b built an earlier revision of the factory file (overlay master 5200a780d6feabdc387f7c106f25f43f4a85a0a8) that ttnn-dn1 overwrites."
 },
 "runtime_env": {
  "GEMMA4_VERIFY_SDPA": "batched",
  "GEMMA4_FUSE_GELU_MUL": "1",
  "GEMMA4_DECODE_KV_ROWS": "1",
  "GEMMA4_DECODE_MM_DN": "1",
  "binding": "Set in the image (layer fast2) and recorded as runtime_env in both proofs; the loader rejects any other numerics environment."
 },
 "build_chain": [
  {
   "layer": "base",
   "image": "ghcr.io/tenstorrent/tt-inference-server/vllm-tt-metal-src-release-ubuntu-22.04-amd64:0.20.0-de59f8a-03fa3af",
   "image_id": "sha256:710de20012a5417ec632e9d1db33b45f81fa9d9961da882a94868d15c5f99735"
  },
  {
   "layer": "p2b",
   "dockerfile": "runtime/Dockerfile.p2b",
   "image_id": "sha256:9fc0b344293da678ef107d570753fea90205968a9cced44ec81831f10163cb3b"
  },
  {
   "layer": "p2c",
   "dockerfile": "runtime/Dockerfile.p2c",
   "image_id": "sha256:b91c7e92ec16c507c916229ba7d675093379fe18ab05c18732eedcd3b91156f2"
  },
  {
   "layer": "ttnn-dn1",
   "dockerfile": "runtime/Dockerfile.ttnn-dn1",
   "build_context": "runtime/ttnn-overlay/ as ttnn/",
   "image_id": "sha256:2807668702d59ed5cfd914e76df10ab1f6276d3fce3e7e1d174040156ec5d8d0"
  },
  {
   "layer": "fast2",
   "dockerfile": "runtime/Dockerfile.fast2",
   "image_id": "sha256:1c23787a9da68bdbb289a190deb86647061c36c7e5fec709be8e358e269ae4b7"
  },
  {
   "layer": "p3d",
   "dockerfile": "runtime/Dockerfile.p3d",
   "image_id": "sha256:331e79b24fa2575c436158be994c6bc0e184d776d15305e82538d2e4d646f544"
  },
  {
   "layer": "p3d-pc",
   "dockerfile": "runtime/Dockerfile.p3d-pc",
   "image_id": "sha256:f3b797e6bf715e30ec116ad25098352c6754aa373617459dec5e34b2ca01c2b8",
   "content": "exact prefix resume (gemma4 tree + vllm_tt_plugin prefix_resume.py / scheduler.py / platform.py / model_runner.py)"
  },
  {
   "layer": "p3d-pc-gate",
   "dockerfile": "runtime/Dockerfile.p3d-pc-gate",
   "image_id": "sha256:b8e8d2e86c6b10440d2127697216f567380c2e185edcafe90c1e4b6e8db57327",
   "content": "confidence-gated draft length (GEMMA4_SPEC_GATE default 4096:0.85,49152:0.95)"
  },
  {
   "layer": "p3d-pc-gate-s1",
   "dockerfile": "runtime/Dockerfile.p3d-pc-gate-s1",
   "image_id": "sha256:7447e3f5a1f1bd9f83cb60c3d4d22cd7fa3d44195b31a53982c63b988e660793",
   "content": "on-device exact sampling, lossless speculative sampling, drafter-KV-row fix; complete gemma4 tree and vllm_tt_plugin package (git d6f9202)"
  }
 ],
 "build_script": "runtime/build.sh",
 "rebuild_note": "Provenance only: intermediate layers copied earlier gemma4 / plugin trees that the p3d-pc-gate-s1 layer overwrites with complete copies; the Dockerfiles reference local intermediate tags. Serve the checksum-pinned runtime-image.tar.gz.",
 "previous_runtime": {
  "tag": "lottolabs/gemma4-12b-tt-p150:p3d",
  "image_id": "sha256:331e79b24fa2575c436158be994c6bc0e184d776d15305e82538d2e4d646f544",
  "gemma4_git_commit": "061e48ddf6a0eab5dac1f958731fcb70768231d1"
 },
 "licenses": {
  "tt-metal": "runtime/licenses/tt-metal/",
  "vllm": "runtime/licenses/vllm/"
 },
 "sampling_source_patches": [
  "evidence/p3d-pc-gate-s1/sampling/source/gemma-sampling-s1.patch",
  "evidence/p3d-pc-gate-s1/sampling/source/gemma-sampling-s2.patch"
 ]
}