ProCreations commited on
Commit
00bd945
·
verified ·
1 Parent(s): 0ecf1af

Pin plain NVFP4 build, source, and bounded validation plan

Browse files
README.md ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ base_model: Agnes-AI/Agnes-3.0-Flash
4
+ base_model_relation: quantized
5
+ pipeline_tag: image-text-to-text
6
+ tags:
7
+ - nvfp4
8
+ - modelopt
9
+ - quantized
10
+ ---
11
+ # Agnes-3.0-Flash Preview — plain NVFP4
12
+
13
+ Private build in progress. The pinned source is the 33B Preview checkpoint with 262144 context. Native evaluation and public release are pending.
reproducibility/bootstrap_build.sh ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env bash
2
+ set -euo pipefail
3
+ mkdir -p /workspace/agnes/code
4
+ python -m pip install --break-system-packages --no-cache-dir uv
5
+ uv venv --system-site-packages /workspace/agnes/venv
6
+ export PATH="/workspace/agnes/venv/bin:$PATH"
7
+ uv pip install --python /workspace/agnes/venv/bin/python \
8
+ 'torch==2.13.0' 'transformers==5.14.0' 'accelerate==1.12.0' \
9
+ 'huggingface_hub==1.18.0' 'pillow>=12' 'sentencepiece>=0.2' 'einops' \
10
+ 'nvidia-modelopt @ https://github.com/NVIDIA/Model-Optimizer/archive/5cae3940402f1ced98069a666b0bec72ec8b33b5.tar.gz'
11
+ python - <<'PY'
12
+ import os
13
+ from huggingface_hub import hf_hub_download
14
+ hf_hub_download('ProCreations/Agnes-3.0-Flash-NVFP4','reproducibility/build.py',
15
+ revision=os.environ['AGNES_CODE_REVISION'],local_dir='/workspace/agnes/code')
16
+ PY
17
+ python -u /workspace/agnes/code/reproducibility/build.py
reproducibility/build.py ADDED
@@ -0,0 +1,208 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Plain data-free NVFP4; deterministic source-supported FFN folding."""
2
+ import collections
3
+ import copy
4
+ import gc
5
+ import hashlib
6
+ import importlib.metadata
7
+ import json
8
+ import os
9
+ import re
10
+ import shutil
11
+ import time
12
+ from pathlib import Path
13
+
14
+ import torch
15
+ from huggingface_hub import HfApi, snapshot_download
16
+ from safetensors import safe_open
17
+ from safetensors.torch import save_file
18
+ from transformers import AutoModelForImageTextToText
19
+ import modelopt.torch.quantization as mtq
20
+ from modelopt.torch.quantization.nn import TensorQuantizer
21
+ from modelopt.torch.export import export_hf_checkpoint
22
+
23
+ SOURCE = 'Agnes-AI/Agnes-3.0-Flash'
24
+ REV = '24f712ce59379b54c4a141d2708c35daf5ff613b'
25
+ TARGET = 'ProCreations/Agnes-3.0-Flash-NVFP4'
26
+ ROOT = Path('/workspace/agnes')
27
+ OUTPUT = ROOT/'export'
28
+ API = HfApi()
29
+ SELECT = re.compile(r'^model\.language_model\.layers\.\d+\.(?:mlp\.(?:gate_proj|up_proj|down_proj)|global_attn\.(?:q_proj|k_proj|v_proj|o_proj))\.weight$')
30
+ REPORT = dict(source=SOURCE, source_revision=REV, target=TARGET,
31
+ status='building', code_revision=os.environ['AGNES_CODE_REVISION'],
32
+ modelopt_revision='5cae3940402f1ced98069a666b0bec72ec8b33b5',
33
+ quantization=dict(method='plain NVFP4 max', training=False, calibration_examples=0,
34
+ calibration_tokens=0, calibration_forward_calls=0, activation_global_scale=1.0,
35
+ block_size=16, activation_block_scales='dynamic E4M3', kv_cache_quantized=False),
36
+ folding=dict(main_width=17408, parallel_width=2048, folded_width=19456,
37
+ gate_up_dimension=0, down_dimension=1, layers=72,
38
+ reason='Match the upstream SGLang BF16 loader before quantization'),
39
+ evaluation=dict(status='pending'))
40
+
41
+ def record():
42
+ (ROOT/'quality_report.json').write_text(json.dumps(REPORT, indent=2)+'\n')
43
+ API.upload_file(repo_id=TARGET, path_or_fileobj=str(ROOT/'quality_report.json'),
44
+ path_in_repo='quality_report.json', commit_message='Record plain NVFP4 build checks')
45
+
46
+ def selected(name):
47
+ return bool(SELECT.fullmatch(name))
48
+
49
+ def main():
50
+ assert API.model_info(TARGET).private
51
+ assert not any(n.endswith('.safetensors') for n in API.list_repo_files(TARGET))
52
+ ROOT.mkdir(parents=True, exist_ok=True)
53
+ torch.set_num_threads(8)
54
+ torch.manual_seed(20260912)
55
+ REPORT['versions'] = {p:importlib.metadata.version(p) for p in ['torch','transformers','nvidia-modelopt','accelerate']}
56
+ REPORT['hardware'] = [torch.cuda.get_device_name(0)]
57
+ record()
58
+ source = Path(snapshot_download(SOURCE, revision=REV, max_workers=8))
59
+ src_index = json.loads((source/'model.safetensors.index.json').read_text())
60
+ src_map = src_index['weight_map']
61
+ model = AutoModelForImageTextToText.from_pretrained(source, trust_remote_code=True,
62
+ dtype=torch.bfloat16, device_map={'':0}, attn_implementation='sdpa').eval()
63
+ assert all(p.device.type == 'cuda' for p in model.parameters())
64
+ assert len(model.model.language_model.layers) == 72
65
+ with torch.no_grad():
66
+ for layer in model.model.language_model.layers:
67
+ mlp = layer.mlp
68
+ assert mlp.parallel_ffn is not None
69
+ for name in ['gate_proj','up_proj','down_proj']:
70
+ original, branch = getattr(mlp,name), getattr(mlp.parallel_ffn,name)
71
+ dim = 1 if name == 'down_proj' else 0
72
+ merged = torch.cat([original.weight,branch.weight], dim=dim)
73
+ # Check both slices before discarding the separate branches.
74
+ a,b = merged.split([original.weight.shape[dim],branch.weight.shape[dim]],dim=dim)
75
+ assert torch.equal(a,original.weight) and torch.equal(b,branch.weight)
76
+ replacement = torch.nn.Linear(merged.shape[1],merged.shape[0],bias=False,
77
+ device=merged.device,dtype=merged.dtype)
78
+ replacement.weight = torch.nn.Parameter(merged,requires_grad=False)
79
+ setattr(mlp,name,replacement)
80
+ mlp.parallel_ffn = None
81
+ text = model.config.text_config
82
+ text.agnes_original_intermediate_size = 17408
83
+ text.agnes_original_parallel_ffn_intermediate_size = 2048
84
+ text.intermediate_size = 19456
85
+ text.parallel_ffn_intermediate_size = 0
86
+ gc.collect(); torch.cuda.empty_cache()
87
+ preset = copy.deepcopy(mtq.NVFP4_DEFAULT_CFG)
88
+ weight = next(x['cfg'] for x in preset['quant_cfg'] if x.get('quantizer_name')=='*weight_quantizer')
89
+ activation = next(x['cfg'] for x in preset['quant_cfg'] if x.get('quantizer_name')=='*input_quantizer')
90
+ activation['constant_amax'] = 2688.0
91
+ rules = [dict(quantizer_name='*',enable=False)]
92
+ targets = [name for name,module in model.named_modules()
93
+ if isinstance(module,torch.nn.Linear) and selected(name+'.weight')]
94
+ assert len(targets) == 288, len(targets)
95
+ for name in targets:
96
+ rules += [dict(quantizer_name=name+'.weight_quantizer',cfg=weight),
97
+ dict(quantizer_name=name+'.input_quantizer',cfg=activation)]
98
+ cfg = dict(quant_cfg=rules,algorithm=dict(method='max',layerwise=dict(enable=False),
99
+ skip_forward_without_activation_calib=True))
100
+ REPORT['quantization']['configuration'] = cfg
101
+ def forbid(*args):
102
+ REPORT['quantization']['calibration_forward_calls'] += 1
103
+ raise RuntimeError('Data-free quantization may not execute calibration forwards')
104
+ hook = model.register_forward_pre_hook(forbid)
105
+ started = time.monotonic()
106
+ with torch.inference_mode():
107
+ mtq.quantize(model,cfg,forward_loop=None)
108
+ enabled = [(name,m) for name,m in model.named_modules()
109
+ if isinstance(m,TensorQuantizer) and m.is_enabled]
110
+ assert len(enabled) == 576, len(enabled)
111
+ assert all(name.rsplit('.',1)[0] in targets for name,m in enabled)
112
+ assert all(getattr(m,'_amax',None) is not None for name,m in enabled if name.endswith('weight_quantizer'))
113
+ REPORT['quantization']['weight_statistics_seconds'] = time.monotonic()-started
114
+ hook.remove()
115
+ REPORT['export_graph_inspection'] = dict(dummy_forward_calls=0,input_tokens=0,quantizers_disabled=True)
116
+ def inspect_probe(module,args):
117
+ assert not any(isinstance(m,TensorQuantizer) and (m.is_enabled or m._if_calib) for m in module.modules())
118
+ assert args[0].shape == (1,2) and torch.equal(args[0],torch.ones_like(args[0]))
119
+ REPORT['export_graph_inspection']['dummy_forward_calls'] += 1
120
+ REPORT['export_graph_inspection']['input_tokens'] += args[0].numel()
121
+ hook = model.register_forward_pre_hook(inspect_probe)
122
+ export_hf_checkpoint(model,dtype=torch.bfloat16,export_dir=OUTPUT,max_shard_size='5GB')
123
+ hook.remove()
124
+ del model,enabled
125
+ gc.collect(); torch.cuda.empty_cache()
126
+ # The upstream Transformers class ignores MTP: retain every original MTP
127
+ # tensor separately, without advertising untested speculative decoding.
128
+ dst_index = json.loads((OUTPUT/'model.safetensors.index.json').read_text())
129
+ mtp = {}
130
+ for name,filename in src_map.items():
131
+ if name.startswith('mtp.'):
132
+ with safe_open(source/filename,framework='pt',device='cpu') as f:
133
+ mtp[name] = f.get_tensor(name).clone()
134
+ assert mtp and not any(name in dst_index['weight_map'] for name in mtp)
135
+ save_file(mtp,str(OUTPUT/'model-mtp.safetensors'),metadata={'format':'pt'})
136
+ dst_index['weight_map'].update({name:'model-mtp.safetensors' for name in mtp})
137
+ dst_index['metadata']['total_size'] += sum(t.numel()*t.element_size() for t in mtp.values())
138
+ (OUTPUT/'model.safetensors.index.json').write_text(json.dumps(dst_index,indent=2)+'\n')
139
+ for file in source.rglob('*'):
140
+ rel=file.relative_to(source)
141
+ if file.is_file() and '.cache' not in rel.parts and not file.name.endswith('.safetensors') and file.name not in ['config.json','model.safetensors.index.json','README.md','.gitattributes']:
142
+ (OUTPUT/rel).parent.mkdir(parents=True,exist_ok=True)
143
+ shutil.copy2(file,OUTPUT/rel)
144
+ # The upstream loader otherwise skips renaming when the branch width is 0.
145
+ # This checkpoint has already folded its branches before quantization.
146
+ for file in OUTPUT.glob('sglang_patch/*/sglang/srt/models/qwen3_5.py'):
147
+ contents=file.read_text()
148
+ before=' if width <= 0:\n yield from weights\n return\n'
149
+ after=' if width <= 0:\n for name, weight in weights:\n yield name.replace(".delta_attn.", ".linear_attn.").replace(".global_attn.", ".self_attn."), weight\n return\n'
150
+ assert contents.count(before)==1
151
+ file.write_text(contents.replace(before,after))
152
+ # Both Transformers names and translated serving names must be excluded.
153
+ for filename in ['config.json','hf_quant_config.json']:
154
+ file=OUTPUT/filename
155
+ obj=json.loads(file.read_text())
156
+ quant=obj['quantization_config'] if filename=='config.json' else obj['quantization']
157
+ excludes=list(quant.get('exclude_modules',[]))
158
+ excludes += ['lm_head','model.visual*','mtp*']
159
+ for i in range(72):
160
+ excludes += [f'model.language_model.layers.{i}.delta_attn*',
161
+ f'model.language_model.layers.{i}.linear_attn*']
162
+ quant['exclude_modules']=sorted(set(excludes))
163
+ file.write_text(json.dumps(obj,indent=2)+'\n')
164
+ preserved=0; preserved_count=0
165
+ for filename in sorted(set(src_map.values())):
166
+ with safe_open(source/filename,framework='pt',device='cpu') as sf:
167
+ for name in sf.keys():
168
+ if selected(name) or '.mlp.parallel_ffn.' in name: continue
169
+ assert name in dst_index['weight_map'],name
170
+ with safe_open(OUTPUT/dst_index['weight_map'][name],framework='pt',device='cpu') as df:
171
+ a,b=sf.get_tensor(name),df.get_tensor(name)
172
+ assert a.dtype==b.dtype and torch.equal(a,b),name
173
+ preserved += a.numel()*a.element_size(); preserved_count+=1
174
+ counts=collections.Counter(); packed=0; inputs=0
175
+ for filename in sorted(set(dst_index['weight_map'].values())):
176
+ with safe_open(OUTPUT/filename,framework='pt',device='cpu') as f:
177
+ for name in f.keys():
178
+ tensor=f.get_tensor(name)
179
+ counts[str(tensor.dtype)]+=tensor.numel()*tensor.element_size()
180
+ if selected(name):
181
+ assert tensor.dtype==torch.uint8,name
182
+ packed+=1
183
+ if 'scale' in name:
184
+ assert torch.isfinite(tensor.float()).all() and (tensor.float()>0).all(),name
185
+ if name.endswith('input_scale'):
186
+ assert torch.equal(tensor,torch.ones_like(tensor)),name
187
+ inputs+=1
188
+ assert packed==len(targets)==288 and inputs==288,(packed,inputs)
189
+ assert REPORT['quantization']['calibration_forward_calls']==0
190
+ REPORT['export']=dict(source_tensor_bytes=src_index['metadata']['total_size'],
191
+ exported_weight_bytes=sum((OUTPUT/f).stat().st_size for f in set(dst_index['weight_map'].values())),
192
+ tensor_bytes_by_dtype=dict(counts),packed_linear_weights=packed,
193
+ bf16_preserved_bytes=preserved,bf16_preserved_tensors=preserved_count,
194
+ all_unquantized_tensors_bitwise_equal=True,mtp_tensors_preserved=len(mtp),
195
+ activation_global_scales_exactly_one=inputs)
196
+ REPORT['status']='packed_export_verified_pending_native_evaluation'
197
+ record()
198
+ shutil.copy2(ROOT/'quality_report.json',OUTPUT/'quality_report.json')
199
+ API.upload_folder(repo_id=TARGET,folder_path=OUTPUT,
200
+ commit_message='Upload verified plain NVFP4 Agnes Preview checkpoint')
201
+ print('AGNES_BUILD_COMPLETE '+json.dumps(REPORT['export']),flush=True)
202
+
203
+ if __name__=='__main__':
204
+ try: main()
205
+ except Exception:
206
+ REPORT['status']='build_failed'
207
+ record()
208
+ raise
reproducibility/dispatch_build.py ADDED
@@ -0,0 +1,52 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """One data-free build; hard timeout and a separate bounded Agnes ledger."""
2
+ import json
3
+ from pathlib import Path
4
+ from huggingface_hub import HfApi, CommitOperationAdd, get_token
5
+
6
+ ROOT=Path(__file__).resolve().parent
7
+ api=HfApi()
8
+ repo='ProCreations/Agnes-3.0-Flash-NVFP4'
9
+ assert api.whoami()['name']=='ProCreations'
10
+ ledger=ROOT/'jobs.jsonl'
11
+ assert not ledger.exists(),'Inspect previous attempts before dispatching again'
12
+ api.create_repo(repo_id=repo,private=True,exist_ok=True)
13
+ assert api.model_info(repo).private
14
+ assert not any(n.endswith('.safetensors') for n in api.list_repo_files(repo))
15
+ plan=dict(source='Agnes-AI/Agnes-3.0-Flash',source_revision='24f712ce59379b54c4a141d2708c35daf5ff613b',
16
+ target=repo,method='plain data-free NVFP4',training=False,calibration_examples=0,
17
+ quantized='All decoder MLP and global attention linear weights; branches folded as in upstream SGLang',
18
+ preserved='Recurrent attention, vision, embeddings, output head, norms, MTP',
19
+ agnes_compute_ceiling_usd=40,prior_nex_compute_upper_bound_usd=98.303050825,
20
+ combined_user_cap_usd=200,publication='Public after native validation')
21
+ (ROOT/'plan.json').write_text(json.dumps(plan,indent=2)+'\n')
22
+ readme='''---
23
+ license: apache-2.0
24
+ base_model: Agnes-AI/Agnes-3.0-Flash
25
+ base_model_relation: quantized
26
+ pipeline_tag: image-text-to-text
27
+ tags:
28
+ - nvfp4
29
+ - modelopt
30
+ - quantized
31
+ ---
32
+ # Agnes-3.0-Flash Preview — plain NVFP4
33
+
34
+ Private build in progress. The pinned source is the 33B Preview checkpoint with 262144 context. Native evaluation and public release are pending.
35
+ '''
36
+ (ROOT/'README.pending.md').write_text(readme)
37
+ names=['build.py','bootstrap_build.sh','dispatch_build.py','plan.json','source_meta.json','runtime_image.json']
38
+ commit=api.create_commit(repo_id=repo,operations=[CommitOperationAdd(path_in_repo='reproducibility/'+n,path_or_fileobj=str(ROOT/n)) for n in names]+[
39
+ CommitOperationAdd(path_in_repo='README.md',path_or_fileobj=str(ROOT/'README.pending.md'))],
40
+ commit_message='Pin plain NVFP4 build, source, and bounded validation plan')
41
+ rate=next(h.unit_cost_usd*60 for h in api.list_jobs_hardware() if h.name=='h200')
42
+ job=api.run_job(image='pytorch/pytorch:2.13.0-cuda13.0-cudnn9-devel',flavor='h200',timeout=5400,
43
+ command=['bash','-c',(ROOT/'bootstrap_build.sh').read_text()],
44
+ env={'AGNES_CODE_REVISION':commit.oid,'HF_HOME':'/workspace/agnes/hf-cache',
45
+ 'PYTHONUNBUFFERED':'1','TOKENIZERS_PARALLELISM':'false','OMP_NUM_THREADS':'8',
46
+ 'HF_HUB_DOWNLOAD_TIMEOUT':'180','TQDM_MININTERVAL':'30',
47
+ 'PYTORCH_CUDA_ALLOC_CONF':'expandable_segments:True'},
48
+ secrets={'HF_TOKEN':get_token()},labels={'project':'agnes-flash-nvfp4','phase':'plain-build'})
49
+ row=dict(job_id=job.id,url=job.url,phase='plain-build',flavor='h200',rate_usd_hour=rate,
50
+ timeout_hours=1.5,maximum_cost_usd=rate*1.5,code_revision=commit.oid)
51
+ ledger.write_text(json.dumps(row)+'\n')
52
+ print(json.dumps(row))
reproducibility/plan.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "source": "Agnes-AI/Agnes-3.0-Flash",
3
+ "source_revision": "24f712ce59379b54c4a141d2708c35daf5ff613b",
4
+ "target": "ProCreations/Agnes-3.0-Flash-NVFP4",
5
+ "method": "plain data-free NVFP4",
6
+ "training": false,
7
+ "calibration_examples": 0,
8
+ "quantized": "All decoder MLP and global attention linear weights; branches folded as in upstream SGLang",
9
+ "preserved": "Recurrent attention, vision, embeddings, output head, norms, MTP",
10
+ "agnes_compute_ceiling_usd": 40,
11
+ "prior_nex_compute_upper_bound_usd": 98.303050825,
12
+ "combined_user_cap_usd": 200,
13
+ "publication": "Public after native validation"
14
+ }
reproducibility/runtime_image.json ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "image": "lmsysorg/sglang",
3
+ "tag": "nightly-dev-20260908-20ca564b",
4
+ "digest": "sha256:9a352a35c973a2357372e85f3bcb5388b6b3c46c1329165987260f3b089647dc",
5
+ "manifest": {
6
+ "schemaVersion": 2,
7
+ "mediaType": "application/vnd.oci.image.index.v1+json",
8
+ "manifests": [
9
+ {
10
+ "mediaType": "application/vnd.oci.image.manifest.v1+json",
11
+ "digest": "sha256:2d53500cb72184020c59fe16aab3512d53b1aa8ee6195bfb2aaf5855d58099df",
12
+ "size": 13687,
13
+ "platform": {
14
+ "architecture": "amd64",
15
+ "os": "linux"
16
+ }
17
+ },
18
+ {
19
+ "mediaType": "application/vnd.oci.image.manifest.v1+json",
20
+ "digest": "sha256:b6792a80cd34d743b2e3be7f088dfb701328a900f64e8008b16279b616f6db46",
21
+ "size": 838,
22
+ "annotations": {
23
+ "vnd.docker.reference.digest": "sha256:2d53500cb72184020c59fe16aab3512d53b1aa8ee6195bfb2aaf5855d58099df",
24
+ "vnd.docker.reference.type": "attestation-manifest"
25
+ },
26
+ "platform": {
27
+ "architecture": "unknown",
28
+ "os": "unknown"
29
+ }
30
+ },
31
+ {
32
+ "mediaType": "application/vnd.oci.image.manifest.v1+json",
33
+ "digest": "sha256:c5697af9d2b89629b8d392509de6085a2c16b492b76acc975660417d86f27491",
34
+ "size": 13673,
35
+ "platform": {
36
+ "architecture": "arm64",
37
+ "os": "linux"
38
+ }
39
+ },
40
+ {
41
+ "mediaType": "application/vnd.oci.image.manifest.v1+json",
42
+ "digest": "sha256:b1b088a5e7a06a4685d684ec7e109e159ddb722fd89dc476e4c0987c585dae7e",
43
+ "size": 838,
44
+ "annotations": {
45
+ "vnd.docker.reference.digest": "sha256:c5697af9d2b89629b8d392509de6085a2c16b492b76acc975660417d86f27491",
46
+ "vnd.docker.reference.type": "attestation-manifest"
47
+ },
48
+ "platform": {
49
+ "architecture": "unknown",
50
+ "os": "unknown"
51
+ }
52
+ }
53
+ ]
54
+ }
55
+ }
reproducibility/source_meta.json ADDED
@@ -0,0 +1,279 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "repo": "Agnes-AI/Agnes-3.0-Flash",
3
+ "revision": "24f712ce59379b54c4a141d2708c35daf5ff613b",
4
+ "private": false,
5
+ "gated": false,
6
+ "card": {
7
+ "language": [
8
+ "en",
9
+ "zh"
10
+ ],
11
+ "library_name": "transformers",
12
+ "license": "apache-2.0",
13
+ "pipeline_tag": "image-text-to-text",
14
+ "tags": [
15
+ "agnes-ai",
16
+ "reasoning",
17
+ "multimodal",
18
+ "long-context",
19
+ "hybrid-attention"
20
+ ]
21
+ },
22
+ "files": [
23
+ {
24
+ "name": ".gitattributes",
25
+ "size": 1519,
26
+ "sha256": null
27
+ },
28
+ {
29
+ "name": "LICENSE",
30
+ "size": 11357,
31
+ "sha256": null
32
+ },
33
+ {
34
+ "name": "README.md",
35
+ "size": 16314,
36
+ "sha256": null
37
+ },
38
+ {
39
+ "name": "README_zh.md",
40
+ "size": 14714,
41
+ "sha256": null
42
+ },
43
+ {
44
+ "name": "assets/agnes_benchmarks.svg",
45
+ "size": 12421,
46
+ "sha256": null
47
+ },
48
+ {
49
+ "name": "assets/agnes_logo.svg",
50
+ "size": 2435,
51
+ "sha256": null
52
+ },
53
+ {
54
+ "name": "chat_template.jinja",
55
+ "size": 8952,
56
+ "sha256": null
57
+ },
58
+ {
59
+ "name": "config.json",
60
+ "size": 4505,
61
+ "sha256": null
62
+ },
63
+ {
64
+ "name": "configuration_agnes.py",
65
+ "size": 7516,
66
+ "sha256": null
67
+ },
68
+ {
69
+ "name": "generation_config.json",
70
+ "size": 202,
71
+ "sha256": null
72
+ },
73
+ {
74
+ "name": "image_processing_agnes.py",
75
+ "size": 8700,
76
+ "sha256": null
77
+ },
78
+ {
79
+ "name": "merges.txt",
80
+ "size": 3353259,
81
+ "sha256": null
82
+ },
83
+ {
84
+ "name": "model-00001-of-00019.safetensors",
85
+ "size": 3966730528,
86
+ "sha256": "5697cc0f8b4223adab6e8fb6043727752c728e485a3621b693381073409e3070"
87
+ },
88
+ {
89
+ "name": "model-00002-of-00019.safetensors",
90
+ "size": 3043080320,
91
+ "sha256": "e3d8adc00d4f6568b257ccdbcfdedde894024eb2a10abd125426fe8440fc563a"
92
+ },
93
+ {
94
+ "name": "model-00003-of-00019.safetensors",
95
+ "size": 2542796952,
96
+ "sha256": "2e1bf62cbcd406eaa64b60d10353e1f0ef4039d0976e56f05cabe953454f9968"
97
+ },
98
+ {
99
+ "name": "model-00004-of-00019.safetensors",
100
+ "size": 3988973120,
101
+ "sha256": "2962bc38c2b69709d6df78cffe9692ab4e761d47ba2f2c0e61c808484a2df825"
102
+ },
103
+ {
104
+ "name": "model-00005-of-00019.safetensors",
105
+ "size": 2099339848,
106
+ "sha256": "ecb1d015bca88250dbc9e167479c5e740de21b5a70fc38cc84b6c3388db9d3a9"
107
+ },
108
+ {
109
+ "name": "model-00006-of-00019.safetensors",
110
+ "size": 3979553664,
111
+ "sha256": "ae9c2e979d10655fded3bbdbfddfb2f72b3d1ac5195c9e214756f694196fd523"
112
+ },
113
+ {
114
+ "name": "model-00007-of-00019.safetensors",
115
+ "size": 2108759344,
116
+ "sha256": "dfdb74b2182b870b602127a64e3b9995f138bfe707168b7339c759623996e2d7"
117
+ },
118
+ {
119
+ "name": "model-00008-of-00019.safetensors",
120
+ "size": 3979553664,
121
+ "sha256": "24db669382d366c90d6c9fd2ce3cfceee836c3ec8a8a3ae62313ce8f3cd55862"
122
+ },
123
+ {
124
+ "name": "model-00009-of-00019.safetensors",
125
+ "size": 2108759344,
126
+ "sha256": "fe9de765012eb1c7bd2dd30f5b2b7579b5eef4de67805023b2d86c2bea8d2511"
127
+ },
128
+ {
129
+ "name": "model-00010-of-00019.safetensors",
130
+ "size": 3979553664,
131
+ "sha256": "a1819ace7b218bdad22a194ed9f09e4d436ba8ff522bf1e09b070d6d2bb8781a"
132
+ },
133
+ {
134
+ "name": "model-00011-of-00019.safetensors",
135
+ "size": 2108759344,
136
+ "sha256": "4a7b4caaedf55e0b4d4bad1524b0f433b5e86ce4a7ab13950c3749dc10b56f5f"
137
+ },
138
+ {
139
+ "name": "model-00012-of-00019.safetensors",
140
+ "size": 3979553664,
141
+ "sha256": "1f937940e0dcd5dc28ab4117fc0437c53e10328b5b5ff52b126bba01da208d7d"
142
+ },
143
+ {
144
+ "name": "model-00013-of-00019.safetensors",
145
+ "size": 2108759344,
146
+ "sha256": "ee54d2d219596d6d8900aaac24a4afd59028b883d57d5eba77ceb8d207c56cb3"
147
+ },
148
+ {
149
+ "name": "model-00014-of-00019.safetensors",
150
+ "size": 3979553664,
151
+ "sha256": "064a0cfd148b1ca7188c83a40a5b30ac5b26e4f3a3982e50948945cfa6872cae"
152
+ },
153
+ {
154
+ "name": "model-00015-of-00019.safetensors",
155
+ "size": 2108759344,
156
+ "sha256": "d395947e2ef2274f151e1b83e8357befa9f40f0bca296d99d1aa69abb703cf51"
157
+ },
158
+ {
159
+ "name": "model-00016-of-00019.safetensors",
160
+ "size": 3979564008,
161
+ "sha256": "4eda4a1568a1a05fc762e7a3a2072c9d11cb860cdd763ac1720aef39f6a24751"
162
+ },
163
+ {
164
+ "name": "model-00017-of-00019.safetensors",
165
+ "size": 2108759344,
166
+ "sha256": "6cdc6d3b849808dfec6f0b7d11a09b74c0c91ba8df852891741e9eb63d33ab9f"
167
+ },
168
+ {
169
+ "name": "model-00018-of-00019.safetensors",
170
+ "size": 3392197360,
171
+ "sha256": "a408f74e0801216c76f34f404e487b7f72177fc34efab5d0f14edf7e7630345f"
172
+ },
173
+ {
174
+ "name": "model-00019-of-00019.safetensors",
175
+ "size": 6088313008,
176
+ "sha256": "48a75e0f87ae40e40ff7190e53576355ca4eea028a64882e302893287ab7b49f"
177
+ },
178
+ {
179
+ "name": "model-parallel-ffn.safetensors",
180
+ "size": 4529878968,
181
+ "sha256": "9ea1ead9b6ea1ebae0d791145e43c727cd5bd7cc8dee19d4c9d95c0eb1ff4f1c"
182
+ },
183
+ {
184
+ "name": "model.safetensors.index.json",
185
+ "size": 145141,
186
+ "sha256": null
187
+ },
188
+ {
189
+ "name": "modeling_agnes.py",
190
+ "size": 70576,
191
+ "sha256": null
192
+ },
193
+ {
194
+ "name": "preprocessor_config.json",
195
+ "size": 476,
196
+ "sha256": null
197
+ },
198
+ {
199
+ "name": "processing_agnes.py",
200
+ "size": 5849,
201
+ "sha256": null
202
+ },
203
+ {
204
+ "name": "serve.sh",
205
+ "size": 1448,
206
+ "sha256": null
207
+ },
208
+ {
209
+ "name": "sglang_patch/README.md",
210
+ "size": 2645,
211
+ "sha256": null
212
+ },
213
+ {
214
+ "name": "sglang_patch/agnes_sglang_config.py",
215
+ "size": 3174,
216
+ "sha256": null
217
+ },
218
+ {
219
+ "name": "sglang_patch/apply_patch.py",
220
+ "size": 4832,
221
+ "sha256": null
222
+ },
223
+ {
224
+ "name": "sglang_patch/nightly-dev-20260908-20ca564b/sglang/srt/configs/agnes.py",
225
+ "size": 3174,
226
+ "sha256": null
227
+ },
228
+ {
229
+ "name": "sglang_patch/nightly-dev-20260908-20ca564b/sglang/srt/models/qwen3_5.py",
230
+ "size": 110213,
231
+ "sha256": null
232
+ },
233
+ {
234
+ "name": "sglang_patch/nightly-dev-20260908-20ca564b/sglang/srt/utils/hf_transformers/common.py",
235
+ "size": 25576,
236
+ "sha256": null
237
+ },
238
+ {
239
+ "name": "sglang_patch/v0.5.19/sglang/srt/configs/agnes.py",
240
+ "size": 3174,
241
+ "sha256": null
242
+ },
243
+ {
244
+ "name": "sglang_patch/v0.5.19/sglang/srt/models/qwen3_5.py",
245
+ "size": 109665,
246
+ "sha256": null
247
+ },
248
+ {
249
+ "name": "sglang_patch/v0.5.19/sglang/srt/utils/hf_transformers/common.py",
250
+ "size": 23485,
251
+ "sha256": null
252
+ },
253
+ {
254
+ "name": "tokenizer.json",
255
+ "size": 9835269,
256
+ "sha256": null
257
+ },
258
+ {
259
+ "name": "tokenizer_config.json",
260
+ "size": 18884,
261
+ "sha256": null
262
+ },
263
+ {
264
+ "name": "video_preprocessor_config.json",
265
+ "size": 460,
266
+ "sha256": null
267
+ },
268
+ {
269
+ "name": "video_processing_agnes.py",
270
+ "size": 8832,
271
+ "sha256": null
272
+ },
273
+ {
274
+ "name": "vocab.json",
275
+ "size": 6722759,
276
+ "sha256": null
277
+ }
278
+ ]
279
+ }