Pin plain NVFP4 build, source, and bounded validation plan
Browse files- README.md +13 -0
- reproducibility/bootstrap_build.sh +17 -0
- reproducibility/build.py +208 -0
- reproducibility/dispatch_build.py +52 -0
- reproducibility/plan.json +14 -0
- reproducibility/runtime_image.json +55 -0
- reproducibility/source_meta.json +279 -0
README.md
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: apache-2.0
|
| 3 |
+
base_model: Agnes-AI/Agnes-3.0-Flash
|
| 4 |
+
base_model_relation: quantized
|
| 5 |
+
pipeline_tag: image-text-to-text
|
| 6 |
+
tags:
|
| 7 |
+
- nvfp4
|
| 8 |
+
- modelopt
|
| 9 |
+
- quantized
|
| 10 |
+
---
|
| 11 |
+
# Agnes-3.0-Flash Preview — plain NVFP4
|
| 12 |
+
|
| 13 |
+
Private build in progress. The pinned source is the 33B Preview checkpoint with 262144 context. Native evaluation and public release are pending.
|
reproducibility/bootstrap_build.sh
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env bash
|
| 2 |
+
set -euo pipefail
|
| 3 |
+
mkdir -p /workspace/agnes/code
|
| 4 |
+
python -m pip install --break-system-packages --no-cache-dir uv
|
| 5 |
+
uv venv --system-site-packages /workspace/agnes/venv
|
| 6 |
+
export PATH="/workspace/agnes/venv/bin:$PATH"
|
| 7 |
+
uv pip install --python /workspace/agnes/venv/bin/python \
|
| 8 |
+
'torch==2.13.0' 'transformers==5.14.0' 'accelerate==1.12.0' \
|
| 9 |
+
'huggingface_hub==1.18.0' 'pillow>=12' 'sentencepiece>=0.2' 'einops' \
|
| 10 |
+
'nvidia-modelopt @ https://github.com/NVIDIA/Model-Optimizer/archive/5cae3940402f1ced98069a666b0bec72ec8b33b5.tar.gz'
|
| 11 |
+
python - <<'PY'
|
| 12 |
+
import os
|
| 13 |
+
from huggingface_hub import hf_hub_download
|
| 14 |
+
hf_hub_download('ProCreations/Agnes-3.0-Flash-NVFP4','reproducibility/build.py',
|
| 15 |
+
revision=os.environ['AGNES_CODE_REVISION'],local_dir='/workspace/agnes/code')
|
| 16 |
+
PY
|
| 17 |
+
python -u /workspace/agnes/code/reproducibility/build.py
|
reproducibility/build.py
ADDED
|
@@ -0,0 +1,208 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Plain data-free NVFP4; deterministic source-supported FFN folding."""
|
| 2 |
+
import collections
|
| 3 |
+
import copy
|
| 4 |
+
import gc
|
| 5 |
+
import hashlib
|
| 6 |
+
import importlib.metadata
|
| 7 |
+
import json
|
| 8 |
+
import os
|
| 9 |
+
import re
|
| 10 |
+
import shutil
|
| 11 |
+
import time
|
| 12 |
+
from pathlib import Path
|
| 13 |
+
|
| 14 |
+
import torch
|
| 15 |
+
from huggingface_hub import HfApi, snapshot_download
|
| 16 |
+
from safetensors import safe_open
|
| 17 |
+
from safetensors.torch import save_file
|
| 18 |
+
from transformers import AutoModelForImageTextToText
|
| 19 |
+
import modelopt.torch.quantization as mtq
|
| 20 |
+
from modelopt.torch.quantization.nn import TensorQuantizer
|
| 21 |
+
from modelopt.torch.export import export_hf_checkpoint
|
| 22 |
+
|
| 23 |
+
SOURCE = 'Agnes-AI/Agnes-3.0-Flash'
|
| 24 |
+
REV = '24f712ce59379b54c4a141d2708c35daf5ff613b'
|
| 25 |
+
TARGET = 'ProCreations/Agnes-3.0-Flash-NVFP4'
|
| 26 |
+
ROOT = Path('/workspace/agnes')
|
| 27 |
+
OUTPUT = ROOT/'export'
|
| 28 |
+
API = HfApi()
|
| 29 |
+
SELECT = re.compile(r'^model\.language_model\.layers\.\d+\.(?:mlp\.(?:gate_proj|up_proj|down_proj)|global_attn\.(?:q_proj|k_proj|v_proj|o_proj))\.weight$')
|
| 30 |
+
REPORT = dict(source=SOURCE, source_revision=REV, target=TARGET,
|
| 31 |
+
status='building', code_revision=os.environ['AGNES_CODE_REVISION'],
|
| 32 |
+
modelopt_revision='5cae3940402f1ced98069a666b0bec72ec8b33b5',
|
| 33 |
+
quantization=dict(method='plain NVFP4 max', training=False, calibration_examples=0,
|
| 34 |
+
calibration_tokens=0, calibration_forward_calls=0, activation_global_scale=1.0,
|
| 35 |
+
block_size=16, activation_block_scales='dynamic E4M3', kv_cache_quantized=False),
|
| 36 |
+
folding=dict(main_width=17408, parallel_width=2048, folded_width=19456,
|
| 37 |
+
gate_up_dimension=0, down_dimension=1, layers=72,
|
| 38 |
+
reason='Match the upstream SGLang BF16 loader before quantization'),
|
| 39 |
+
evaluation=dict(status='pending'))
|
| 40 |
+
|
| 41 |
+
def record():
|
| 42 |
+
(ROOT/'quality_report.json').write_text(json.dumps(REPORT, indent=2)+'\n')
|
| 43 |
+
API.upload_file(repo_id=TARGET, path_or_fileobj=str(ROOT/'quality_report.json'),
|
| 44 |
+
path_in_repo='quality_report.json', commit_message='Record plain NVFP4 build checks')
|
| 45 |
+
|
| 46 |
+
def selected(name):
|
| 47 |
+
return bool(SELECT.fullmatch(name))
|
| 48 |
+
|
| 49 |
+
def main():
|
| 50 |
+
assert API.model_info(TARGET).private
|
| 51 |
+
assert not any(n.endswith('.safetensors') for n in API.list_repo_files(TARGET))
|
| 52 |
+
ROOT.mkdir(parents=True, exist_ok=True)
|
| 53 |
+
torch.set_num_threads(8)
|
| 54 |
+
torch.manual_seed(20260912)
|
| 55 |
+
REPORT['versions'] = {p:importlib.metadata.version(p) for p in ['torch','transformers','nvidia-modelopt','accelerate']}
|
| 56 |
+
REPORT['hardware'] = [torch.cuda.get_device_name(0)]
|
| 57 |
+
record()
|
| 58 |
+
source = Path(snapshot_download(SOURCE, revision=REV, max_workers=8))
|
| 59 |
+
src_index = json.loads((source/'model.safetensors.index.json').read_text())
|
| 60 |
+
src_map = src_index['weight_map']
|
| 61 |
+
model = AutoModelForImageTextToText.from_pretrained(source, trust_remote_code=True,
|
| 62 |
+
dtype=torch.bfloat16, device_map={'':0}, attn_implementation='sdpa').eval()
|
| 63 |
+
assert all(p.device.type == 'cuda' for p in model.parameters())
|
| 64 |
+
assert len(model.model.language_model.layers) == 72
|
| 65 |
+
with torch.no_grad():
|
| 66 |
+
for layer in model.model.language_model.layers:
|
| 67 |
+
mlp = layer.mlp
|
| 68 |
+
assert mlp.parallel_ffn is not None
|
| 69 |
+
for name in ['gate_proj','up_proj','down_proj']:
|
| 70 |
+
original, branch = getattr(mlp,name), getattr(mlp.parallel_ffn,name)
|
| 71 |
+
dim = 1 if name == 'down_proj' else 0
|
| 72 |
+
merged = torch.cat([original.weight,branch.weight], dim=dim)
|
| 73 |
+
# Check both slices before discarding the separate branches.
|
| 74 |
+
a,b = merged.split([original.weight.shape[dim],branch.weight.shape[dim]],dim=dim)
|
| 75 |
+
assert torch.equal(a,original.weight) and torch.equal(b,branch.weight)
|
| 76 |
+
replacement = torch.nn.Linear(merged.shape[1],merged.shape[0],bias=False,
|
| 77 |
+
device=merged.device,dtype=merged.dtype)
|
| 78 |
+
replacement.weight = torch.nn.Parameter(merged,requires_grad=False)
|
| 79 |
+
setattr(mlp,name,replacement)
|
| 80 |
+
mlp.parallel_ffn = None
|
| 81 |
+
text = model.config.text_config
|
| 82 |
+
text.agnes_original_intermediate_size = 17408
|
| 83 |
+
text.agnes_original_parallel_ffn_intermediate_size = 2048
|
| 84 |
+
text.intermediate_size = 19456
|
| 85 |
+
text.parallel_ffn_intermediate_size = 0
|
| 86 |
+
gc.collect(); torch.cuda.empty_cache()
|
| 87 |
+
preset = copy.deepcopy(mtq.NVFP4_DEFAULT_CFG)
|
| 88 |
+
weight = next(x['cfg'] for x in preset['quant_cfg'] if x.get('quantizer_name')=='*weight_quantizer')
|
| 89 |
+
activation = next(x['cfg'] for x in preset['quant_cfg'] if x.get('quantizer_name')=='*input_quantizer')
|
| 90 |
+
activation['constant_amax'] = 2688.0
|
| 91 |
+
rules = [dict(quantizer_name='*',enable=False)]
|
| 92 |
+
targets = [name for name,module in model.named_modules()
|
| 93 |
+
if isinstance(module,torch.nn.Linear) and selected(name+'.weight')]
|
| 94 |
+
assert len(targets) == 288, len(targets)
|
| 95 |
+
for name in targets:
|
| 96 |
+
rules += [dict(quantizer_name=name+'.weight_quantizer',cfg=weight),
|
| 97 |
+
dict(quantizer_name=name+'.input_quantizer',cfg=activation)]
|
| 98 |
+
cfg = dict(quant_cfg=rules,algorithm=dict(method='max',layerwise=dict(enable=False),
|
| 99 |
+
skip_forward_without_activation_calib=True))
|
| 100 |
+
REPORT['quantization']['configuration'] = cfg
|
| 101 |
+
def forbid(*args):
|
| 102 |
+
REPORT['quantization']['calibration_forward_calls'] += 1
|
| 103 |
+
raise RuntimeError('Data-free quantization may not execute calibration forwards')
|
| 104 |
+
hook = model.register_forward_pre_hook(forbid)
|
| 105 |
+
started = time.monotonic()
|
| 106 |
+
with torch.inference_mode():
|
| 107 |
+
mtq.quantize(model,cfg,forward_loop=None)
|
| 108 |
+
enabled = [(name,m) for name,m in model.named_modules()
|
| 109 |
+
if isinstance(m,TensorQuantizer) and m.is_enabled]
|
| 110 |
+
assert len(enabled) == 576, len(enabled)
|
| 111 |
+
assert all(name.rsplit('.',1)[0] in targets for name,m in enabled)
|
| 112 |
+
assert all(getattr(m,'_amax',None) is not None for name,m in enabled if name.endswith('weight_quantizer'))
|
| 113 |
+
REPORT['quantization']['weight_statistics_seconds'] = time.monotonic()-started
|
| 114 |
+
hook.remove()
|
| 115 |
+
REPORT['export_graph_inspection'] = dict(dummy_forward_calls=0,input_tokens=0,quantizers_disabled=True)
|
| 116 |
+
def inspect_probe(module,args):
|
| 117 |
+
assert not any(isinstance(m,TensorQuantizer) and (m.is_enabled or m._if_calib) for m in module.modules())
|
| 118 |
+
assert args[0].shape == (1,2) and torch.equal(args[0],torch.ones_like(args[0]))
|
| 119 |
+
REPORT['export_graph_inspection']['dummy_forward_calls'] += 1
|
| 120 |
+
REPORT['export_graph_inspection']['input_tokens'] += args[0].numel()
|
| 121 |
+
hook = model.register_forward_pre_hook(inspect_probe)
|
| 122 |
+
export_hf_checkpoint(model,dtype=torch.bfloat16,export_dir=OUTPUT,max_shard_size='5GB')
|
| 123 |
+
hook.remove()
|
| 124 |
+
del model,enabled
|
| 125 |
+
gc.collect(); torch.cuda.empty_cache()
|
| 126 |
+
# The upstream Transformers class ignores MTP: retain every original MTP
|
| 127 |
+
# tensor separately, without advertising untested speculative decoding.
|
| 128 |
+
dst_index = json.loads((OUTPUT/'model.safetensors.index.json').read_text())
|
| 129 |
+
mtp = {}
|
| 130 |
+
for name,filename in src_map.items():
|
| 131 |
+
if name.startswith('mtp.'):
|
| 132 |
+
with safe_open(source/filename,framework='pt',device='cpu') as f:
|
| 133 |
+
mtp[name] = f.get_tensor(name).clone()
|
| 134 |
+
assert mtp and not any(name in dst_index['weight_map'] for name in mtp)
|
| 135 |
+
save_file(mtp,str(OUTPUT/'model-mtp.safetensors'),metadata={'format':'pt'})
|
| 136 |
+
dst_index['weight_map'].update({name:'model-mtp.safetensors' for name in mtp})
|
| 137 |
+
dst_index['metadata']['total_size'] += sum(t.numel()*t.element_size() for t in mtp.values())
|
| 138 |
+
(OUTPUT/'model.safetensors.index.json').write_text(json.dumps(dst_index,indent=2)+'\n')
|
| 139 |
+
for file in source.rglob('*'):
|
| 140 |
+
rel=file.relative_to(source)
|
| 141 |
+
if file.is_file() and '.cache' not in rel.parts and not file.name.endswith('.safetensors') and file.name not in ['config.json','model.safetensors.index.json','README.md','.gitattributes']:
|
| 142 |
+
(OUTPUT/rel).parent.mkdir(parents=True,exist_ok=True)
|
| 143 |
+
shutil.copy2(file,OUTPUT/rel)
|
| 144 |
+
# The upstream loader otherwise skips renaming when the branch width is 0.
|
| 145 |
+
# This checkpoint has already folded its branches before quantization.
|
| 146 |
+
for file in OUTPUT.glob('sglang_patch/*/sglang/srt/models/qwen3_5.py'):
|
| 147 |
+
contents=file.read_text()
|
| 148 |
+
before=' if width <= 0:\n yield from weights\n return\n'
|
| 149 |
+
after=' if width <= 0:\n for name, weight in weights:\n yield name.replace(".delta_attn.", ".linear_attn.").replace(".global_attn.", ".self_attn."), weight\n return\n'
|
| 150 |
+
assert contents.count(before)==1
|
| 151 |
+
file.write_text(contents.replace(before,after))
|
| 152 |
+
# Both Transformers names and translated serving names must be excluded.
|
| 153 |
+
for filename in ['config.json','hf_quant_config.json']:
|
| 154 |
+
file=OUTPUT/filename
|
| 155 |
+
obj=json.loads(file.read_text())
|
| 156 |
+
quant=obj['quantization_config'] if filename=='config.json' else obj['quantization']
|
| 157 |
+
excludes=list(quant.get('exclude_modules',[]))
|
| 158 |
+
excludes += ['lm_head','model.visual*','mtp*']
|
| 159 |
+
for i in range(72):
|
| 160 |
+
excludes += [f'model.language_model.layers.{i}.delta_attn*',
|
| 161 |
+
f'model.language_model.layers.{i}.linear_attn*']
|
| 162 |
+
quant['exclude_modules']=sorted(set(excludes))
|
| 163 |
+
file.write_text(json.dumps(obj,indent=2)+'\n')
|
| 164 |
+
preserved=0; preserved_count=0
|
| 165 |
+
for filename in sorted(set(src_map.values())):
|
| 166 |
+
with safe_open(source/filename,framework='pt',device='cpu') as sf:
|
| 167 |
+
for name in sf.keys():
|
| 168 |
+
if selected(name) or '.mlp.parallel_ffn.' in name: continue
|
| 169 |
+
assert name in dst_index['weight_map'],name
|
| 170 |
+
with safe_open(OUTPUT/dst_index['weight_map'][name],framework='pt',device='cpu') as df:
|
| 171 |
+
a,b=sf.get_tensor(name),df.get_tensor(name)
|
| 172 |
+
assert a.dtype==b.dtype and torch.equal(a,b),name
|
| 173 |
+
preserved += a.numel()*a.element_size(); preserved_count+=1
|
| 174 |
+
counts=collections.Counter(); packed=0; inputs=0
|
| 175 |
+
for filename in sorted(set(dst_index['weight_map'].values())):
|
| 176 |
+
with safe_open(OUTPUT/filename,framework='pt',device='cpu') as f:
|
| 177 |
+
for name in f.keys():
|
| 178 |
+
tensor=f.get_tensor(name)
|
| 179 |
+
counts[str(tensor.dtype)]+=tensor.numel()*tensor.element_size()
|
| 180 |
+
if selected(name):
|
| 181 |
+
assert tensor.dtype==torch.uint8,name
|
| 182 |
+
packed+=1
|
| 183 |
+
if 'scale' in name:
|
| 184 |
+
assert torch.isfinite(tensor.float()).all() and (tensor.float()>0).all(),name
|
| 185 |
+
if name.endswith('input_scale'):
|
| 186 |
+
assert torch.equal(tensor,torch.ones_like(tensor)),name
|
| 187 |
+
inputs+=1
|
| 188 |
+
assert packed==len(targets)==288 and inputs==288,(packed,inputs)
|
| 189 |
+
assert REPORT['quantization']['calibration_forward_calls']==0
|
| 190 |
+
REPORT['export']=dict(source_tensor_bytes=src_index['metadata']['total_size'],
|
| 191 |
+
exported_weight_bytes=sum((OUTPUT/f).stat().st_size for f in set(dst_index['weight_map'].values())),
|
| 192 |
+
tensor_bytes_by_dtype=dict(counts),packed_linear_weights=packed,
|
| 193 |
+
bf16_preserved_bytes=preserved,bf16_preserved_tensors=preserved_count,
|
| 194 |
+
all_unquantized_tensors_bitwise_equal=True,mtp_tensors_preserved=len(mtp),
|
| 195 |
+
activation_global_scales_exactly_one=inputs)
|
| 196 |
+
REPORT['status']='packed_export_verified_pending_native_evaluation'
|
| 197 |
+
record()
|
| 198 |
+
shutil.copy2(ROOT/'quality_report.json',OUTPUT/'quality_report.json')
|
| 199 |
+
API.upload_folder(repo_id=TARGET,folder_path=OUTPUT,
|
| 200 |
+
commit_message='Upload verified plain NVFP4 Agnes Preview checkpoint')
|
| 201 |
+
print('AGNES_BUILD_COMPLETE '+json.dumps(REPORT['export']),flush=True)
|
| 202 |
+
|
| 203 |
+
if __name__=='__main__':
|
| 204 |
+
try: main()
|
| 205 |
+
except Exception:
|
| 206 |
+
REPORT['status']='build_failed'
|
| 207 |
+
record()
|
| 208 |
+
raise
|
reproducibility/dispatch_build.py
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""One data-free build; hard timeout and a separate bounded Agnes ledger."""
|
| 2 |
+
import json
|
| 3 |
+
from pathlib import Path
|
| 4 |
+
from huggingface_hub import HfApi, CommitOperationAdd, get_token
|
| 5 |
+
|
| 6 |
+
ROOT=Path(__file__).resolve().parent
|
| 7 |
+
api=HfApi()
|
| 8 |
+
repo='ProCreations/Agnes-3.0-Flash-NVFP4'
|
| 9 |
+
assert api.whoami()['name']=='ProCreations'
|
| 10 |
+
ledger=ROOT/'jobs.jsonl'
|
| 11 |
+
assert not ledger.exists(),'Inspect previous attempts before dispatching again'
|
| 12 |
+
api.create_repo(repo_id=repo,private=True,exist_ok=True)
|
| 13 |
+
assert api.model_info(repo).private
|
| 14 |
+
assert not any(n.endswith('.safetensors') for n in api.list_repo_files(repo))
|
| 15 |
+
plan=dict(source='Agnes-AI/Agnes-3.0-Flash',source_revision='24f712ce59379b54c4a141d2708c35daf5ff613b',
|
| 16 |
+
target=repo,method='plain data-free NVFP4',training=False,calibration_examples=0,
|
| 17 |
+
quantized='All decoder MLP and global attention linear weights; branches folded as in upstream SGLang',
|
| 18 |
+
preserved='Recurrent attention, vision, embeddings, output head, norms, MTP',
|
| 19 |
+
agnes_compute_ceiling_usd=40,prior_nex_compute_upper_bound_usd=98.303050825,
|
| 20 |
+
combined_user_cap_usd=200,publication='Public after native validation')
|
| 21 |
+
(ROOT/'plan.json').write_text(json.dumps(plan,indent=2)+'\n')
|
| 22 |
+
readme='''---
|
| 23 |
+
license: apache-2.0
|
| 24 |
+
base_model: Agnes-AI/Agnes-3.0-Flash
|
| 25 |
+
base_model_relation: quantized
|
| 26 |
+
pipeline_tag: image-text-to-text
|
| 27 |
+
tags:
|
| 28 |
+
- nvfp4
|
| 29 |
+
- modelopt
|
| 30 |
+
- quantized
|
| 31 |
+
---
|
| 32 |
+
# Agnes-3.0-Flash Preview — plain NVFP4
|
| 33 |
+
|
| 34 |
+
Private build in progress. The pinned source is the 33B Preview checkpoint with 262144 context. Native evaluation and public release are pending.
|
| 35 |
+
'''
|
| 36 |
+
(ROOT/'README.pending.md').write_text(readme)
|
| 37 |
+
names=['build.py','bootstrap_build.sh','dispatch_build.py','plan.json','source_meta.json','runtime_image.json']
|
| 38 |
+
commit=api.create_commit(repo_id=repo,operations=[CommitOperationAdd(path_in_repo='reproducibility/'+n,path_or_fileobj=str(ROOT/n)) for n in names]+[
|
| 39 |
+
CommitOperationAdd(path_in_repo='README.md',path_or_fileobj=str(ROOT/'README.pending.md'))],
|
| 40 |
+
commit_message='Pin plain NVFP4 build, source, and bounded validation plan')
|
| 41 |
+
rate=next(h.unit_cost_usd*60 for h in api.list_jobs_hardware() if h.name=='h200')
|
| 42 |
+
job=api.run_job(image='pytorch/pytorch:2.13.0-cuda13.0-cudnn9-devel',flavor='h200',timeout=5400,
|
| 43 |
+
command=['bash','-c',(ROOT/'bootstrap_build.sh').read_text()],
|
| 44 |
+
env={'AGNES_CODE_REVISION':commit.oid,'HF_HOME':'/workspace/agnes/hf-cache',
|
| 45 |
+
'PYTHONUNBUFFERED':'1','TOKENIZERS_PARALLELISM':'false','OMP_NUM_THREADS':'8',
|
| 46 |
+
'HF_HUB_DOWNLOAD_TIMEOUT':'180','TQDM_MININTERVAL':'30',
|
| 47 |
+
'PYTORCH_CUDA_ALLOC_CONF':'expandable_segments:True'},
|
| 48 |
+
secrets={'HF_TOKEN':get_token()},labels={'project':'agnes-flash-nvfp4','phase':'plain-build'})
|
| 49 |
+
row=dict(job_id=job.id,url=job.url,phase='plain-build',flavor='h200',rate_usd_hour=rate,
|
| 50 |
+
timeout_hours=1.5,maximum_cost_usd=rate*1.5,code_revision=commit.oid)
|
| 51 |
+
ledger.write_text(json.dumps(row)+'\n')
|
| 52 |
+
print(json.dumps(row))
|
reproducibility/plan.json
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"source": "Agnes-AI/Agnes-3.0-Flash",
|
| 3 |
+
"source_revision": "24f712ce59379b54c4a141d2708c35daf5ff613b",
|
| 4 |
+
"target": "ProCreations/Agnes-3.0-Flash-NVFP4",
|
| 5 |
+
"method": "plain data-free NVFP4",
|
| 6 |
+
"training": false,
|
| 7 |
+
"calibration_examples": 0,
|
| 8 |
+
"quantized": "All decoder MLP and global attention linear weights; branches folded as in upstream SGLang",
|
| 9 |
+
"preserved": "Recurrent attention, vision, embeddings, output head, norms, MTP",
|
| 10 |
+
"agnes_compute_ceiling_usd": 40,
|
| 11 |
+
"prior_nex_compute_upper_bound_usd": 98.303050825,
|
| 12 |
+
"combined_user_cap_usd": 200,
|
| 13 |
+
"publication": "Public after native validation"
|
| 14 |
+
}
|
reproducibility/runtime_image.json
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"image": "lmsysorg/sglang",
|
| 3 |
+
"tag": "nightly-dev-20260908-20ca564b",
|
| 4 |
+
"digest": "sha256:9a352a35c973a2357372e85f3bcb5388b6b3c46c1329165987260f3b089647dc",
|
| 5 |
+
"manifest": {
|
| 6 |
+
"schemaVersion": 2,
|
| 7 |
+
"mediaType": "application/vnd.oci.image.index.v1+json",
|
| 8 |
+
"manifests": [
|
| 9 |
+
{
|
| 10 |
+
"mediaType": "application/vnd.oci.image.manifest.v1+json",
|
| 11 |
+
"digest": "sha256:2d53500cb72184020c59fe16aab3512d53b1aa8ee6195bfb2aaf5855d58099df",
|
| 12 |
+
"size": 13687,
|
| 13 |
+
"platform": {
|
| 14 |
+
"architecture": "amd64",
|
| 15 |
+
"os": "linux"
|
| 16 |
+
}
|
| 17 |
+
},
|
| 18 |
+
{
|
| 19 |
+
"mediaType": "application/vnd.oci.image.manifest.v1+json",
|
| 20 |
+
"digest": "sha256:b6792a80cd34d743b2e3be7f088dfb701328a900f64e8008b16279b616f6db46",
|
| 21 |
+
"size": 838,
|
| 22 |
+
"annotations": {
|
| 23 |
+
"vnd.docker.reference.digest": "sha256:2d53500cb72184020c59fe16aab3512d53b1aa8ee6195bfb2aaf5855d58099df",
|
| 24 |
+
"vnd.docker.reference.type": "attestation-manifest"
|
| 25 |
+
},
|
| 26 |
+
"platform": {
|
| 27 |
+
"architecture": "unknown",
|
| 28 |
+
"os": "unknown"
|
| 29 |
+
}
|
| 30 |
+
},
|
| 31 |
+
{
|
| 32 |
+
"mediaType": "application/vnd.oci.image.manifest.v1+json",
|
| 33 |
+
"digest": "sha256:c5697af9d2b89629b8d392509de6085a2c16b492b76acc975660417d86f27491",
|
| 34 |
+
"size": 13673,
|
| 35 |
+
"platform": {
|
| 36 |
+
"architecture": "arm64",
|
| 37 |
+
"os": "linux"
|
| 38 |
+
}
|
| 39 |
+
},
|
| 40 |
+
{
|
| 41 |
+
"mediaType": "application/vnd.oci.image.manifest.v1+json",
|
| 42 |
+
"digest": "sha256:b1b088a5e7a06a4685d684ec7e109e159ddb722fd89dc476e4c0987c585dae7e",
|
| 43 |
+
"size": 838,
|
| 44 |
+
"annotations": {
|
| 45 |
+
"vnd.docker.reference.digest": "sha256:c5697af9d2b89629b8d392509de6085a2c16b492b76acc975660417d86f27491",
|
| 46 |
+
"vnd.docker.reference.type": "attestation-manifest"
|
| 47 |
+
},
|
| 48 |
+
"platform": {
|
| 49 |
+
"architecture": "unknown",
|
| 50 |
+
"os": "unknown"
|
| 51 |
+
}
|
| 52 |
+
}
|
| 53 |
+
]
|
| 54 |
+
}
|
| 55 |
+
}
|
reproducibility/source_meta.json
ADDED
|
@@ -0,0 +1,279 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"repo": "Agnes-AI/Agnes-3.0-Flash",
|
| 3 |
+
"revision": "24f712ce59379b54c4a141d2708c35daf5ff613b",
|
| 4 |
+
"private": false,
|
| 5 |
+
"gated": false,
|
| 6 |
+
"card": {
|
| 7 |
+
"language": [
|
| 8 |
+
"en",
|
| 9 |
+
"zh"
|
| 10 |
+
],
|
| 11 |
+
"library_name": "transformers",
|
| 12 |
+
"license": "apache-2.0",
|
| 13 |
+
"pipeline_tag": "image-text-to-text",
|
| 14 |
+
"tags": [
|
| 15 |
+
"agnes-ai",
|
| 16 |
+
"reasoning",
|
| 17 |
+
"multimodal",
|
| 18 |
+
"long-context",
|
| 19 |
+
"hybrid-attention"
|
| 20 |
+
]
|
| 21 |
+
},
|
| 22 |
+
"files": [
|
| 23 |
+
{
|
| 24 |
+
"name": ".gitattributes",
|
| 25 |
+
"size": 1519,
|
| 26 |
+
"sha256": null
|
| 27 |
+
},
|
| 28 |
+
{
|
| 29 |
+
"name": "LICENSE",
|
| 30 |
+
"size": 11357,
|
| 31 |
+
"sha256": null
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"name": "README.md",
|
| 35 |
+
"size": 16314,
|
| 36 |
+
"sha256": null
|
| 37 |
+
},
|
| 38 |
+
{
|
| 39 |
+
"name": "README_zh.md",
|
| 40 |
+
"size": 14714,
|
| 41 |
+
"sha256": null
|
| 42 |
+
},
|
| 43 |
+
{
|
| 44 |
+
"name": "assets/agnes_benchmarks.svg",
|
| 45 |
+
"size": 12421,
|
| 46 |
+
"sha256": null
|
| 47 |
+
},
|
| 48 |
+
{
|
| 49 |
+
"name": "assets/agnes_logo.svg",
|
| 50 |
+
"size": 2435,
|
| 51 |
+
"sha256": null
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"name": "chat_template.jinja",
|
| 55 |
+
"size": 8952,
|
| 56 |
+
"sha256": null
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
"name": "config.json",
|
| 60 |
+
"size": 4505,
|
| 61 |
+
"sha256": null
|
| 62 |
+
},
|
| 63 |
+
{
|
| 64 |
+
"name": "configuration_agnes.py",
|
| 65 |
+
"size": 7516,
|
| 66 |
+
"sha256": null
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"name": "generation_config.json",
|
| 70 |
+
"size": 202,
|
| 71 |
+
"sha256": null
|
| 72 |
+
},
|
| 73 |
+
{
|
| 74 |
+
"name": "image_processing_agnes.py",
|
| 75 |
+
"size": 8700,
|
| 76 |
+
"sha256": null
|
| 77 |
+
},
|
| 78 |
+
{
|
| 79 |
+
"name": "merges.txt",
|
| 80 |
+
"size": 3353259,
|
| 81 |
+
"sha256": null
|
| 82 |
+
},
|
| 83 |
+
{
|
| 84 |
+
"name": "model-00001-of-00019.safetensors",
|
| 85 |
+
"size": 3966730528,
|
| 86 |
+
"sha256": "5697cc0f8b4223adab6e8fb6043727752c728e485a3621b693381073409e3070"
|
| 87 |
+
},
|
| 88 |
+
{
|
| 89 |
+
"name": "model-00002-of-00019.safetensors",
|
| 90 |
+
"size": 3043080320,
|
| 91 |
+
"sha256": "e3d8adc00d4f6568b257ccdbcfdedde894024eb2a10abd125426fe8440fc563a"
|
| 92 |
+
},
|
| 93 |
+
{
|
| 94 |
+
"name": "model-00003-of-00019.safetensors",
|
| 95 |
+
"size": 2542796952,
|
| 96 |
+
"sha256": "2e1bf62cbcd406eaa64b60d10353e1f0ef4039d0976e56f05cabe953454f9968"
|
| 97 |
+
},
|
| 98 |
+
{
|
| 99 |
+
"name": "model-00004-of-00019.safetensors",
|
| 100 |
+
"size": 3988973120,
|
| 101 |
+
"sha256": "2962bc38c2b69709d6df78cffe9692ab4e761d47ba2f2c0e61c808484a2df825"
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"name": "model-00005-of-00019.safetensors",
|
| 105 |
+
"size": 2099339848,
|
| 106 |
+
"sha256": "ecb1d015bca88250dbc9e167479c5e740de21b5a70fc38cc84b6c3388db9d3a9"
|
| 107 |
+
},
|
| 108 |
+
{
|
| 109 |
+
"name": "model-00006-of-00019.safetensors",
|
| 110 |
+
"size": 3979553664,
|
| 111 |
+
"sha256": "ae9c2e979d10655fded3bbdbfddfb2f72b3d1ac5195c9e214756f694196fd523"
|
| 112 |
+
},
|
| 113 |
+
{
|
| 114 |
+
"name": "model-00007-of-00019.safetensors",
|
| 115 |
+
"size": 2108759344,
|
| 116 |
+
"sha256": "dfdb74b2182b870b602127a64e3b9995f138bfe707168b7339c759623996e2d7"
|
| 117 |
+
},
|
| 118 |
+
{
|
| 119 |
+
"name": "model-00008-of-00019.safetensors",
|
| 120 |
+
"size": 3979553664,
|
| 121 |
+
"sha256": "24db669382d366c90d6c9fd2ce3cfceee836c3ec8a8a3ae62313ce8f3cd55862"
|
| 122 |
+
},
|
| 123 |
+
{
|
| 124 |
+
"name": "model-00009-of-00019.safetensors",
|
| 125 |
+
"size": 2108759344,
|
| 126 |
+
"sha256": "fe9de765012eb1c7bd2dd30f5b2b7579b5eef4de67805023b2d86c2bea8d2511"
|
| 127 |
+
},
|
| 128 |
+
{
|
| 129 |
+
"name": "model-00010-of-00019.safetensors",
|
| 130 |
+
"size": 3979553664,
|
| 131 |
+
"sha256": "a1819ace7b218bdad22a194ed9f09e4d436ba8ff522bf1e09b070d6d2bb8781a"
|
| 132 |
+
},
|
| 133 |
+
{
|
| 134 |
+
"name": "model-00011-of-00019.safetensors",
|
| 135 |
+
"size": 2108759344,
|
| 136 |
+
"sha256": "4a7b4caaedf55e0b4d4bad1524b0f433b5e86ce4a7ab13950c3749dc10b56f5f"
|
| 137 |
+
},
|
| 138 |
+
{
|
| 139 |
+
"name": "model-00012-of-00019.safetensors",
|
| 140 |
+
"size": 3979553664,
|
| 141 |
+
"sha256": "1f937940e0dcd5dc28ab4117fc0437c53e10328b5b5ff52b126bba01da208d7d"
|
| 142 |
+
},
|
| 143 |
+
{
|
| 144 |
+
"name": "model-00013-of-00019.safetensors",
|
| 145 |
+
"size": 2108759344,
|
| 146 |
+
"sha256": "ee54d2d219596d6d8900aaac24a4afd59028b883d57d5eba77ceb8d207c56cb3"
|
| 147 |
+
},
|
| 148 |
+
{
|
| 149 |
+
"name": "model-00014-of-00019.safetensors",
|
| 150 |
+
"size": 3979553664,
|
| 151 |
+
"sha256": "064a0cfd148b1ca7188c83a40a5b30ac5b26e4f3a3982e50948945cfa6872cae"
|
| 152 |
+
},
|
| 153 |
+
{
|
| 154 |
+
"name": "model-00015-of-00019.safetensors",
|
| 155 |
+
"size": 2108759344,
|
| 156 |
+
"sha256": "d395947e2ef2274f151e1b83e8357befa9f40f0bca296d99d1aa69abb703cf51"
|
| 157 |
+
},
|
| 158 |
+
{
|
| 159 |
+
"name": "model-00016-of-00019.safetensors",
|
| 160 |
+
"size": 3979564008,
|
| 161 |
+
"sha256": "4eda4a1568a1a05fc762e7a3a2072c9d11cb860cdd763ac1720aef39f6a24751"
|
| 162 |
+
},
|
| 163 |
+
{
|
| 164 |
+
"name": "model-00017-of-00019.safetensors",
|
| 165 |
+
"size": 2108759344,
|
| 166 |
+
"sha256": "6cdc6d3b849808dfec6f0b7d11a09b74c0c91ba8df852891741e9eb63d33ab9f"
|
| 167 |
+
},
|
| 168 |
+
{
|
| 169 |
+
"name": "model-00018-of-00019.safetensors",
|
| 170 |
+
"size": 3392197360,
|
| 171 |
+
"sha256": "a408f74e0801216c76f34f404e487b7f72177fc34efab5d0f14edf7e7630345f"
|
| 172 |
+
},
|
| 173 |
+
{
|
| 174 |
+
"name": "model-00019-of-00019.safetensors",
|
| 175 |
+
"size": 6088313008,
|
| 176 |
+
"sha256": "48a75e0f87ae40e40ff7190e53576355ca4eea028a64882e302893287ab7b49f"
|
| 177 |
+
},
|
| 178 |
+
{
|
| 179 |
+
"name": "model-parallel-ffn.safetensors",
|
| 180 |
+
"size": 4529878968,
|
| 181 |
+
"sha256": "9ea1ead9b6ea1ebae0d791145e43c727cd5bd7cc8dee19d4c9d95c0eb1ff4f1c"
|
| 182 |
+
},
|
| 183 |
+
{
|
| 184 |
+
"name": "model.safetensors.index.json",
|
| 185 |
+
"size": 145141,
|
| 186 |
+
"sha256": null
|
| 187 |
+
},
|
| 188 |
+
{
|
| 189 |
+
"name": "modeling_agnes.py",
|
| 190 |
+
"size": 70576,
|
| 191 |
+
"sha256": null
|
| 192 |
+
},
|
| 193 |
+
{
|
| 194 |
+
"name": "preprocessor_config.json",
|
| 195 |
+
"size": 476,
|
| 196 |
+
"sha256": null
|
| 197 |
+
},
|
| 198 |
+
{
|
| 199 |
+
"name": "processing_agnes.py",
|
| 200 |
+
"size": 5849,
|
| 201 |
+
"sha256": null
|
| 202 |
+
},
|
| 203 |
+
{
|
| 204 |
+
"name": "serve.sh",
|
| 205 |
+
"size": 1448,
|
| 206 |
+
"sha256": null
|
| 207 |
+
},
|
| 208 |
+
{
|
| 209 |
+
"name": "sglang_patch/README.md",
|
| 210 |
+
"size": 2645,
|
| 211 |
+
"sha256": null
|
| 212 |
+
},
|
| 213 |
+
{
|
| 214 |
+
"name": "sglang_patch/agnes_sglang_config.py",
|
| 215 |
+
"size": 3174,
|
| 216 |
+
"sha256": null
|
| 217 |
+
},
|
| 218 |
+
{
|
| 219 |
+
"name": "sglang_patch/apply_patch.py",
|
| 220 |
+
"size": 4832,
|
| 221 |
+
"sha256": null
|
| 222 |
+
},
|
| 223 |
+
{
|
| 224 |
+
"name": "sglang_patch/nightly-dev-20260908-20ca564b/sglang/srt/configs/agnes.py",
|
| 225 |
+
"size": 3174,
|
| 226 |
+
"sha256": null
|
| 227 |
+
},
|
| 228 |
+
{
|
| 229 |
+
"name": "sglang_patch/nightly-dev-20260908-20ca564b/sglang/srt/models/qwen3_5.py",
|
| 230 |
+
"size": 110213,
|
| 231 |
+
"sha256": null
|
| 232 |
+
},
|
| 233 |
+
{
|
| 234 |
+
"name": "sglang_patch/nightly-dev-20260908-20ca564b/sglang/srt/utils/hf_transformers/common.py",
|
| 235 |
+
"size": 25576,
|
| 236 |
+
"sha256": null
|
| 237 |
+
},
|
| 238 |
+
{
|
| 239 |
+
"name": "sglang_patch/v0.5.19/sglang/srt/configs/agnes.py",
|
| 240 |
+
"size": 3174,
|
| 241 |
+
"sha256": null
|
| 242 |
+
},
|
| 243 |
+
{
|
| 244 |
+
"name": "sglang_patch/v0.5.19/sglang/srt/models/qwen3_5.py",
|
| 245 |
+
"size": 109665,
|
| 246 |
+
"sha256": null
|
| 247 |
+
},
|
| 248 |
+
{
|
| 249 |
+
"name": "sglang_patch/v0.5.19/sglang/srt/utils/hf_transformers/common.py",
|
| 250 |
+
"size": 23485,
|
| 251 |
+
"sha256": null
|
| 252 |
+
},
|
| 253 |
+
{
|
| 254 |
+
"name": "tokenizer.json",
|
| 255 |
+
"size": 9835269,
|
| 256 |
+
"sha256": null
|
| 257 |
+
},
|
| 258 |
+
{
|
| 259 |
+
"name": "tokenizer_config.json",
|
| 260 |
+
"size": 18884,
|
| 261 |
+
"sha256": null
|
| 262 |
+
},
|
| 263 |
+
{
|
| 264 |
+
"name": "video_preprocessor_config.json",
|
| 265 |
+
"size": 460,
|
| 266 |
+
"sha256": null
|
| 267 |
+
},
|
| 268 |
+
{
|
| 269 |
+
"name": "video_processing_agnes.py",
|
| 270 |
+
"size": 8832,
|
| 271 |
+
"sha256": null
|
| 272 |
+
},
|
| 273 |
+
{
|
| 274 |
+
"name": "vocab.json",
|
| 275 |
+
"size": 6722759,
|
| 276 |
+
"sha256": null
|
| 277 |
+
}
|
| 278 |
+
]
|
| 279 |
+
}
|