Ruurd commited on
Commit
e2402e0
·
verified ·
1 Parent(s): fd8178d

Deploy BYOD-Llama-3.1-8B full-precision demo

Browse files
app.py CHANGED
@@ -25,6 +25,8 @@ SESSION = load_hub_adapter_session(
25
  MODEL_REPO_ID,
26
  device_name="cuda",
27
  quantization="none",
 
 
28
  )
29
  print(f"Loaded {DISPLAY_NAME} ({SESSION.compute_dtype}, unquantized).")
30
 
 
25
  MODEL_REPO_ID,
26
  device_name="cuda",
27
  quantization="none",
28
+ # A forward pass is only valid after @spaces.GPU has allocated hardware.
29
+ preflight=False,
30
  )
31
  print(f"Loaded {DISPLAY_NAME} ({SESSION.compute_dtype}, unquantized).")
32
 
src/diffusion_lm/.DS_Store CHANGED
Binary files a/src/diffusion_lm/.DS_Store and b/src/diffusion_lm/.DS_Store differ
 
src/diffusion_lm/inference.py CHANGED
@@ -82,6 +82,7 @@ def _load_adapter_path(
82
  adapter_path: str | Path,
83
  device_name: str = "auto",
84
  quantization: str | None = None,
 
85
  ) -> InferenceSession:
86
  """Load an adapter directory that has already been resolved and validated."""
87
  adapter_path = Path(adapter_path).expanduser().resolve()
@@ -166,7 +167,8 @@ def _load_adapter_path(
166
  model.eval()
167
  mask_info = validate_mask_token(tokenizer, str(run_config.get("mask_token", "MASK")))
168
  session = InferenceSession(model, tokenizer, device, adapter_path, run_config, mask_info["mask_token_id"], "4bit" if use_4bit else "none", str(compute_dtype).removeprefix("torch."))
169
- preflight_session(session)
 
170
  return session
171
 
172
 
@@ -182,6 +184,7 @@ def load_hub_adapter_session(
182
  quantization: str | None = None,
183
  revision: str | None = None,
184
  cache_dir: str | Path | None = None,
 
185
  ) -> InferenceSession:
186
  """Download and load a BYOD adapter from the Hugging Face Hub."""
187
  try:
@@ -197,7 +200,7 @@ def load_hub_adapter_session(
197
  cache_dir=str(cache_dir) if cache_dir is not None else None,
198
  token=os.getenv("HF_TOKEN"),
199
  )
200
- return _load_adapter_path(adapter_path, device_name, quantization)
201
 
202
 
203
  def load_merged_session(
 
82
  adapter_path: str | Path,
83
  device_name: str = "auto",
84
  quantization: str | None = None,
85
+ preflight: bool = True,
86
  ) -> InferenceSession:
87
  """Load an adapter directory that has already been resolved and validated."""
88
  adapter_path = Path(adapter_path).expanduser().resolve()
 
167
  model.eval()
168
  mask_info = validate_mask_token(tokenizer, str(run_config.get("mask_token", "MASK")))
169
  session = InferenceSession(model, tokenizer, device, adapter_path, run_config, mask_info["mask_token_id"], "4bit" if use_4bit else "none", str(compute_dtype).removeprefix("torch."))
170
+ if preflight:
171
+ preflight_session(session)
172
  return session
173
 
174
 
 
184
  quantization: str | None = None,
185
  revision: str | None = None,
186
  cache_dir: str | Path | None = None,
187
+ preflight: bool = True,
188
  ) -> InferenceSession:
189
  """Download and load a BYOD adapter from the Hugging Face Hub."""
190
  try:
 
200
  cache_dir=str(cache_dir) if cache_dir is not None else None,
201
  token=os.getenv("HF_TOKEN"),
202
  )
203
+ return _load_adapter_path(adapter_path, device_name, quantization, preflight=preflight)
204
 
205
 
206
  def load_merged_session(