Spaces:
Running on Zero
Running on Zero
Deploy BYOD-Llama-3.1-8B full-precision demo
Browse files- app.py +2 -0
- src/diffusion_lm/.DS_Store +0 -0
- src/diffusion_lm/inference.py +5 -2
app.py
CHANGED
|
@@ -25,6 +25,8 @@ SESSION = load_hub_adapter_session(
|
|
| 25 |
MODEL_REPO_ID,
|
| 26 |
device_name="cuda",
|
| 27 |
quantization="none",
|
|
|
|
|
|
|
| 28 |
)
|
| 29 |
print(f"Loaded {DISPLAY_NAME} ({SESSION.compute_dtype}, unquantized).")
|
| 30 |
|
|
|
|
| 25 |
MODEL_REPO_ID,
|
| 26 |
device_name="cuda",
|
| 27 |
quantization="none",
|
| 28 |
+
# A forward pass is only valid after @spaces.GPU has allocated hardware.
|
| 29 |
+
preflight=False,
|
| 30 |
)
|
| 31 |
print(f"Loaded {DISPLAY_NAME} ({SESSION.compute_dtype}, unquantized).")
|
| 32 |
|
src/diffusion_lm/.DS_Store
CHANGED
|
Binary files a/src/diffusion_lm/.DS_Store and b/src/diffusion_lm/.DS_Store differ
|
|
|
src/diffusion_lm/inference.py
CHANGED
|
@@ -82,6 +82,7 @@ def _load_adapter_path(
|
|
| 82 |
adapter_path: str | Path,
|
| 83 |
device_name: str = "auto",
|
| 84 |
quantization: str | None = None,
|
|
|
|
| 85 |
) -> InferenceSession:
|
| 86 |
"""Load an adapter directory that has already been resolved and validated."""
|
| 87 |
adapter_path = Path(adapter_path).expanduser().resolve()
|
|
@@ -166,7 +167,8 @@ def _load_adapter_path(
|
|
| 166 |
model.eval()
|
| 167 |
mask_info = validate_mask_token(tokenizer, str(run_config.get("mask_token", "MASK")))
|
| 168 |
session = InferenceSession(model, tokenizer, device, adapter_path, run_config, mask_info["mask_token_id"], "4bit" if use_4bit else "none", str(compute_dtype).removeprefix("torch."))
|
| 169 |
-
|
|
|
|
| 170 |
return session
|
| 171 |
|
| 172 |
|
|
@@ -182,6 +184,7 @@ def load_hub_adapter_session(
|
|
| 182 |
quantization: str | None = None,
|
| 183 |
revision: str | None = None,
|
| 184 |
cache_dir: str | Path | None = None,
|
|
|
|
| 185 |
) -> InferenceSession:
|
| 186 |
"""Download and load a BYOD adapter from the Hugging Face Hub."""
|
| 187 |
try:
|
|
@@ -197,7 +200,7 @@ def load_hub_adapter_session(
|
|
| 197 |
cache_dir=str(cache_dir) if cache_dir is not None else None,
|
| 198 |
token=os.getenv("HF_TOKEN"),
|
| 199 |
)
|
| 200 |
-
return _load_adapter_path(adapter_path, device_name, quantization)
|
| 201 |
|
| 202 |
|
| 203 |
def load_merged_session(
|
|
|
|
| 82 |
adapter_path: str | Path,
|
| 83 |
device_name: str = "auto",
|
| 84 |
quantization: str | None = None,
|
| 85 |
+
preflight: bool = True,
|
| 86 |
) -> InferenceSession:
|
| 87 |
"""Load an adapter directory that has already been resolved and validated."""
|
| 88 |
adapter_path = Path(adapter_path).expanduser().resolve()
|
|
|
|
| 167 |
model.eval()
|
| 168 |
mask_info = validate_mask_token(tokenizer, str(run_config.get("mask_token", "MASK")))
|
| 169 |
session = InferenceSession(model, tokenizer, device, adapter_path, run_config, mask_info["mask_token_id"], "4bit" if use_4bit else "none", str(compute_dtype).removeprefix("torch."))
|
| 170 |
+
if preflight:
|
| 171 |
+
preflight_session(session)
|
| 172 |
return session
|
| 173 |
|
| 174 |
|
|
|
|
| 184 |
quantization: str | None = None,
|
| 185 |
revision: str | None = None,
|
| 186 |
cache_dir: str | Path | None = None,
|
| 187 |
+
preflight: bool = True,
|
| 188 |
) -> InferenceSession:
|
| 189 |
"""Download and load a BYOD adapter from the Hugging Face Hub."""
|
| 190 |
try:
|
|
|
|
| 200 |
cache_dir=str(cache_dir) if cache_dir is not None else None,
|
| 201 |
token=os.getenv("HF_TOKEN"),
|
| 202 |
)
|
| 203 |
+
return _load_adapter_path(adapter_path, device_name, quantization, preflight=preflight)
|
| 204 |
|
| 205 |
|
| 206 |
def load_merged_session(
|