Text Generation
Transformers
Safetensors
lfm2
liquid
lfm2.5
edge
parallel-constrained-decoding
structured-generation
classification
inference-only
modal
conversational
Instructions to use monotykamary/LFM2.5-2.6B-RLCD with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use monotykamary/LFM2.5-2.6B-RLCD with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="monotykamary/LFM2.5-2.6B-RLCD") messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# pip install -U transformers accelerate # Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("monotykamary/LFM2.5-2.6B-RLCD") model = AutoModelForCausalLM.from_pretrained("monotykamary/LFM2.5-2.6B-RLCD", device_map="auto") messages = [ {"role": "user", "content": "Who are you?"}, ] inputs = tokenizer.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=256) print(tokenizer.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use monotykamary/LFM2.5-2.6B-RLCD with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "monotykamary/LFM2.5-2.6B-RLCD" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "monotykamary/LFM2.5-2.6B-RLCD", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/monotykamary/LFM2.5-2.6B-RLCD
- SGLang
How to use monotykamary/LFM2.5-2.6B-RLCD with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "monotykamary/LFM2.5-2.6B-RLCD" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "monotykamary/LFM2.5-2.6B-RLCD", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "monotykamary/LFM2.5-2.6B-RLCD" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "monotykamary/LFM2.5-2.6B-RLCD", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use monotykamary/LFM2.5-2.6B-RLCD with Docker Model Runner:
docker model run hf.co/monotykamary/LFM2.5-2.6B-RLCD
Download lfm25_pcd_modal.py from monotykamary/LFM2.5-2.6B-RLCD: direct link, hf CLI and curl.
- Browser
- Download file 5.15 kB
-
https://huggingface.co/monotykamary/LFM2.5-2.6B-RLCD/resolve/main/lfm25_pcd_modal.py
- Command line
-
hf download hf://monotykamary/LFM2.5-2.6B-RLCD/lfm25_pcd_modal.py
-
curl -L -o lfm25_pcd_modal.py https://huggingface.co/monotykamary/LFM2.5-2.6B-RLCD/resolve/main/lfm25_pcd_modal.py
5.15 kB
| """Frugal PCD validation/benchmarking and optional authenticated serving on Modal. | |
| modal run lfm25_pcd_modal.py --task prepare | |
| modal run lfm25_pcd_modal.py --task validate | |
| modal run lfm25_pcd_modal.py --task benchmark | |
| Publication is a separate, explicit --task publish operation. | |
| """ | |
| import json | |
| from pathlib import Path | |
| import modal | |
| app = modal.App("lfm25-26b-pcd") | |
| root = Path(__file__).parent | |
| cache = modal.Volume.from_name("huggingface-cache", create_if_missing=True) | |
| results = modal.Volume.from_name("lfm25-26b-pcd-results", create_if_missing=True) | |
| base = modal.Image.debian_slim(python_version="3.11").env( | |
| { | |
| "HF_HOME": "/root/.cache/huggingface", | |
| "TOKENIZERS_PARALLELISM": "false", | |
| "HF_HUB_DISABLE_TELEMETRY": "1", | |
| "OMP_NUM_THREADS": "4", | |
| } | |
| ) | |
| cpu_image = base.uv_pip_install("huggingface-hub==1.31.0", "pyyaml==6.0.3").add_local_python_source( | |
| "pcd" | |
| ) | |
| gpu_image = ( | |
| base.pip_install_from_requirements(str(root / "requirements-pcd.txt")) | |
| .add_local_python_source("pcd") | |
| .add_local_dir(root / "tests" / "pcd", "/root/tests/pcd") | |
| ) | |
| def prepare(): | |
| from pcd.publishing import prepare as prepare_snapshot | |
| info = prepare_snapshot() | |
| cache.commit() | |
| return info | |
| class PCDModel: | |
| precision: str = modal.parameter(default="float16") | |
| def load(self): | |
| import torch | |
| from pcd.engine import Engine | |
| if self.precision not in {"float16", "bfloat16", "float32"}: | |
| raise ValueError("unsupported precision") | |
| torch.set_num_threads(4) | |
| torch.set_float32_matmul_precision("highest") | |
| self.engine = Engine( | |
| device="cuda", dtype=self.precision, attention="sdpa", local_files_only=True | |
| ) | |
| def run(self, task: str, suite: str = "diagnostic", repeats: int = 3): | |
| from pcd.benchmark import execute | |
| report = execute(self.engine, task=task, suite=suite, repeats=repeats) | |
| path = Path("/results") / (report["run_id"] + ".json") | |
| path.write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n") | |
| results.commit() | |
| return report | |
| def infer(self, context: str, schema: dict, mode: str = "token"): | |
| return self.engine.constrained(context, schema, mode=mode) | |
| def extract(self, payload: dict): | |
| import jsonschema | |
| from fastapi import HTTPException | |
| try: | |
| if set(payload) - {"context", "schema", "mode"}: | |
| raise ValueError("unsupported request keys") | |
| return self.engine.constrained( | |
| payload["context"], payload["schema"], mode=payload.get("mode", "token") | |
| ) | |
| except (ValueError, KeyError, TypeError, jsonschema.SchemaError) as exc: | |
| raise HTTPException(status_code=422, detail=str(exc)) from None | |
| release_dir = root / "release" / "pcd" | |
| publish_image = ( | |
| cpu_image.add_local_dir(release_dir, "/release-src") if release_dir.is_dir() else cpu_image | |
| ) | |
| def publish(public: bool = False): | |
| from pcd.publishing import publish as publish_bundle | |
| result = publish_bundle(public=public) | |
| results.commit() | |
| return result | |
| def main( | |
| task: str = "validate", | |
| suite: str = "diagnostic", | |
| repeats: int = 3, | |
| precision: str = "float16", | |
| public: bool = False, | |
| ): | |
| if task not in {"prepare", "validate", "benchmark", "probe", "publish"}: | |
| raise ValueError("task must be prepare, validate, benchmark, probe, or publish") | |
| if repeats < 1 or repeats > 10: | |
| raise ValueError("repeats must be between 1 and 10") | |
| if public and task != "publish": | |
| raise ValueError("--public only applies to --task publish") | |
| if task == "publish": | |
| if not release_dir.is_dir(): | |
| raise ValueError("stage the release with scripts/build_pcd_release.py first") | |
| report = publish.remote(public) | |
| elif task == "prepare": | |
| report = prepare.remote() | |
| else: | |
| report = PCDModel(precision=precision).run.remote(task, suite, repeats) | |
| output = root / "results" / "pcd" | |
| output.mkdir(parents=True, exist_ok=True) | |
| path = output / (report.get("run_id", task) + ".json") | |
| path.write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n") | |
| print(json.dumps({"report": str(path), "summary": report.get("summary", report)}, indent=2)) | |