Text Generation
Transformers
Safetensors
lfm2
liquid
lfm2.5
edge
parallel-constrained-decoding
structured-generation
classification
inference-only
modal
conversational
Instructions to use monotykamary/LFM2.5-2.6B-RLCD with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use monotykamary/LFM2.5-2.6B-RLCD with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="monotykamary/LFM2.5-2.6B-RLCD") messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# pip install -U transformers accelerate # Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("monotykamary/LFM2.5-2.6B-RLCD") model = AutoModelForCausalLM.from_pretrained("monotykamary/LFM2.5-2.6B-RLCD", device_map="auto") messages = [ {"role": "user", "content": "Who are you?"}, ] inputs = tokenizer.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=256) print(tokenizer.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use monotykamary/LFM2.5-2.6B-RLCD with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "monotykamary/LFM2.5-2.6B-RLCD" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "monotykamary/LFM2.5-2.6B-RLCD", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/monotykamary/LFM2.5-2.6B-RLCD
- SGLang
How to use monotykamary/LFM2.5-2.6B-RLCD with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "monotykamary/LFM2.5-2.6B-RLCD" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "monotykamary/LFM2.5-2.6B-RLCD", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "monotykamary/LFM2.5-2.6B-RLCD" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "monotykamary/LFM2.5-2.6B-RLCD", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use monotykamary/LFM2.5-2.6B-RLCD with Docker Model Runner:
docker model run hf.co/monotykamary/LFM2.5-2.6B-RLCD
Download pcd/tasks.py from monotykamary/LFM2.5-2.6B-RLCD: direct link, hf CLI and curl.
- Browser
- Download file 6.71 kB
-
https://huggingface.co/monotykamary/LFM2.5-2.6B-RLCD/resolve/main/pcd/tasks.py
- Command line
-
hf download hf://monotykamary/LFM2.5-2.6B-RLCD/pcd/tasks.py
-
curl -L -o tasks.py https://huggingface.co/monotykamary/LFM2.5-2.6B-RLCD/resolve/main/pcd/tasks.py
6.71 kB
| """Fixed hand-authored diagnostics, not representative production benchmarks.""" | |
| def object_schema(properties): | |
| return { | |
| "type": "object", | |
| "properties": properties, | |
| "required": list(properties), | |
| "additionalProperties": False, | |
| } | |
| SUPPORT = object_schema( | |
| { | |
| "topic": { | |
| "type": "string", | |
| "enum": ["billing", "technical", "shipping"], | |
| "description": "Main issue", | |
| }, | |
| "urgent": { | |
| "type": "boolean", | |
| "description": "True only if immediate action is explicitly requested", | |
| }, | |
| "refund": { | |
| "type": "boolean", | |
| "description": "Whether the customer explicitly requests a refund", | |
| }, | |
| } | |
| ) | |
| ROUTING = object_schema( | |
| { | |
| "route": { | |
| "type": "string", | |
| "enum": ["north", "north west", "south"], | |
| "description": "Requested route", | |
| }, | |
| "service": { | |
| "type": "string", | |
| "enum": ["express", "express plus", "standard"], | |
| "description": "Requested service level", | |
| }, | |
| "insured": {"type": "boolean", "description": "Whether insurance is requested"}, | |
| } | |
| ) | |
| SENTIMENT = object_schema( | |
| { | |
| "sentiment": { | |
| "type": "string", | |
| "enum": ["positive", "negative", "neutral"], | |
| "description": "Sentiment of the message", | |
| }, | |
| "language": { | |
| "type": "string", | |
| "enum": ["English", "French", "Spanish"], | |
| "description": "Language used", | |
| }, | |
| "question": { | |
| "type": "boolean", | |
| "description": "Whether the message asks a question", | |
| }, | |
| } | |
| ) | |
| def diagnostic(): | |
| groups = [ | |
| ( | |
| "support", | |
| SUPPORT, | |
| [ | |
| ( | |
| "I was charged twice. Please refund the duplicate charge. No hurry.", | |
| ["billing", False, True], | |
| ), | |
| ( | |
| "The server is down. Please fix it immediately. I do not want a refund.", | |
| ["technical", True, False], | |
| ), | |
| ( | |
| "Where is my parcel? A status update next week is fine.", | |
| ["shipping", False, False], | |
| ), | |
| ( | |
| "My delivery never arrived. Refund me today, this is urgent.", | |
| ["shipping", True, True], | |
| ), | |
| ], | |
| ), | |
| ( | |
| "routing", | |
| ROUTING, | |
| [ | |
| ( | |
| "Route north west using express plus. Include insurance.", | |
| ["north west", "express plus", True], | |
| ), | |
| ( | |
| "Send south with standard service, without insurance.", | |
| ["south", "standard", False], | |
| ), | |
| ( | |
| "Use north and express, not express plus. No insurance.", | |
| ["north", "express", False], | |
| ), | |
| ( | |
| "Route north west. Standard service. Insurance is required.", | |
| ["north west", "standard", True], | |
| ), | |
| ], | |
| ), | |
| ( | |
| "sentiment", | |
| SENTIMENT, | |
| [ | |
| ( | |
| "I love this product. It works wonderfully!", | |
| ["positive", "English", False], | |
| ), | |
| ( | |
| "Ce produit est horrible. Pouvez-vous le remplacer ?", | |
| ["negative", "French", True], | |
| ), | |
| ("¿Cuál es el horario de apertura?", ["neutral", "Spanish", True]), | |
| ( | |
| "The office opens at nine and closes at five.", | |
| ["neutral", "English", False], | |
| ), | |
| ], | |
| ), | |
| ] | |
| return [ | |
| { | |
| "id": f"{group}-{i}", | |
| "schema": schema, | |
| "context": context, | |
| "expected": dict(zip(schema["properties"], values)), | |
| } | |
| for group, schema, rows in groups | |
| for i, (context, values) in enumerate(rows) | |
| ] | |
| def audit(): | |
| """Small post-freeze audit; never use these labels to select the prompt.""" | |
| cases = [ | |
| ( | |
| SUPPORT, | |
| "Could you explain the monthly subscription price? This isn't urgent, and I'm not asking for money back.", | |
| ["billing", False, False], | |
| ), | |
| ( | |
| SUPPORT, | |
| "Please refund my broken software purchase right now; it crashes whenever I open it.", | |
| ["technical", True, True], | |
| ), | |
| ( | |
| ROUTING, | |
| "Use standard delivery to the north. Add insurance.", | |
| ["north", "standard", True], | |
| ), | |
| ( | |
| ROUTING, | |
| "Express plus to the south, insurance declined.", | |
| ["south", "express plus", False], | |
| ), | |
| (SENTIMENT, "¡Excelente servicio, muchas gracias!", ["positive", "Spanish", False]), | |
| (SENTIMENT, "Le train part à huit heures.", ["neutral", "French", False]), | |
| ] | |
| return [ | |
| { | |
| "id": f"audit-{i}", | |
| "schema": schema, | |
| "context": context, | |
| "expected": dict(zip(schema["properties"], values)), | |
| } | |
| for i, (schema, context, values) in enumerate(cases) | |
| ] | |
| def stress(): | |
| cases = [] | |
| for n in (4, 12, 28): | |
| properties = { | |
| f"flag_{i:02}": { | |
| "type": "boolean", | |
| "description": f"Value of flag_{i:02}, true iff enabled", | |
| } | |
| for i in range(n) | |
| } | |
| expected = {name: i % 3 != 0 for i, name in enumerate(properties)} | |
| context = "Configuration report.\n" + "\n".join( | |
| f"{key}: {'enabled' if value else 'disabled'}" for key, value in expected.items() | |
| ) | |
| cases.append( | |
| { | |
| "id": f"fields-{n}", | |
| "schema": object_schema(properties), | |
| "context": context, | |
| "expected": expected, | |
| } | |
| ) | |
| for count in (64, 255): | |
| choices = [f"category-{i:03}" for i in range(count)] | |
| cases.append( | |
| { | |
| "id": f"enum-{count}", | |
| "schema": object_schema({"category": {"type": "string", "enum": choices}}), | |
| "context": f"The assigned category is category-{count - 7:03}. Return that exact category.", | |
| "expected": {"category": f"category-{count - 7:03}"}, | |
| } | |
| ) | |
| return cases | |