Instructions to use RWKV/RWKV7-G1j-1.5B-20260831 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use RWKV/RWKV7-G1j-1.5B-20260831 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="RWKV/RWKV7-G1j-1.5B-20260831", trust_remote_code=True) messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# Load model directly from transformers import AutoModelForCausalLM model = AutoModelForCausalLM.from_pretrained("RWKV/RWKV7-G1j-1.5B-20260831", trust_remote_code=True, device_map="auto") - RWKV
How to use RWKV/RWKV7-G1j-1.5B-20260831 with RWKV:
# No code snippets available yet for this library. # To use this model, check the repository files and the library's documentation. # Want to help? PRs adding snippets are welcome at: # https://github.com/huggingface/huggingface.js
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use RWKV/RWKV7-G1j-1.5B-20260831 with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "RWKV/RWKV7-G1j-1.5B-20260831" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "RWKV/RWKV7-G1j-1.5B-20260831", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/RWKV/RWKV7-G1j-1.5B-20260831
- SGLang
How to use RWKV/RWKV7-G1j-1.5B-20260831 with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "RWKV/RWKV7-G1j-1.5B-20260831" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "RWKV/RWKV7-G1j-1.5B-20260831", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "RWKV/RWKV7-G1j-1.5B-20260831" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "RWKV/RWKV7-G1j-1.5B-20260831", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use RWKV/RWKV7-G1j-1.5B-20260831 with Docker Model Runner:
docker model run hf.co/RWKV/RWKV7-G1j-1.5B-20260831
Use chunked WKV by default
Browse filesApply rwkv-publisher upstream commit 0e23b39775d6017638d0234d153bd263c411c229. The existing chunked prefill path becomes the default; weights remain unchanged.
- README.md +6 -0
- config.json +1 -1
- configuration_rwkv7.py +2 -2
- release-manifest.json +19 -18
README.md
CHANGED
|
@@ -132,6 +132,12 @@ model = AutoModelForCausalLM.from_pretrained(
|
|
| 132 |
The recurrent cache returned by the model can be passed back for incremental
|
| 133 |
decoding. Use an `attention_mask` for padded batches.
|
| 134 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 135 |
## Chat quickstart
|
| 136 |
|
| 137 |
```python
|
|
|
|
| 132 |
The recurrent cache returned by the model can be passed back for incremental
|
| 133 |
decoding. Use an `attention_mask` for padded batches.
|
| 134 |
|
| 135 |
+
The model defaults to the chunk-parallel WKV path for efficient multi-token
|
| 136 |
+
prefill. To reproduce the portable token-order reference path, set
|
| 137 |
+
`model.config.wkv_implementation = "eager"` before the first forward pass.
|
| 138 |
+
Chunked execution changes floating-point operation order, so small numerical
|
| 139 |
+
differences from eager execution are expected.
|
| 140 |
+
|
| 141 |
## Chat quickstart
|
| 142 |
|
| 143 |
```python
|
config.json
CHANGED
|
@@ -26,6 +26,6 @@
|
|
| 26 |
"use_cache": true,
|
| 27 |
"v_low_rank_dim": 64,
|
| 28 |
"vocab_size": 65536,
|
| 29 |
-
"wkv_implementation": "
|
| 30 |
"wkv_state_dtype": "float32"
|
| 31 |
}
|
|
|
|
| 26 |
"use_cache": true,
|
| 27 |
"v_low_rank_dim": 64,
|
| 28 |
"vocab_size": 65536,
|
| 29 |
+
"wkv_implementation": "chunked",
|
| 30 |
"wkv_state_dtype": "float32"
|
| 31 |
}
|
configuration_rwkv7.py
CHANGED
|
@@ -70,7 +70,7 @@ class Rwkv7Config(PreTrainedConfig):
|
|
| 70 |
the whole sequence, so a narrow state drifts; `"float32"` with fp16
|
| 71 |
activations is the combination the reference implementation uses.
|
| 72 |
`"float16"`/`"bfloat16"` trade that for a smaller state.
|
| 73 |
-
wkv_implementation (`str`, *optional*, defaults to `"
|
| 74 |
Which WKV recurrence to use, by name, from
|
| 75 |
`models.rwkv7.modeling_rwkv7.RWKV7_WKV_FUNCTIONS`. `"eager"` is the
|
| 76 |
exact portable PyTorch path and preserves the reference token order for
|
|
@@ -120,7 +120,7 @@ class Rwkv7Config(PreTrainedConfig):
|
|
| 120 |
tie_word_embeddings: bool = False
|
| 121 |
use_cache: bool = True
|
| 122 |
wkv_state_dtype: str = "float32"
|
| 123 |
-
wkv_implementation: str = "
|
| 124 |
bos_token_id: int | None = 0
|
| 125 |
eos_token_id: int | None = 0
|
| 126 |
pad_token_id: int | None = 0
|
|
|
|
| 70 |
the whole sequence, so a narrow state drifts; `"float32"` with fp16
|
| 71 |
activations is the combination the reference implementation uses.
|
| 72 |
`"float16"`/`"bfloat16"` trade that for a smaller state.
|
| 73 |
+
wkv_implementation (`str`, *optional*, defaults to `"chunked"`):
|
| 74 |
Which WKV recurrence to use, by name, from
|
| 75 |
`models.rwkv7.modeling_rwkv7.RWKV7_WKV_FUNCTIONS`. `"eager"` is the
|
| 76 |
exact portable PyTorch path and preserves the reference token order for
|
|
|
|
| 120 |
tie_word_embeddings: bool = False
|
| 121 |
use_cache: bool = True
|
| 122 |
wkv_state_dtype: str = "float32"
|
| 123 |
+
wkv_implementation: str = "chunked"
|
| 124 |
bos_token_id: int | None = 0
|
| 125 |
eos_token_id: int | None = 0
|
| 126 |
pad_token_id: int | None = 0
|
release-manifest.json
CHANGED
|
@@ -36,8 +36,8 @@
|
|
| 36 |
},
|
| 37 |
"README.md": {
|
| 38 |
"role": "model_card",
|
| 39 |
-
"sha256": "
|
| 40 |
-
"size_bytes":
|
| 41 |
},
|
| 42 |
"assets/rwkv-logo.webp": {
|
| 43 |
"role": "media",
|
|
@@ -51,13 +51,13 @@
|
|
| 51 |
},
|
| 52 |
"config.json": {
|
| 53 |
"role": "model_config",
|
| 54 |
-
"sha256": "
|
| 55 |
-
"size_bytes":
|
| 56 |
},
|
| 57 |
"configuration_rwkv7.py": {
|
| 58 |
"role": "model_code",
|
| 59 |
-
"sha256": "
|
| 60 |
-
"size_bytes":
|
| 61 |
},
|
| 62 |
"generation_config.json": {
|
| 63 |
"role": "model_config",
|
|
@@ -169,37 +169,38 @@
|
|
| 169 |
"provenance": "locked-profile"
|
| 170 |
},
|
| 171 |
"model_code": {
|
| 172 |
-
"format_version":
|
|
|
|
|
|
|
|
|
|
| 173 |
"patches": [
|
|
|
|
| 174 |
"model-neutral-configuration-docstring",
|
| 175 |
"layer-zero-value-residual-buffers",
|
| 176 |
"trainer-past-key-values-placeholder",
|
| 177 |
"trl-position-ids-packing-boundaries",
|
| 178 |
"labels-disable-recurrent-cache"
|
| 179 |
],
|
| 180 |
-
"source_repository": "https://github.com/huggingface/transformers.git",
|
| 181 |
-
"source_revision": "4ad9ed0747ed6ba75c787e8f9040dcd64b166ee2",
|
| 182 |
"sources": {
|
| 183 |
"configuration_rwkv7.py": {
|
| 184 |
"asset_path": "model_code/configuration_rwkv7.py",
|
| 185 |
-
"output_sha256": "ad5db5a9d335015e316ae5a4ff3913c48b592424513881c4379f2ad732116fd9",
|
| 186 |
"repository_path": "src/transformers/models/rwkv7/configuration_rwkv7.py",
|
| 187 |
-
"source_sha256": "6f5b92c5fe7498ad22b0054a2f735a7ca82e7577436f4ad32f0fc27d1e900fdd"
|
|
|
|
| 188 |
},
|
| 189 |
"modeling_rwkv7.py": {
|
| 190 |
"asset_path": "model_code/modeling_rwkv7.py",
|
| 191 |
-
"output_sha256": "8b8a3459b40a33424b592abff360485577a8f19419be411e3dceb817a51ebe46",
|
| 192 |
"repository_path": "src/transformers/models/rwkv7/modeling_rwkv7.py",
|
| 193 |
-
"source_sha256": "3e8e5af7c4eba0b5de1496aef44773d7ac1bb4d96756e6f55efaf29453d67952"
|
|
|
|
| 194 |
},
|
| 195 |
"tokenization_rwkv7.py": {
|
| 196 |
"asset_path": "model_code/tokenization_rwkv7.py",
|
| 197 |
-
"output_sha256": "20c29d0f8f27003889570f54a1cd53482dd608a034e87673f6e43cf5e7c8d173",
|
| 198 |
"repository_path": "rwkv-publisher/src/rwkv_publisher/assets/model_code/tokenization_rwkv7.py",
|
| 199 |
-
"source_sha256": "20c29d0f8f27003889570f54a1cd53482dd608a034e87673f6e43cf5e7c8d173"
|
|
|
|
| 200 |
}
|
| 201 |
-
}
|
| 202 |
-
"transformers_min_version": "5.15"
|
| 203 |
},
|
| 204 |
"profile": {
|
| 205 |
"checkpoint": "g1j-1.5b-20260831",
|
|
@@ -372,4 +373,4 @@
|
|
| 372 |
"sha256": "c43176881caf85fe22ad654ab02e7519260d560f3d20420ab590adb0c823860f",
|
| 373 |
"size_bytes": 3055444605
|
| 374 |
}
|
| 375 |
-
}
|
|
|
|
| 36 |
},
|
| 37 |
"README.md": {
|
| 38 |
"role": "model_card",
|
| 39 |
+
"sha256": "9e71f101f9d45fb457d197117bb0df15b8d6cb65e2acbbdf3cfa25c54141e36d",
|
| 40 |
+
"size_bytes": 12164
|
| 41 |
},
|
| 42 |
"assets/rwkv-logo.webp": {
|
| 43 |
"role": "media",
|
|
|
|
| 51 |
},
|
| 52 |
"config.json": {
|
| 53 |
"role": "model_config",
|
| 54 |
+
"sha256": "95f3524181116fca466ce5b2e8c827dcf1a026ebfb4967d3ad28af9feee96d1f",
|
| 55 |
+
"size_bytes": 781
|
| 56 |
},
|
| 57 |
"configuration_rwkv7.py": {
|
| 58 |
"role": "model_code",
|
| 59 |
+
"sha256": "14050c631e27d66cd2f52ca5ec53b62eef04e602c4f1fc4e3696c328bc8e200e",
|
| 60 |
+
"size_bytes": 7431
|
| 61 |
},
|
| 62 |
"generation_config.json": {
|
| 63 |
"role": "model_config",
|
|
|
|
| 169 |
"provenance": "locked-profile"
|
| 170 |
},
|
| 171 |
"model_code": {
|
| 172 |
+
"format_version": 5,
|
| 173 |
+
"source_repository": "https://github.com/huggingface/transformers.git",
|
| 174 |
+
"source_revision": "4ad9ed0747ed6ba75c787e8f9040dcd64b166ee2",
|
| 175 |
+
"transformers_min_version": "5.15",
|
| 176 |
"patches": [
|
| 177 |
+
"chunked-wkv-default",
|
| 178 |
"model-neutral-configuration-docstring",
|
| 179 |
"layer-zero-value-residual-buffers",
|
| 180 |
"trainer-past-key-values-placeholder",
|
| 181 |
"trl-position-ids-packing-boundaries",
|
| 182 |
"labels-disable-recurrent-cache"
|
| 183 |
],
|
|
|
|
|
|
|
| 184 |
"sources": {
|
| 185 |
"configuration_rwkv7.py": {
|
| 186 |
"asset_path": "model_code/configuration_rwkv7.py",
|
|
|
|
| 187 |
"repository_path": "src/transformers/models/rwkv7/configuration_rwkv7.py",
|
| 188 |
+
"source_sha256": "6f5b92c5fe7498ad22b0054a2f735a7ca82e7577436f4ad32f0fc27d1e900fdd",
|
| 189 |
+
"output_sha256": "76f832fa713fd7e117273a67a86225fe2faebc6d80a6757113484def3a5671b4"
|
| 190 |
},
|
| 191 |
"modeling_rwkv7.py": {
|
| 192 |
"asset_path": "model_code/modeling_rwkv7.py",
|
|
|
|
| 193 |
"repository_path": "src/transformers/models/rwkv7/modeling_rwkv7.py",
|
| 194 |
+
"source_sha256": "3e8e5af7c4eba0b5de1496aef44773d7ac1bb4d96756e6f55efaf29453d67952",
|
| 195 |
+
"output_sha256": "8b8a3459b40a33424b592abff360485577a8f19419be411e3dceb817a51ebe46"
|
| 196 |
},
|
| 197 |
"tokenization_rwkv7.py": {
|
| 198 |
"asset_path": "model_code/tokenization_rwkv7.py",
|
|
|
|
| 199 |
"repository_path": "rwkv-publisher/src/rwkv_publisher/assets/model_code/tokenization_rwkv7.py",
|
| 200 |
+
"source_sha256": "20c29d0f8f27003889570f54a1cd53482dd608a034e87673f6e43cf5e7c8d173",
|
| 201 |
+
"output_sha256": "20c29d0f8f27003889570f54a1cd53482dd608a034e87673f6e43cf5e7c8d173"
|
| 202 |
}
|
| 203 |
+
}
|
|
|
|
| 204 |
},
|
| 205 |
"profile": {
|
| 206 |
"checkpoint": "g1j-1.5b-20260831",
|
|
|
|
| 373 |
"sha256": "c43176881caf85fe22ad654ab02e7519260d560f3d20420ab590adb0c823860f",
|
| 374 |
"size_bytes": 3055444605
|
| 375 |
}
|
| 376 |
+
}
|