Any-to-Any
Transformers
ONNX
Safetensors
English
Chinese
multimodal
audio
video
speech
streaming
full-duplex
long-video
custom-code
Instructions to use inclusionAI/Realtime-Venus with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use inclusionAI/Realtime-Venus with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("inclusionAI/Realtime-Venus", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download Realtime-Venus-Audio/special_tokens_map.json from inclusionAI/Realtime-Venus: direct link, hf CLI and curl.
- Browser
- Download file 2.33 kB
-
https://huggingface.co/inclusionAI/Realtime-Venus/resolve/main/Realtime-Venus-Audio/special_tokens_map.json
- Command line
-
hf download hf://inclusionAI/Realtime-Venus/Realtime-Venus-Audio/special_tokens_map.json
-
curl -L -o special_tokens_map.json https://huggingface.co/inclusionAI/Realtime-Venus/resolve/main/Realtime-Venus-Audio/special_tokens_map.json
2.33 kB
| { | |
| "additional_special_tokens": [ | |
| "<unk>", | |
| "<image>", | |
| "</image>", | |
| "<ref>", | |
| "</ref>", | |
| "<box>", | |
| "</box>", | |
| "<quad>", | |
| "</quad>", | |
| "<point>", | |
| "</point>", | |
| "<slice>", | |
| "</slice>", | |
| "<image_id>", | |
| "</image_id>", | |
| "<unit>", | |
| "</unit>", | |
| "<answer>", | |
| "</answer>", | |
| "<focus>", | |
| "</focus>", | |
| "<line>", | |
| "</line>", | |
| "<perception>", | |
| "</perception>", | |
| "<source_image>", | |
| "</source_image>", | |
| "<image_save_to>", | |
| "</image_save_to>", | |
| "<|audio_start|>", | |
| "<|audio|>", | |
| "<|audio_end|>", | |
| "<|spk_bos|>", | |
| "<|spk|>", | |
| "<|spk_eos|>", | |
| "<|tts_bos|>", | |
| "<|tts_eos|>", | |
| "<|listen|>", | |
| "<|speak|>", | |
| "<|interrupt|>", | |
| "<|vad_start|>", | |
| "<|vad_end|>", | |
| "<|emotion_start|>", | |
| "<|emotion_end|>", | |
| "<|speed_start|>", | |
| "<|speed_end|>", | |
| "<|pitch_start|>", | |
| "<|pitch_end|>", | |
| "<|chunk_eos|>", | |
| "<|chunk_bos|>", | |
| "<|chunk_tts_bos|>", | |
| "<|chunk_tts_eos|>", | |
| "<|tts_pad|>", | |
| "<|timbre_7|>", | |
| "<|timbre_8|>", | |
| "<|timbre_9|>", | |
| "<|timbre_10|>", | |
| "<|timbre_11|>", | |
| "<|timbre_12|>", | |
| "<|timbre_13|>", | |
| "<|timbre_14|>", | |
| "<|timbre_15|>", | |
| "<|timbre_16|>", | |
| "<|timbre_17|>", | |
| "<|timbre_18|>", | |
| "<|timbre_19|>", | |
| "<|timbre_20|>", | |
| "<|timbre_21|>", | |
| "<|timbre_22|>", | |
| "<|timbre_23|>", | |
| "<|timbre_24|>", | |
| "<|timbre_25|>", | |
| "<|timbre_26|>", | |
| "<|timbre_27|>", | |
| "<|timbre_28|>", | |
| "<|timbre_29|>", | |
| "<|timbre_30|>", | |
| "<|timbre_31|>", | |
| "<delegate>", | |
| "</delegate>", | |
| "<backend>", | |
| "</backend>", | |
| { | |
| "content": "<|im_end|>", | |
| "lstrip": false, | |
| "normalized": false, | |
| "rstrip": false, | |
| "single_word": false | |
| } | |
| ], | |
| "bos_token": { | |
| "content": "<|im_start|>", | |
| "lstrip": false, | |
| "normalized": false, | |
| "rstrip": false, | |
| "single_word": false | |
| }, | |
| "eos_token": { | |
| "content": "<|im_end|>", | |
| "lstrip": false, | |
| "normalized": false, | |
| "rstrip": false, | |
| "single_word": false | |
| }, | |
| "pad_token": { | |
| "content": "<|endoftext|>", | |
| "lstrip": false, | |
| "normalized": false, | |
| "rstrip": false, | |
| "single_word": false | |
| }, | |
| "unk_token": { | |
| "content": "<unk>", | |
| "lstrip": false, | |
| "normalized": false, | |
| "rstrip": false, | |
| "single_word": false | |
| } | |
| } | |