Harryji168 commited on
Commit
1b58237
·
verified ·
1 Parent(s): 522a25d

Deploy UniVideo Studio - app.py

Browse files
Files changed (1) hide show
  1. app.py +237 -0
app.py ADDED
@@ -0,0 +1,237 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import gradio as gr
2
+ import os
3
+ import requests
4
+ from PIL import Image
5
+ import io
6
+
7
+ # Note: UniVideo requires significant GPU resources
8
+ # This Space provides a UI that would connect to the Inference API
9
+ # For local deployment, you'd need to follow the full installation from:
10
+ # https://github.com/KlingTeam/UniVideo
11
+
12
+ HF_TOKEN = os.getenv("HF_TOKEN", "")
13
+ MODEL_ID = "KlingTeam/UniVideo"
14
+
15
+ def call_inference_api(task, **kwargs):
16
+ """
17
+ Call the Hugging Face Inference API for UniVideo
18
+ Note: As of now, UniVideo may not be available via standard Inference API
19
+ This is a template for when it becomes available
20
+ """
21
+ api_url = f"https://api-inference.huggingface.co/models/{MODEL_ID}"
22
+ headers = {"Authorization": f"Bearer {HF_TOKEN}"} if HF_TOKEN else {}
23
+
24
+ try:
25
+ response = requests.post(api_url, headers=headers, json=kwargs, timeout=120)
26
+ if response.status_code == 200:
27
+ return response.content
28
+ else:
29
+ return None, f"API Error: {response.status_code} - {response.text}"
30
+ except Exception as e:
31
+ return None, f"Error: {str(e)}"
32
+
33
+ def text_to_video(prompt, duration=5):
34
+ """Generate video from text prompt"""
35
+ return None, """
36
+ ⚠️ **UniVideo requires GPU resources**
37
+
38
+ This model is not yet available via the standard Inference API.
39
+
40
+ **To use UniVideo:**
41
+ 1. Clone the repository: https://github.com/KlingTeam/UniVideo
42
+ 2. Follow installation instructions
43
+ 3. Download the checkpoint from: https://huggingface.co/KlingTeam/UniVideo
44
+ 4. Run locally with GPU (recommended: A100 or H100)
45
+
46
+ **Alternative:** Upgrade this Space to GPU hardware and implement the full pipeline.
47
+
48
+ Your prompt: "{}"
49
+ """.format(prompt)
50
+
51
+ def image_to_video(image, prompt):
52
+ """Animate image to video"""
53
+ if image is None:
54
+ return None, "Please upload an image"
55
+
56
+ return None, """
57
+ ⚠️ **UniVideo Image-to-Video requires GPU**
58
+
59
+ This feature requires running the full UniVideo model locally.
60
+
61
+ **Quick Start:**
62
+ ```bash
63
+ cd univideo
64
+ python univideo_inference.py --task i2v --config configs/univideo_qwen2p5vl7b_hidden_hunyuanvideo.yaml
65
+ ```
66
+
67
+ Your prompt: "{}"
68
+ """.format(prompt)
69
+
70
+ def video_understanding(video):
71
+ """Understand and describe video content"""
72
+ if video is None:
73
+ return "Please upload a video"
74
+
75
+ return """
76
+ ⚠️ **UniVideo Understanding requires GPU**
77
+
78
+ To analyze videos with UniVideo:
79
+ ```bash
80
+ python univideo_inference.py --task understanding --config configs/univideo_qwen2p5vl7b_hidden_hunyuanvideo.yaml
81
+ ```
82
+ """
83
+
84
+ def video_editing(video, instruction):
85
+ """Edit video based on text instruction"""
86
+ if video is None:
87
+ return None, "Please upload a video"
88
+
89
+ return None, f"""
90
+ ⚠️ **UniVideo Video Editing requires GPU**
91
+
92
+ To edit videos with UniVideo:
93
+ ```bash
94
+ python univideo_inference.py --task v2v_edit --config configs/univideo_qwen2p5vl7b_hidden_hunyuanvideo.yaml
95
+ ```
96
+
97
+ Your instruction: "{instruction}"
98
+ """
99
+
100
+ # Gradio Interface
101
+ with gr.Blocks(theme=gr.themes.Soft(), title="UniVideo Studio") as demo:
102
+ gr.Markdown("# 🎬 UniVideo Studio")
103
+ gr.Markdown("**Unified Understanding, Generation, and Editing for Videos**")
104
+ gr.Markdown("⚠️ *Note: This Space requires GPU hardware. Currently showing setup instructions.*")
105
+
106
+ with gr.Tabs():
107
+ # Text-to-Video Tab
108
+ with gr.Tab("📝 Text-to-Video"):
109
+ with gr.Row():
110
+ with gr.Column():
111
+ t2v_prompt = gr.Textbox(
112
+ label="Prompt",
113
+ placeholder="Describe the video you want to generate...",
114
+ lines=3
115
+ )
116
+ t2v_duration = gr.Slider(3, 10, value=5, step=1, label="Duration (seconds)")
117
+ t2v_btn = gr.Button("Generate Video", variant="primary")
118
+
119
+ with gr.Column():
120
+ t2v_output = gr.Video(label="Generated Video")
121
+ t2v_status = gr.Textbox(label="Status", lines=10)
122
+
123
+ t2v_btn.click(
124
+ fn=text_to_video,
125
+ inputs=[t2v_prompt, t2v_duration],
126
+ outputs=[t2v_output, t2v_status]
127
+ )
128
+
129
+ # Image-to-Video Tab
130
+ with gr.Tab("🖼️ Image-to-Video"):
131
+ with gr.Row():
132
+ with gr.Column():
133
+ i2v_image = gr.Image(label="Input Image", type="pil")
134
+ i2v_prompt = gr.Textbox(
135
+ label="Motion Description",
136
+ placeholder="Describe how the image should animate...",
137
+ lines=2
138
+ )
139
+ i2v_btn = gr.Button("Animate Image", variant="primary")
140
+
141
+ with gr.Column():
142
+ i2v_output = gr.Video(label="Generated Video")
143
+ i2v_status = gr.Textbox(label="Status", lines=10)
144
+
145
+ i2v_btn.click(
146
+ fn=image_to_video,
147
+ inputs=[i2v_image, i2v_prompt],
148
+ outputs=[i2v_output, i2v_status]
149
+ )
150
+
151
+ # Video Understanding Tab
152
+ with gr.Tab("🧠 Video Understanding"):
153
+ with gr.Row():
154
+ with gr.Column():
155
+ vu_video = gr.Video(label="Input Video")
156
+ vu_btn = gr.Button("Analyze Video", variant="primary")
157
+
158
+ with gr.Column():
159
+ vu_output = gr.Textbox(label="Analysis", lines=15)
160
+
161
+ vu_btn.click(
162
+ fn=video_understanding,
163
+ inputs=[vu_video],
164
+ outputs=[vu_output]
165
+ )
166
+
167
+ # Video Editing Tab
168
+ with gr.Tab("✂️ Video Editing"):
169
+ with gr.Row():
170
+ with gr.Column():
171
+ ve_video = gr.Video(label="Input Video")
172
+ ve_instruction = gr.Textbox(
173
+ label="Editing Instruction",
174
+ placeholder="Describe how to edit the video...",
175
+ lines=2
176
+ )
177
+ ve_btn = gr.Button("Edit Video", variant="primary")
178
+
179
+ with gr.Column():
180
+ ve_output = gr.Video(label="Edited Video")
181
+ ve_status = gr.Textbox(label="Status", lines=10)
182
+
183
+ ve_btn.click(
184
+ fn=video_editing,
185
+ inputs=[ve_video, ve_instruction],
186
+ outputs=[ve_output, ve_status]
187
+ )
188
+
189
+ gr.Markdown("""
190
+ ## 📚 Resources
191
+
192
+ - **Paper**: [UniVideo: Unified Understanding, Generation, and Editing for Videos](https://arxiv.org/abs/2510.08377)
193
+ - **GitHub**: [KlingTeam/UniVideo](https://github.com/KlingTeam/UniVideo)
194
+ - **Model**: [KlingTeam/UniVideo on Hugging Face](https://huggingface.co/KlingTeam/UniVideo)
195
+ - **Demo Page**: [UniVideo Project Page](https://congwei1230.github.io/UniVideo/)
196
+
197
+ ## 💡 To Run UniVideo Locally
198
+
199
+ 1. **Clone Repository**:
200
+ ```bash
201
+ git clone https://github.com/KlingTeam/UniVideo.git
202
+ cd UniVideo
203
+ ```
204
+
205
+ 2. **Install Dependencies**:
206
+ ```bash
207
+ conda env create -f environment.yml
208
+ conda activate univideo
209
+ ```
210
+
211
+ 3. **Download Checkpoint**:
212
+ ```bash
213
+ python download_ckpt.py
214
+ ```
215
+
216
+ 4. **Run Inference**:
217
+ ```bash
218
+ cd univideo
219
+ python univideo_inference.py --task i2v --config configs/univideo_qwen2p5vl7b_hidden_hunyuanvideo.yaml
220
+ ```
221
+
222
+ ## ⚙️ Hardware Requirements
223
+
224
+ - **Recommended**: NVIDIA A100 (80GB) or H100
225
+ - **Minimum**: NVIDIA GPU with 24GB+ VRAM
226
+ - **CPU**: Not supported (model is too large)
227
+
228
+ ## 🔧 To Enable This Space
229
+
230
+ 1. Upgrade Space hardware to GPU (Settings → Hardware)
231
+ 2. Implement full UniVideo pipeline in `app.py`
232
+ 3. Add model checkpoint loading
233
+ 4. Configure inference parameters
234
+ """)
235
+
236
+ if __name__ == "__main__":
237
+ demo.launch()