File size: 19,586 Bytes
e6a95be
904e4ad
 
e6a95be
 
904e4ad
 
 
 
 
 
 
e6a95be
 
ab3b519
 
 
 
e6a95be
 
 
 
 
 
 
 
904e4ad
 
 
 
 
 
 
 
 
7270705
904e4ad
 
7123409
f5e3ffa
7123409
904e4ad
 
f5e3ffa
904e4ad
e6a95be
 
904e4ad
 
7123409
 
 
 
 
 
e6a95be
 
 
 
 
 
 
4578f11
 
 
e6a95be
 
 
2a6abee
 
 
 
 
 
 
 
 
 
 
904e4ad
f5e3ffa
3546667
7270705
3546667
 
 
 
 
 
 
 
 
f5e3ffa
 
904e4ad
 
f5e3ffa
3546667
f5e3ffa
 
 
 
3546667
 
 
 
 
 
f5e3ffa
904e4ad
 
e6a95be
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
55d9178
e6a95be
 
 
 
 
 
 
 
 
904e4ad
 
 
7270705
904e4ad
 
 
 
 
 
f5e3ffa
3546667
629c301
904e4ad
e6a95be
 
 
629c301
 
 
 
e6a95be
 
 
 
 
7270705
e6a95be
 
 
 
 
 
 
 
 
629c301
e6a95be
 
 
 
 
f5e3ffa
 
 
e6a95be
f5e3ffa
e6a95be
f5e3ffa
e6a95be
 
3d9fab2
3546667
e6a95be
3d9fab2
3546667
 
e6a95be
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
904e4ad
 
 
 
7270705
904e4ad
 
 
 
 
 
 
f5e3ffa
819311d
629c301
904e4ad
 
 
 
 
7270705
904e4ad
 
 
e6a95be
 
 
 
 
 
3d9fab2
e6a95be
7270705
e6a95be
 
 
 
 
 
 
 
 
 
 
ab3b519
e6a95be
 
ab3b519
e6a95be
 
 
 
 
 
 
 
 
904e4ad
ab3b519
 
904e4ad
 
ab3b519
 
 
 
 
 
 
 
 
904e4ad
e6a95be
 
 
 
 
 
3d9fab2
e6a95be
 
 
 
904e4ad
 
 
9fbf3f5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
904e4ad
 
 
 
2a6abee
 
 
 
de8e11b
 
 
 
2a6abee
 
 
 
 
 
 
 
 
 
 
 
 
 
3d9fab2
2a6abee
 
 
3d9fab2
2a6abee
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
904e4ad
 
 
 
 
 
 
f5e3ffa
 
 
 
 
 
819311d
441c4d5
 
3546667
 
819311d
904e4ad
7270705
 
4aabec2
7270705
 
904e4ad
 
 
 
 
9fbf3f5
904e4ad
 
 
 
 
 
 
 
 
7270705
904e4ad
 
 
 
 
 
 
f5e3ffa
819311d
904e4ad
 
 
 
9fbf3f5
 
 
 
 
904e4ad
acbba8c
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
import os
import random
import tempfile
import threading
import time as _time

import gradio as gr
import numpy as np
import spaces
import torch
from diffusers.utils.export_utils import export_to_video

from logging_utils import (
    LogUploader,
    print_export_done,
    print_export_error,
    print_export_start,
    print_frames_info,
    print_infer_done,
    print_infer_error,
    print_infer_start,
    print_stage_done,
    print_stage_error,
    print_stage_start,
    print_startup_env,
)
from model.pipeline import (
    DEFAULT_NEGATIVE_PROMPT,
    DEFAULT_PROMPT,
    FIXED_FPS,
    MAX_DURATION,
    MIN_DURATION,
    estimate_generation_seconds,
    get_num_frames,
    load_pipeline,
    resize_and_crop_to_match,
    resize_image,
)
from postprocess.interpolation import ensure_weights_downloaded as ensure_rife_weights_downloaded
from postprocess.interpolation import interpolate_frames
from postprocess.upscale.upscale import ensure_weights_downloaded as ensure_upscale_weights_downloaded

MAX_SEED = np.iinfo(np.int32).max
FRAME_MULTIPLIER_CHOICES = [FIXED_FPS, FIXED_FPS * 2, FIXED_FPS * 4, FIXED_FPS * 8]

print_startup_env()

pipe = load_pipeline()

# RIFE/upscale weight fetches are CPU/network-only (no CUDA needed) β€” done here, at process
# startup, so a cold container doesn't pay for them out of the metered @spaces.GPU allocation
# on its first interpolation/upscale request.
ensure_rife_weights_downloaded()
ensure_upscale_weights_downloaded()

# Debug logging (LOG-2): gated entirely by LOG_HF_TOKEN/LOG_DATASET_REPO β€” a silent no-op when
# either is unset. Deliberately not the deploy HF_TOKEN: this uses its own write-scoped token so a
# leak has a much smaller blast radius (append-only on one dataset repo, not the whole Space repo).
_log_uploader = LogUploader(
    token=os.environ.get("LOG_HF_TOKEN"),
    repo_id=os.environ.get("LOG_DATASET_REPO"),
    max_bytes=int(float(os.environ.get("LOG_STORAGE_CAP_GB", "10")) * 1024**3),
    # Hub's commit endpoint rejects pushes once a directory holds >10000 files; stay well under
    # that per-directory cap (each logged stem adds one file to data/, images/, and videos/).
    max_files=int(os.environ.get("LOG_MAX_FILES", "8000")),
    batch_interval=int(os.environ.get("LOG_BATCH_INTERVAL", "60")),
)

# SPACE_ID is set automatically by Hugging Face Spaces ("namespace/space_name") β€” used instead of a
# hardcoded handle so the privacy notice stays correct for anyone who duplicates this Space.
_SPACE_ID = os.environ.get("SPACE_ID")
_OPERATOR_MD = (
    f"[the Space operator](https://huggingface.co/{_SPACE_ID.split('/')[0]})" if _SPACE_ID else "the Space operator"
)
_COMMUNITY_MD = (
    f"[this Space's Community tab](https://huggingface.co/spaces/{_SPACE_ID}/discussions)"
    if _SPACE_ID else "this Space's Community tab"
)


def _gpu_duration(
    resized_image,
    processed_last_image,
    prompt,
    negative_prompt,
    steps,
    num_frames,
    guidance_scale,
    seed,
    frame_multiplier,
    upscale_output,
    progress,
):
    duration = estimate_generation_seconds(
        resized_image.width, resized_image.height, num_frames, int(steps), float(guidance_scale)
    )
    frame_factor = frame_multiplier // FIXED_FPS
    out_frames = num_frames
    if frame_factor > 1:
        # Matches the reference project's RIFE-time heuristic: ~0.02s per extra interpolated frame.
        extra_frames = (num_frames * frame_factor) - num_frames
        duration += extra_frames * 0.02
        out_frames += extra_frames
    if upscale_output:
        # Rough per-frame tiled-SR heuristic (4xLSDIRCompact, fp16); not yet empirically calibrated
        # on the dev Space (same caveat as estimate_generation_seconds above), plus a flat allowance
        # for first-call weight download/model load.
        duration += 20 + out_frames * 0.3
    return duration


def _apply_interpolation(raw_frames, frame_factor, progress):
    if frame_factor <= 1:
        return list(raw_frames), FIXED_FPS

    def _report_interpolation_progress(done: int, total: int) -> None:
        progress(0.7 + 0.2 * (done / total), desc=f"Interpolating frames ({done}/{total})...")

    print_stage_start("interpolation")
    t0 = _time.perf_counter()
    try:
        interpolated = interpolate_frames(
            raw_frames, multiplier=int(frame_factor), progress_callback=_report_interpolation_progress
        )
        print_stage_done("interpolation", _time.perf_counter() - t0)
        return interpolated, FIXED_FPS * frame_factor
    except Exception as e:
        # QF-2: don't discard a successfully generated base video over a post-processing failure.
        print_stage_error("interpolation", e)
        gr.Warning(f"Frame interpolation failed ({e}); returning the video without it.")
        return list(raw_frames), FIXED_FPS


def _apply_upscale(frames, progress):
    from postprocess.upscale import upscale_frames

    def _report_upscale_progress(done: int, total: int) -> None:
        progress(0.9 + 0.09 * (done / total), desc=f"Upscaling 4x ({done}/{total} frames)...")

    progress(0.9, desc=f"Upscaling 4x (0/{len(frames)} frames)...")
    print_stage_start("upscaling")
    t0 = _time.perf_counter()
    try:
        upscaled = upscale_frames(frames, progress_callback=_report_upscale_progress)
        print_stage_done("upscaling", _time.perf_counter() - t0)
        return upscaled
    except Exception as e:
        # QF-2: a failed upscale still returns the pre-upscale (interpolated/base) result.
        print_stage_error("upscaling", e)
        gr.Warning(f"Upscaling failed ({e}); returning the video without it.")
        return None


@spaces.GPU(duration=_gpu_duration)
def run_inference(
    resized_image,
    processed_last_image,
    prompt,
    negative_prompt,
    steps,
    num_frames,
    guidance_scale,
    seed,
    frame_multiplier,
    upscale_output,
    progress=gr.Progress(),
):
    print_infer_start(prompt, negative_prompt, seed, steps, guidance_scale, frame_multiplier, upscale_output)
    t_start = _time.perf_counter()

    def _report_generation_progress(pipe_, step_index, timestep, callback_kwargs):
        progress(0.7 * (step_index + 1) / int(steps), desc=f"Generating ({step_index + 1}/{int(steps)} steps)...")
        return callback_kwargs

    print_stage_start("generation")
    t0 = _time.perf_counter()
    try:
        result = pipe(
            image=resized_image,
            last_image=processed_last_image,
            prompt=prompt,
            negative_prompt=negative_prompt,
            height=resized_image.height,
            width=resized_image.width,
            num_frames=num_frames,
            guidance_scale=float(guidance_scale),
            num_inference_steps=int(steps),
            generator=torch.Generator(device="cuda").manual_seed(seed),
            output_type="np",
            callback_on_step_end=_report_generation_progress,
        )
    except Exception as e:
        print_stage_error("generation", e)
        raise
    print_stage_done("generation", _time.perf_counter() - t0)
    raw_frames = result.frames[0]  # (T, H, W, C) float32 in [0, 1]

    frame_factor = frame_multiplier // FIXED_FPS
    final_frames, final_fps = _apply_interpolation(raw_frames, frame_factor, progress)

    upscaled_frames = _apply_upscale(final_frames, progress) if upscale_output else None

    if upscaled_frames is None:
        print_infer_done(_time.perf_counter() - t_start, final_fps, len(final_frames))
        return final_frames, final_fps

    print_infer_done(_time.perf_counter() - t_start, final_fps, len(upscaled_frames))
    return upscaled_frames, final_fps


def _spawn_log(
    input_image,
    output_frames,
    output_fps,
    prompt,
    negative_prompt,
    seed,
    steps,
    guidance_scale,
    interpolation_enabled,
    interpolation_multiplier,
    upscale_enabled,
    output_width,
    output_height,
    generation_duration_seconds,
    success,
    error_message="",
):
    # Run on a background thread so committing to the Hub never adds latency to the request
    # (LOG-6) β€” must be spawned here (in the CPU-facing function), not inside the @spaces.GPU
    # run_inference call, since that runs in a subprocess that exits (killing any thread it spawns)
    # as soon as run_inference returns.
    threading.Thread(
        target=_log_uploader.log_inference,
        args=(
            input_image, output_frames, output_fps, prompt, negative_prompt, seed, steps,
            guidance_scale, interpolation_enabled, interpolation_multiplier, upscale_enabled,
            output_width, output_height, generation_duration_seconds, success, error_message,
        ),
        daemon=True,
    ).start()


def generate_video(
    input_image,
    last_image,
    prompt,
    negative_prompt,
    duration_seconds,
    steps,
    guidance_scale,
    seed,
    randomize_seed,
    frame_multiplier,
    upscale_output,
    progress=gr.Progress(),
):
    if input_image is None:
        raise gr.Error("Please upload an input image.")

    resized_image = resize_image(input_image)
    processed_last_image = resize_and_crop_to_match(last_image, resized_image) if last_image is not None else None
    num_frames = get_num_frames(duration_seconds)
    current_seed = random.randint(0, MAX_SEED) if randomize_seed else int(seed)

    frame_factor = frame_multiplier // FIXED_FPS
    interpolation_enabled = frame_factor > 1
    interpolation_multiplier = frame_factor if interpolation_enabled else 1

    t0 = _time.perf_counter()
    try:
        final_frames, final_fps = run_inference(
            resized_image,
            processed_last_image,
            prompt,
            negative_prompt,
            steps,
            num_frames,
            guidance_scale,
            current_seed,
            frame_multiplier,
            upscale_output,
            progress,
        )
    except Exception as e:
        import traceback as _tb
        duration = _time.perf_counter() - t0
        print_infer_error(e, duration)
        print(f"[infer] traceback:\n{_tb.format_exc()}", flush=True)
        if _log_uploader.enabled:
            _spawn_log(
                resized_image, None, None, prompt, negative_prompt, current_seed, steps,
                guidance_scale, interpolation_enabled, interpolation_multiplier, upscale_output,
                None, None, duration, False, str(e),
            )
        raise

    duration = _time.perf_counter() - t0

    print_frames_info("final_frames", final_frames, final_fps)

    with tempfile.NamedTemporaryFile(suffix=".mp4", delete=False) as tmpfile:
        video_path = tmpfile.name
    print_export_start(video_path)
    t_export = _time.perf_counter()
    try:
        export_to_video(final_frames, video_path, fps=final_fps, quality=6)
    except Exception as e:
        import traceback as _tb
        print_export_error(e, _tb.format_exc())
        raise
    print_export_done(_time.perf_counter() - t_export)

    if _log_uploader.enabled:
        # Interpolation/upscale failures are already caught-and-warned (not raised) inside
        # run_inference β€” a video was still produced, so this still counts as success=True.
        # interpolation_enabled/upscale_output reflect what the user requested, not whether an
        # internal fallback fired.
        _spawn_log(
            resized_image, final_frames, final_fps, prompt, negative_prompt, current_seed,
            steps, guidance_scale, interpolation_enabled, interpolation_multiplier, upscale_output,
            resized_image.width, resized_image.height, duration, True,
        )

    return video_path, video_path, current_seed


def preview_effect(input_image, duration_seconds, frame_multiplier, upscale_output):
    """QF-3: show the resulting fps/resolution/duration before the user hits Generate."""
    num_frames = get_num_frames(duration_seconds)
    frame_factor = frame_multiplier // FIXED_FPS
    out_fps = FIXED_FPS * frame_factor if frame_factor > 1 else FIXED_FPS
    out_frame_count = ((num_frames - 1) * frame_factor + 1) if frame_factor > 1 else num_frames

    if input_image is not None:
        w, h = resize_image(input_image).size
    else:
        w = h = None

    if upscale_output and w is not None:
        w, h = w * 4, h * 4

    resolution = f"{w}Γ—{h}px" if w is not None else "resolution depends on the uploaded image"
    return f"**Output:** ~{out_frame_count} frames @ {out_fps} fps, {resolution}"


with gr.Blocks() as demo:
    gr.Markdown("# High Quality Video Generation")
    gr.Markdown("Turn a still image into a short video clip guided by a text prompt.")

    gr.Markdown(
        "- Service availability and response times may vary as this Space runs on shared GPU infrastructure.\n"
        "- When using this Space, please comply with the "
        "[Hugging Face Content Policy](https://huggingface.co/content-policy).\n"
        "- This Space and its outputs are provided **\"as is\"**, without warranties of any kind. You are "
        "solely responsible for the content you generate and for how you use or share it; the operator "
        "accepts no liability for any loss, damage, or claim arising from use of this Space or its "
        "outputs.\n"
        f"- To monitor application performance and improve quality, input data (image/prompt) and generated "
        f"outputs are logged solely for debugging purposes and retained in a private Hugging Face dataset "
        f"accessible only to {_OPERATOR_MD}, automatically pruned once the dataset reaches its configured "
        f"limit (oldest records deleted first). This processing is based on legitimate interest (GDPR Art. "
        f"6(1)(f)) and the data is not shared with or sold to any third party. To request access to or "
        f"deletion of your data, open a discussion on {_COMMUNITY_MD}. See the **Privacy Policy** below for "
        "full details of your rights.\n"
        "- No user account or identifying information (such as IP address or session data) is collected."
    )

    with gr.Accordion("Privacy Policy", open=False):
        gr.Markdown(
            f"**Space:** High Quality Video Generation  \n"
            f"**Operator:** {_OPERATOR_MD}  \n"
            f"**Last updated:** 2026-09-18\n"
            "\n"
            "#### What data is collected\n"
            "When you submit a request, the following is logged: your uploaded input image, the generated "
            "output video, your prompt and negative prompt, seed, inference "
            "steps, guidance scale, frame-interpolation and upscaling settings, output resolution/fps/duration, "
            "generation duration, and success/error status. No user account, IP address, or session data is "
            "collected.\n"
            "\n"
            "#### Why it is collected\n"
            "Solely for debugging and monitoring application performance (legitimate interest, GDPR Art. "
            "6(1)(f)).\n"
            "\n"
            "#### Where it is stored\n"
            f"In a private Hugging Face dataset accessible only to {_OPERATOR_MD}. The data is not shared with "
            "or sold to any third party.\n"
            "\n"
            "#### How long it is kept\n"
            "Entries are automatically pruned once the dataset reaches its configured storage cap (default "
            "10GB) β€” the oldest entries are deleted first. The operator may also delete data manually at any "
            "time.\n"
            "\n"
            "#### Your rights\n"
            "Use of this Space and submission of images is entirely voluntary. Under GDPR you have the right "
            "to:\n"
            "- **Access** (Art. 15): request a copy of data held about you\n"
            "- **Erasure** (Art. 17): request deletion of your data\n"
            "- **Restriction** (Art. 18): request that processing be limited\n"
            "- **Portability** (Art. 20): receive your data in a machine-readable format\n"
            "- **Objection** (Art. 21): object to processing based on legitimate interest\n"
            "\n"
            f"To exercise any of these rights, open a discussion on {_COMMUNITY_MD}.\n"
            "\n"
            "#### Right to complain\n"
            "You have the right to lodge a complaint with your national data protection authority, for example: "
            "ICO (UK), CNIL (France), BfDI (Germany), or your local EU member state authority listed at "
            "[edpb.europa.eu](https://www.edpb.europa.eu/about-edpb/about-edpb/members_en)."
        )

    with gr.Row():
        with gr.Column():
            input_image_component = gr.Image(type="pil", label="Input Image", sources=["upload", "clipboard"])
            prompt_input = gr.Textbox(label="Prompt", value=DEFAULT_PROMPT)
            duration_input = gr.Slider(
                minimum=MIN_DURATION, maximum=MAX_DURATION, step=0.1, value=3.5, label="Duration (seconds)"
            )
            frame_multiplier_input = gr.Dropdown(
                choices=FRAME_MULTIPLIER_CHOICES,
                value=FIXED_FPS,
                label="Video Fluidity (Frames per Second)",
                info="Extra frames are generated with RIFE frame interpolation to smooth motion.",
            )
            upscale_checkbox = gr.Checkbox(
                label="Upscale 4Γ—",
                value=True,
                info="Upscales the finished video 4x on GPU after generation. Runs after frame "
                "interpolation, within the same GPU allocation, and adds to the GPU quota used.",
            )
            with gr.Accordion("Advanced Settings", open=False):
                last_image_component = gr.Image(
                    type="pil",
                    label="Last Image (Optional) β€” end frame to animate towards",
                    sources=["upload", "clipboard"],
                )
                negative_prompt_input = gr.Textbox(label="Negative Prompt", value=DEFAULT_NEGATIVE_PROMPT, lines=3)
                steps_slider = gr.Slider(minimum=1, maximum=30, step=1, value=6, label="Inference Steps")
                guidance_scale_input = gr.Slider(minimum=0.0, maximum=10.0, step=0.5, value=1.0, label="Guidance Scale")
                seed_input = gr.Slider(label="Seed", minimum=0, maximum=MAX_SEED, step=1, value=42)
                randomize_seed_checkbox = gr.Checkbox(label="Randomize seed", value=True)
            effect_preview = gr.Markdown()
            generate_button = gr.Button("Generate Video", variant="primary")
        with gr.Column():
            video_output = gr.Video(label="Generated Video", autoplay=True)
            file_output = gr.File(label="Download Video")

    generate_button.click(
        fn=generate_video,
        inputs=[
            input_image_component,
            last_image_component,
            prompt_input,
            negative_prompt_input,
            duration_input,
            steps_slider,
            guidance_scale_input,
            seed_input,
            randomize_seed_checkbox,
            frame_multiplier_input,
            upscale_checkbox,
        ],
        outputs=[video_output, file_output, seed_input],
    )

    preview_inputs = [input_image_component, duration_input, frame_multiplier_input, upscale_checkbox]
    for component in preview_inputs:
        component.change(fn=preview_effect, inputs=preview_inputs, outputs=effect_preview)
    demo.load(fn=preview_effect, inputs=preview_inputs, outputs=effect_preview)

if __name__ == "__main__":
    demo.queue().launch(show_error=True, ssr_mode=False)