multimodalart HF Staff commited on
Commit
1110d75
·
verified ·
1 Parent(s): 6d1aa02

Size the GPU reservation to the measured request

Browse files
Files changed (1) hide show
  1. app.py +12 -4
app.py CHANGED
@@ -91,8 +91,13 @@ MAX_IMAGE_SLOTS, OPEN_IMAGE_SLOTS = 3, 2
91
 
92
  # Seconds of GPU one request needs, from the packed sequence it is about to denoise: linear in the rows for the
93
  # matmuls, quadratic for the attention. Fit shared with the other MiniMax-H3 Spaces on this pool.
94
- STEP_LINEAR, STEP_QUADRATIC, SAFETY = 1.1745e-4, 3.8396e-9, 1.3
95
- PLACEMENT_ALLOWANCE = int(os.environ.get("H3_PLACEMENT_ALLOWANCE", "90"))
 
 
 
 
 
96
  AUDIO_LATENTS_PER_SECOND, AUDIO_CHANNELS = 40, 2
97
  REFERENCE_IMAGE_SHORT_EDGE, CANVAS_MULTIPLE = 2048, 32
98
  DECODE_BASE, DECODE_PER_DEFAULT_CANVAS, DEFAULT_CANVAS_PIXELS = 15, 25, 960 * 544 * 124
@@ -223,7 +228,10 @@ def reference_rows(image_paths: list[str]) -> int:
223
 
224
 
225
  def denoise_seconds(sequence: int, steps: int) -> float:
226
- return int(steps) * (STEP_LINEAR * sequence + STEP_QUADRATIC * sequence**2) * SAFETY
 
 
 
227
 
228
 
229
  def get_duration(prompt_embeds, text_token_tags, image_paths, height, width, num_frames, steps, weight, seed, **_):
@@ -497,7 +505,7 @@ def compose_prompt(references, action, camera, hud, soundscape, music, seconds)
497
  "subject_definitions:\n" + "\n".join(definitions),
498
  (
499
  "summary:\n"
500
- f"[reference generation] A {float(seconds):g}-second photorealistic {spec['genre']} action RPG "
501
  f"gameplay sequence{place}, where {subject} {action}, {spec['summary']}."
502
  ),
503
  "retention_analysis:\n" + "\n".join(retention),
 
91
 
92
  # Seconds of GPU one request needs, from the packed sequence it is about to denoise: linear in the rows for the
93
  # matmuls, quadratic for the attention. Fit shared with the other MiniMax-H3 Spaces on this pool.
94
+ STEP_LINEAR, STEP_QUADRATIC, SAFETY = 1.1745e-4, 3.8396e-9, 1.15
95
+ # The fit above was measured with the first-block cache off. With it on — the default — roughly a third of the
96
+ # forwards skip the trunk, and the measured request below came in well under the uncached prediction. Both numbers
97
+ # are calibrated against this Space's own smoke test: S=40696 at 20 steps reserved 445s under the old 1.3/90/no-FBC
98
+ # constants and actually used 250s of GPU including placement, so a visitor's quota was paying for 195 idle seconds.
99
+ FBC_SPEEDUP = float(os.environ.get("H3_FBC_SPEEDUP", "1.45"))
100
+ PLACEMENT_ALLOWANCE = int(os.environ.get("H3_PLACEMENT_ALLOWANCE", "60"))
101
  AUDIO_LATENTS_PER_SECOND, AUDIO_CHANNELS = 40, 2
102
  REFERENCE_IMAGE_SHORT_EDGE, CANVAS_MULTIPLE = 2048, 32
103
  DECODE_BASE, DECODE_PER_DEFAULT_CANVAS, DEFAULT_CANVAS_PIXELS = 15, 25, 960 * 544 * 124
 
228
 
229
 
230
  def denoise_seconds(sequence: int, steps: int) -> float:
231
+ import h3_fbc
232
+
233
+ uncached = int(steps) * (STEP_LINEAR * sequence + STEP_QUADRATIC * sequence**2) * SAFETY
234
+ return uncached / (FBC_SPEEDUP if h3_fbc.ENABLED else 1.0)
235
 
236
 
237
  def get_duration(prompt_embeds, text_token_tags, image_paths, height, width, num_frames, steps, weight, seed, **_):
 
505
  "subject_definitions:\n" + "\n".join(definitions),
506
  (
507
  "summary:\n"
508
+ f"[reference generation] A {float(seconds):.1f}-second photorealistic {spec['genre']} action RPG "
509
  f"gameplay sequence{place}, where {subject} {action}, {spec['summary']}."
510
  ),
511
  "retention_analysis:\n" + "\n".join(retention),