someone-in-the-world Claude Sonnet 5 commited on
Commit
4578f11
·
unverified ·
1 Parent(s): b505376

Prune log repo on file count, not just storage size (#44)

Browse files

The Hub commit endpoint rejects pushes once a directory holds more than
10000 files. Storage-size-only retention didn't bound this: data/ and
images/ files are tiny relative to videos/, so thousands of stems
accumulated in those directories long before the byte cap ever
triggered a prune, eventually hitting the Hub's per-directory limit.

Add a LOG_MAX_FILES cap (default 8000 stems) that prunes alongside the
existing LOG_STORAGE_CAP_GB, whichever is hit first.


Claude-Session: https://claude.ai/code/session_01WHfM7C6kNJh15dWfYt8agd

Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com>

Files changed (2) hide show
  1. app.py +3 -0
  2. logging_utils.py +22 -11
app.py CHANGED
@@ -61,6 +61,9 @@ _log_uploader = LogUploader(
61
  token=os.environ.get("LOG_HF_TOKEN"),
62
  repo_id=os.environ.get("LOG_DATASET_REPO"),
63
  max_bytes=int(float(os.environ.get("LOG_STORAGE_CAP_GB", "10")) * 1024**3),
 
 
 
64
  batch_interval=int(os.environ.get("LOG_BATCH_INTERVAL", "60")),
65
  )
66
 
 
61
  token=os.environ.get("LOG_HF_TOKEN"),
62
  repo_id=os.environ.get("LOG_DATASET_REPO"),
63
  max_bytes=int(float(os.environ.get("LOG_STORAGE_CAP_GB", "10")) * 1024**3),
64
+ # Hub's commit endpoint rejects pushes once a directory holds >10000 files; stay well under
65
+ # that per-directory cap (each logged stem adds one file to data/, images/, and videos/).
66
+ max_files=int(os.environ.get("LOG_MAX_FILES", "8000")),
67
  batch_interval=int(os.environ.get("LOG_BATCH_INTERVAL", "60")),
68
  )
69
 
logging_utils.py CHANGED
@@ -3,8 +3,11 @@ Face Hub dataset repo (LOG-2 through LOG-9).
3
 
4
  Modeled on the FireRed-Image-Edit-1.0-Fast reference project's logging_utils.py/inference.py
5
  pipeline, adapted for video: logged media is referenced by path rather than embedded as bytes
6
- (LOG-5), the logged output video is the pre-upscale result (LOG-4), and retention is a
7
- total-storage-size cap rather than a file-count cap (LOG-7).
 
 
 
8
  """
9
 
10
  import json
@@ -271,19 +274,26 @@ def _group_by_stem(existing: list[tuple[str, int]]) -> dict[str, dict[str, Any]]
271
  return groups
272
 
273
 
274
- def _build_delete_ops_by_size(existing: list[tuple[str, int]], new_batch_size: int,
275
- max_bytes: int) -> list[CommitOperationDelete]:
276
  groups = _group_by_stem(existing)
277
- total = sum(g["size"] for g in groups.values()) + new_batch_size
278
- if max_bytes <= 0 or total <= max_bytes:
 
 
 
 
 
 
279
  return []
280
  ops: list[CommitOperationDelete] = []
281
  for stem in sorted(groups):
282
- if total <= max_bytes:
283
  break
284
  group = groups[stem]
285
  ops.extend(CommitOperationDelete(path_in_repo=p) for p in group["paths"])
286
- total -= group["size"]
 
287
  return ops
288
 
289
 
@@ -313,10 +323,12 @@ def _delete_temp_files(paths: list[str]) -> None:
313
 
314
 
315
  class LogUploader:
316
- def __init__(self, token: str | None, repo_id: str | None, max_bytes: int, batch_interval: int = 60) -> None:
 
317
  self._token = token
318
  self._repo_id = repo_id
319
  self._max_bytes = max_bytes
 
320
  self._batch_interval = batch_interval
321
  self._pending: list[tuple[str, str]] = []
322
  self._lock = threading.Lock()
@@ -412,9 +424,8 @@ class LogUploader:
412
  api = HfApi(token=self._token)
413
  api.create_repo(repo_id=self._repo_id, repo_type="dataset", private=True, exist_ok=True)
414
  existing = _list_existing_files_with_sizes(api, self._repo_id)
415
- new_batch_size = sum(os.path.getsize(local) for _, local in batch)
416
  add_ops = [CommitOperationAdd(path_in_repo=p, path_or_fileobj=local) for p, local in batch]
417
- del_ops = _build_delete_ops_by_size(existing, new_batch_size, self._max_bytes)
418
  api.create_commit(
419
  repo_id=self._repo_id, repo_type="dataset",
420
  operations=[*add_ops, *del_ops],
 
3
 
4
  Modeled on the FireRed-Image-Edit-1.0-Fast reference project's logging_utils.py/inference.py
5
  pipeline, adapted for video: logged media is referenced by path rather than embedded as bytes
6
+ (LOG-5), the logged output video is the pre-upscale result (LOG-4), and retention prunes on
7
+ both a total-storage-size cap and a per-directory file-count cap (LOG-7) — the latter exists
8
+ because the Hub commit endpoint rejects pushes once a directory (data/, images/, videos/)
9
+ holds more than 10000 files, which a size-only cap doesn't bound since data/ and images/
10
+ files are tiny compared to videos/.
11
  """
12
 
13
  import json
 
274
  return groups
275
 
276
 
277
+ def _build_delete_ops(existing: list[tuple[str, int]], new_batch: list[tuple[str, str]],
278
+ max_bytes: int, max_files: int) -> list[CommitOperationDelete]:
279
  groups = _group_by_stem(existing)
280
+ new_stems = {_stem_of(p) for p, _ in new_batch}
281
+ total_bytes = sum(g["size"] for g in groups.values()) + sum(os.path.getsize(local) for _, local in new_batch)
282
+ total_files = len(groups) + len(new_stems)
283
+
284
+ def _over_cap() -> bool:
285
+ return (max_bytes > 0 and total_bytes > max_bytes) or (max_files > 0 and total_files > max_files)
286
+
287
+ if not _over_cap():
288
  return []
289
  ops: list[CommitOperationDelete] = []
290
  for stem in sorted(groups):
291
+ if not _over_cap():
292
  break
293
  group = groups[stem]
294
  ops.extend(CommitOperationDelete(path_in_repo=p) for p in group["paths"])
295
+ total_bytes -= group["size"]
296
+ total_files -= 1
297
  return ops
298
 
299
 
 
323
 
324
 
325
  class LogUploader:
326
+ def __init__(self, token: str | None, repo_id: str | None, max_bytes: int, max_files: int,
327
+ batch_interval: int = 60) -> None:
328
  self._token = token
329
  self._repo_id = repo_id
330
  self._max_bytes = max_bytes
331
+ self._max_files = max_files
332
  self._batch_interval = batch_interval
333
  self._pending: list[tuple[str, str]] = []
334
  self._lock = threading.Lock()
 
424
  api = HfApi(token=self._token)
425
  api.create_repo(repo_id=self._repo_id, repo_type="dataset", private=True, exist_ok=True)
426
  existing = _list_existing_files_with_sizes(api, self._repo_id)
 
427
  add_ops = [CommitOperationAdd(path_in_repo=p, path_or_fileobj=local) for p, local in batch]
428
+ del_ops = _build_delete_ops(existing, batch, self._max_bytes, self._max_files)
429
  api.create_commit(
430
  repo_id=self._repo_id, repo_type="dataset",
431
  operations=[*add_ops, *del_ops],