Spaces:
Running on Zero
Running on Zero
Prune log repo on file count, not just storage size (#44)
Browse filesThe Hub commit endpoint rejects pushes once a directory holds more than
10000 files. Storage-size-only retention didn't bound this: data/ and
images/ files are tiny relative to videos/, so thousands of stems
accumulated in those directories long before the byte cap ever
triggered a prune, eventually hitting the Hub's per-directory limit.
Add a LOG_MAX_FILES cap (default 8000 stems) that prunes alongside the
existing LOG_STORAGE_CAP_GB, whichever is hit first.
Claude-Session: https://claude.ai/code/session_01WHfM7C6kNJh15dWfYt8agd
Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com>
- app.py +3 -0
- logging_utils.py +22 -11
app.py
CHANGED
|
@@ -61,6 +61,9 @@ _log_uploader = LogUploader(
|
|
| 61 |
token=os.environ.get("LOG_HF_TOKEN"),
|
| 62 |
repo_id=os.environ.get("LOG_DATASET_REPO"),
|
| 63 |
max_bytes=int(float(os.environ.get("LOG_STORAGE_CAP_GB", "10")) * 1024**3),
|
|
|
|
|
|
|
|
|
|
| 64 |
batch_interval=int(os.environ.get("LOG_BATCH_INTERVAL", "60")),
|
| 65 |
)
|
| 66 |
|
|
|
|
| 61 |
token=os.environ.get("LOG_HF_TOKEN"),
|
| 62 |
repo_id=os.environ.get("LOG_DATASET_REPO"),
|
| 63 |
max_bytes=int(float(os.environ.get("LOG_STORAGE_CAP_GB", "10")) * 1024**3),
|
| 64 |
+
# Hub's commit endpoint rejects pushes once a directory holds >10000 files; stay well under
|
| 65 |
+
# that per-directory cap (each logged stem adds one file to data/, images/, and videos/).
|
| 66 |
+
max_files=int(os.environ.get("LOG_MAX_FILES", "8000")),
|
| 67 |
batch_interval=int(os.environ.get("LOG_BATCH_INTERVAL", "60")),
|
| 68 |
)
|
| 69 |
|
logging_utils.py
CHANGED
|
@@ -3,8 +3,11 @@ Face Hub dataset repo (LOG-2 through LOG-9).
|
|
| 3 |
|
| 4 |
Modeled on the FireRed-Image-Edit-1.0-Fast reference project's logging_utils.py/inference.py
|
| 5 |
pipeline, adapted for video: logged media is referenced by path rather than embedded as bytes
|
| 6 |
-
(LOG-5), the logged output video is the pre-upscale result (LOG-4), and retention
|
| 7 |
-
total-storage-size cap
|
|
|
|
|
|
|
|
|
|
| 8 |
"""
|
| 9 |
|
| 10 |
import json
|
|
@@ -271,19 +274,26 @@ def _group_by_stem(existing: list[tuple[str, int]]) -> dict[str, dict[str, Any]]
|
|
| 271 |
return groups
|
| 272 |
|
| 273 |
|
| 274 |
-
def
|
| 275 |
-
|
| 276 |
groups = _group_by_stem(existing)
|
| 277 |
-
|
| 278 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 279 |
return []
|
| 280 |
ops: list[CommitOperationDelete] = []
|
| 281 |
for stem in sorted(groups):
|
| 282 |
-
if
|
| 283 |
break
|
| 284 |
group = groups[stem]
|
| 285 |
ops.extend(CommitOperationDelete(path_in_repo=p) for p in group["paths"])
|
| 286 |
-
|
|
|
|
| 287 |
return ops
|
| 288 |
|
| 289 |
|
|
@@ -313,10 +323,12 @@ def _delete_temp_files(paths: list[str]) -> None:
|
|
| 313 |
|
| 314 |
|
| 315 |
class LogUploader:
|
| 316 |
-
def __init__(self, token: str | None, repo_id: str | None, max_bytes: int,
|
|
|
|
| 317 |
self._token = token
|
| 318 |
self._repo_id = repo_id
|
| 319 |
self._max_bytes = max_bytes
|
|
|
|
| 320 |
self._batch_interval = batch_interval
|
| 321 |
self._pending: list[tuple[str, str]] = []
|
| 322 |
self._lock = threading.Lock()
|
|
@@ -412,9 +424,8 @@ class LogUploader:
|
|
| 412 |
api = HfApi(token=self._token)
|
| 413 |
api.create_repo(repo_id=self._repo_id, repo_type="dataset", private=True, exist_ok=True)
|
| 414 |
existing = _list_existing_files_with_sizes(api, self._repo_id)
|
| 415 |
-
new_batch_size = sum(os.path.getsize(local) for _, local in batch)
|
| 416 |
add_ops = [CommitOperationAdd(path_in_repo=p, path_or_fileobj=local) for p, local in batch]
|
| 417 |
-
del_ops =
|
| 418 |
api.create_commit(
|
| 419 |
repo_id=self._repo_id, repo_type="dataset",
|
| 420 |
operations=[*add_ops, *del_ops],
|
|
|
|
| 3 |
|
| 4 |
Modeled on the FireRed-Image-Edit-1.0-Fast reference project's logging_utils.py/inference.py
|
| 5 |
pipeline, adapted for video: logged media is referenced by path rather than embedded as bytes
|
| 6 |
+
(LOG-5), the logged output video is the pre-upscale result (LOG-4), and retention prunes on
|
| 7 |
+
both a total-storage-size cap and a per-directory file-count cap (LOG-7) — the latter exists
|
| 8 |
+
because the Hub commit endpoint rejects pushes once a directory (data/, images/, videos/)
|
| 9 |
+
holds more than 10000 files, which a size-only cap doesn't bound since data/ and images/
|
| 10 |
+
files are tiny compared to videos/.
|
| 11 |
"""
|
| 12 |
|
| 13 |
import json
|
|
|
|
| 274 |
return groups
|
| 275 |
|
| 276 |
|
| 277 |
+
def _build_delete_ops(existing: list[tuple[str, int]], new_batch: list[tuple[str, str]],
|
| 278 |
+
max_bytes: int, max_files: int) -> list[CommitOperationDelete]:
|
| 279 |
groups = _group_by_stem(existing)
|
| 280 |
+
new_stems = {_stem_of(p) for p, _ in new_batch}
|
| 281 |
+
total_bytes = sum(g["size"] for g in groups.values()) + sum(os.path.getsize(local) for _, local in new_batch)
|
| 282 |
+
total_files = len(groups) + len(new_stems)
|
| 283 |
+
|
| 284 |
+
def _over_cap() -> bool:
|
| 285 |
+
return (max_bytes > 0 and total_bytes > max_bytes) or (max_files > 0 and total_files > max_files)
|
| 286 |
+
|
| 287 |
+
if not _over_cap():
|
| 288 |
return []
|
| 289 |
ops: list[CommitOperationDelete] = []
|
| 290 |
for stem in sorted(groups):
|
| 291 |
+
if not _over_cap():
|
| 292 |
break
|
| 293 |
group = groups[stem]
|
| 294 |
ops.extend(CommitOperationDelete(path_in_repo=p) for p in group["paths"])
|
| 295 |
+
total_bytes -= group["size"]
|
| 296 |
+
total_files -= 1
|
| 297 |
return ops
|
| 298 |
|
| 299 |
|
|
|
|
| 323 |
|
| 324 |
|
| 325 |
class LogUploader:
|
| 326 |
+
def __init__(self, token: str | None, repo_id: str | None, max_bytes: int, max_files: int,
|
| 327 |
+
batch_interval: int = 60) -> None:
|
| 328 |
self._token = token
|
| 329 |
self._repo_id = repo_id
|
| 330 |
self._max_bytes = max_bytes
|
| 331 |
+
self._max_files = max_files
|
| 332 |
self._batch_interval = batch_interval
|
| 333 |
self._pending: list[tuple[str, str]] = []
|
| 334 |
self._lock = threading.Lock()
|
|
|
|
| 424 |
api = HfApi(token=self._token)
|
| 425 |
api.create_repo(repo_id=self._repo_id, repo_type="dataset", private=True, exist_ok=True)
|
| 426 |
existing = _list_existing_files_with_sizes(api, self._repo_id)
|
|
|
|
| 427 |
add_ops = [CommitOperationAdd(path_in_repo=p, path_or_fileobj=local) for p, local in batch]
|
| 428 |
+
del_ops = _build_delete_ops(existing, batch, self._max_bytes, self._max_files)
|
| 429 |
api.create_commit(
|
| 430 |
repo_id=self._repo_id, repo_type="dataset",
|
| 431 |
operations=[*add_ops, *del_ops],
|