eustlb's picture
eustlb HF Staff
Chart audio-to-audio model creation on the Hub
8671f98 verified
Raw History Blame Contribute Delete
3.32 kB
"""Refresh data.json from the public Hub API.
Counts every public model repo currently tagged `text-to-speech` by the
month it was created. No auth, no VPN — this hits huggingface.co.
Usage:
python3 fetch_data.py # the tag in task.json
python3 fetch_data.py text-to-audio # or any other tag
"""
from __future__ import annotations
import calendar
import datetime as dt
import json
import re
import sys
import urllib.parse
import urllib.request
from collections import Counter
from pathlib import Path
API = "https://huggingface.co/api/models"
# A release nobody bookmarks is not a release the field noticed. Likes are the
# only engagement signal the listing API returns, so they stand in for "notable".
NOTABLE_LIKES = 25
def fetch(task: str) -> tuple[Counter, Counter]:
"""Returns (all models per month, models with >=NOTABLE_LIKES per month)."""
url = (f"{API}?{urllib.parse.urlencode({'pipeline_tag': task, 'limit': 1000})}"
"&expand[]=createdAt&expand[]=likes")
months: Counter = Counter()
notable: Counter = Counter()
page = 0
while url:
req = urllib.request.Request(url, headers={"User-Agent": "tts-space/fetch_data"})
with urllib.request.urlopen(req, timeout=120) as r:
rows = json.loads(r.read())
link = r.headers.get("Link", "")
for m in rows:
created = m.get("createdAt")
if created:
months[created[:7]] += 1
if m.get("likes", 0) >= NOTABLE_LIKES:
notable[created[:7]] += 1
page += 1
print(f" page {page}: {len(rows)} rows, {sum(months.values())} total", file=sys.stderr)
nxt = re.search(r'<([^>]+)>;\s*rel="next"', link)
url = nxt.group(1) if nxt else None
return months, notable
def main() -> None:
cfg = json.loads((Path(__file__).with_name("task.json")).read_text())
task = sys.argv[1] if len(sys.argv) > 1 else cfg["tag"]
months, notable = fetch(task)
today = dt.date.today()
current = today.strftime("%Y-%m")
days_in_month = calendar.monthrange(today.year, today.month)[1]
series = [[m, months[m]] for m in sorted(months)]
notable_series = [[m, notable[m]] for m in sorted(months)]
out = {
"task": task,
"retrieved": today.isoformat(),
"total": sum(months.values()),
# The current month is only partly elapsed — the page marks it and
# projects a full-month pace from these two numbers.
"notable_likes": NOTABLE_LIKES,
"partial_month": current,
"partial_days": today.day,
"days_in_partial_month": days_in_month,
"months": series,
"notable": notable_series,
}
# One month per line keeps the diff readable when this is re-run.
head = {k: v for k, v in out.items() if k != "months"}
head = {k: v for k, v in head.items() if k != "notable"}
body = ",\n".join(f' ["{m}", {n}, {c}]' for (m, n), (_, c) in zip(series, notable_series))
text = (json.dumps(head, indent=2)[:-2]
+ ',\n "months": [\n' + body + "\n ]\n}\n")
target = Path(__file__).with_name("data.json")
target.write_text(text)
print(f"wrote {target} — {out['total']} models across {len(series)} months")
if __name__ == "__main__":
main()