Download mcp/embeddings.py from mgbam/MCP_Res: direct link, hf CLI and curl.
- Browser
- Download file 1.07 kB
-
https://huggingface.co/spaces/mgbam/MCP_Res/resolve/49fee0b83cc6421d58b3568b62846bb4cc8cd6c0/mcp/embeddings.py
- Command line
-
hf download hf://spaces/mgbam/MCP_Res@49fee0b83cc6421d58b3568b62846bb4cc8cd6c0/mcp/embeddings.py
-
curl -L -o embeddings.py https://huggingface.co/spaces/mgbam/MCP_Res/resolve/49fee0b83cc6421d58b3568b62846bb4cc8cd6c0/mcp/embeddings.py
1.07 kB
| # ββ mcp/embeddings.py βββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| import os, asyncio | |
| from huggingface_hub import InferenceClient | |
| from sklearn.cluster import KMeans | |
| # Use your HF token for pro endpoints | |
| HF_TOKEN = os.getenv("HF_TOKEN") | |
| EMBED_MODEL = "sentence-transformers/all-mpnet-base-v2" | |
| client = InferenceClient(token=HF_TOKEN) | |
| async def embed_texts(texts: list[str]) -> list[list[float]]: | |
| """ | |
| Compute embeddings for a list of texts via HF Inference API. | |
| """ | |
| def _embed(t): | |
| return client.embed(model=EMBED_MODEL, inputs=t) | |
| # run in threadpool | |
| tasks = [asyncio.to_thread(_embed, t) for t in texts] | |
| return await asyncio.gather(*tasks) | |
| async def cluster_embeddings(embs: list[list[float]], n_clusters: int = 5) -> list[int]: | |
| """ | |
| Cluster embeddings into n_clusters, return list of cluster labels. | |
| """ | |
| kmeans = KMeans(n_clusters=n_clusters, random_state=0) | |
| return kmeans.fit_predict(embs).tolist() | |