Download project.json from RKB109/rag-evaluation-lab-20260720-model: direct link, hf CLI and curl.
- Browser
- Download file 1.93 kB
-
https://huggingface.co/RKB109/rag-evaluation-lab-20260720-model/resolve/9f198fa971b13576c8f8c23a3f4b70ed1abaf9e3/project.json
- Command line
-
hf download hf://RKB109/rag-evaluation-lab-20260720-model@9f198fa971b13576c8f8c23a3f4b70ed1abaf9e3/project.json
-
curl -L -o project.json https://huggingface.co/RKB109/rag-evaluation-lab-20260720-model/resolve/9f198fa971b13576c8f8c23a3f4b70ed1abaf9e3/project.json
1.93 kB
| { | |
| "name": "RAG Evaluation Lab", | |
| "problem": "RAG systems often ship without a stable regression set or failure taxonomy.", | |
| "domain": "ai-evaluation", | |
| "architecture": "classifier", | |
| "hugging_face_tasks": [ | |
| "text-classification", | |
| "question-answering", | |
| "text-ranking", | |
| "summarization" | |
| ], | |
| "recommended_stack": [ | |
| "FastAPI for evaluation jobs and reports", | |
| "Ragas-style retrieval and faithfulness metrics", | |
| "MLflow for experiment and artifact tracking", | |
| "PostgreSQL for versioned evaluation cases", | |
| "OpenTelemetry plus Phoenix for trace inspection", | |
| "GitHub Actions for threshold-based release gates" | |
| ], | |
| "real_world_data_sources": [ | |
| { | |
| "name": "GitHub REST API", | |
| "url": "https://api.github.com/repos/huggingface/transformers/issues?state=open&per_page=5", | |
| "purpose": "Real technical questions for evaluation-set construction" | |
| }, | |
| { | |
| "name": "arXiv API", | |
| "url": "https://export.arxiv.org/api/query?search_query=all:retrieval%20augmented%20generation&start=0&max_results=5", | |
| "purpose": "Public RAG literature for grounded-answer cases" | |
| } | |
| ], | |
| "job_description_skills": [ | |
| "LLM and RAG evaluation design", | |
| "Golden datasets and failure taxonomies", | |
| "Experiment tracking and model release gates", | |
| "Trace-level diagnosis and prompt regression testing", | |
| "Statistical comparison of AI system versions" | |
| ], | |
| "impact_targets": [ | |
| "Detect 100% of seeded unsupported-answer regressions", | |
| "Track retrieval, faithfulness, citation, latency, and cost metrics", | |
| "Fail CI when any critical metric drops beyond tolerance", | |
| "Produce comparable evaluation reports for every model version" | |
| ], | |
| "baseline_evaluation": { | |
| "test_examples": 4, | |
| "accuracy": 0.75, | |
| "synthetic_evaluation": true | |
| }, | |
| "estimated_delivery": "8-12 weeks for one engineer", | |
| "generated_baseline_is_production_ready": false | |
| } | |