{"slug":"rag-eval-design","source_name":"rag-eval-design","name":"RAG / retrieval eval design","description":"Designs an evaluation for a retrieval-augmented (RAG) system — scoring the retriever separately from the generator, requiring a groundedness check, and building a labeled golden set with hard negatives. Use when someone wants to test, benchmark, or set up an eval for a system that looks up documents/passages and then writes an answer from them. Do NOT use for general eval or rubric authoring on non-retrieval tasks (that is eval-writer / prompt-eval-rubric-writer), or for making an agent answer-from-sources at runtime (that is rag-grounding-and-citation).","version":1,"lift":{"pass_rate_delta_pts":null,"pass_rate_pct":53.8,"total_cases":13,"passed_cases":7,"tokens_delta_pct":591.3,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.5-flash","grading_method":"judged","completed_at":"2026-07-10T16:52:53.477042+00:00"},"skill_score":0.4126,"benchmark_models":[{"model":"gemini-3.5-flash","headline":false,"delta_pts":null,"with_pass_pct":null,"without_pass_pct":null,"tokens_delta_pct":591.3,"turns_delta_pct":0,"total_cases":13,"cases_aggregated":13,"verdict":"mixed","never_hurt":true,"completed_at":"2026-07-10T16:52:53.477042+00:00","run_id":"88384e3c-9268-4027-b360-0c2c1b3a278e","version_number":1,"is_latest_version":true,"gate":"not_comparable"}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":false},"license":null,"install_count":6,"manifest_hash":"8e26a2b85aba9f0084e30c557c0aa95ed1a18eb3e082365bd20481e878ec0de0","raw_url":"https://app.decimal.ai/s/rag-eval-design/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/rag-eval-design"}