{"slug":"mkurman-openai-evals","source_name":"mkurman/openai-evals","name":"Mkurman/Openai Evals","description":"LLM evaluation framework and registry (OpenAI Evals). Framework for evaluating LLMs and LLM-based systems with a registry of community-contributed eval templates. Supports model-graded evals, classification, simple completion matching, and custom completion functions. Use for systematic LLM quality testing, regression detection, and prompt engineering validation.","version":1,"lift":{"pass_rate_delta_pts":18.18,"pass_rate_pct":95.5,"total_cases":22,"passed_cases":21,"tokens_delta_pct":9.5,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-21T10:24:18.753994+00:00"},"skill_score":0.9545,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":18.18,"with_pass_pct":95.5,"without_pass_pct":77.3,"tokens_delta_pct":9.5,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"mixed","never_hurt":true,"completed_at":"2026-08-21T10:24:18.753994+00:00","run_id":"97e77e7e-042f-4b11-807b-0f9003f9c0ce","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT license","install_count":0,"manifest_hash":"1051cfe91444cd0c52842275ab2efaa32485306000c33a0dcdc469942125c851","raw_url":"https://app.decimal.ai/s/mkurman-openai-evals/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/mkurman-openai-evals"}