{"slug":"ancoleman-evaluating-llms","source_name":"ancoleman/evaluating-llms","name":"Ancoleman/Evaluating Llms","description":"Evaluate LLM systems using automated metrics, LLM-as-judge, and benchmarks. Use when testing prompt quality, validating RAG pipelines, measuring safety (hallucinations, bias), or comparing models for production deployment.","version":1,"lift":{"pass_rate_delta_pts":18.18,"pass_rate_pct":72.7,"total_cases":22,"passed_cases":16,"tokens_delta_pct":181.8,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-21T12:15:45.985717+00:00"},"skill_score":0.7273,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":18.18,"with_pass_pct":72.7,"without_pass_pct":54.5,"tokens_delta_pct":181.8,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"mixed","never_hurt":true,"completed_at":"2026-08-21T12:15:45.985717+00:00","run_id":"f45de548-eba4-4436-844f-802c14930d41","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT","install_count":0,"manifest_hash":"a76528b4e7ee16e01096f7f2464f411bac5e5bae51d9530982b41c6df0d36b41","raw_url":"https://app.decimal.ai/s/ancoleman-evaluating-llms/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/ancoleman-evaluating-llms"}