{"slug":"wshobson-llm-evaluation","source_name":"wshobson/llm-evaluation","name":"Wshobson/LLM Evaluation","description":"Implement comprehensive evaluation strategies for LLM applications using automated metrics, human feedback, and benchmarking. Use when testing LLM performance, measuring AI application quality, or establishing evaluation frameworks.","version":1,"lift":{"pass_rate_delta_pts":0,"pass_rate_pct":82.6,"total_cases":23,"passed_cases":19,"tokens_delta_pct":34.6,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-07T12:19:52.885574+00:00"},"skill_score":null,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":0,"with_pass_pct":82.6,"without_pass_pct":82.6,"tokens_delta_pct":34.6,"turns_delta_pct":0,"total_cases":23,"cases_aggregated":23,"verdict":"mixed","never_hurt":true,"completed_at":"2026-08-07T12:19:52.885574+00:00","run_id":"344a0078-62d2-4ef5-80d8-3ae34cca6353","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT","install_count":0,"manifest_hash":"87c3193383841fc553e7f67e9a18db97000fabe6d33a3949041b0e6a1c8e6237","raw_url":"https://app.decimal.ai/s/wshobson-llm-evaluation/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/wshobson-llm-evaluation"}