{"slug":"davila7-agent-evaluation","source_name":"davila7/agent-evaluation","name":"Davila7/Agent Evaluation","description":"Testing and benchmarking LLM agents including behavioral testing, capability assessment, reliability metrics, and production monitoring—where even top agents achieve less than 50% on real-world benchmarks Use when: agent testing, agent evaluation, benchmark agents, agent reliability, test agent.","version":1,"lift":{"pass_rate_delta_pts":13.64,"pass_rate_pct":81.8,"total_cases":22,"passed_cases":18,"tokens_delta_pct":23.6,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-12T14:02:40.397133+00:00"},"skill_score":null,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":13.64,"with_pass_pct":81.8,"without_pass_pct":68.2,"tokens_delta_pct":23.6,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"mixed","never_hurt":false,"completed_at":"2026-08-12T14:02:40.397133+00:00","run_id":"abe10bad-6e81-45a0-a229-f05a17ecd535","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT","install_count":0,"manifest_hash":"582d9ac7a750932d21fc848960cd68c7c910cdfd61baf68ac78e56c090e0afb2","raw_url":"https://app.decimal.ai/s/davila7-agent-evaluation/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/davila7-agent-evaluation"}