{"slug":"agentscope-ai-meta-eval","source_name":"agentscope-ai/meta-eval","name":"Agentscope AI/Meta Eval","description":"Use when the user wants to build an evaluation system for an LLM/agent application but doesn't know where to start — they have traces, prompts, RAG pipelines, or nothing at all. Also use when the user mentions evaluation, eval, benchmarking, testing LLM quality, measuring agent performance, assessing RAG accuracy, or wants to compare prompts/models. This skill is the entry router: it asks diagnostic questions then recommends which sub-skill (local workflow) to use next.","version":1,"lift":{"pass_rate_delta_pts":72.73,"pass_rate_pct":86.4,"total_cases":22,"passed_cases":19,"tokens_delta_pct":93.9,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-15T19:37:20.774594+00:00"},"skill_score":0.8636,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":72.73,"with_pass_pct":86.4,"without_pass_pct":13.6,"tokens_delta_pct":93.9,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"mixed","never_hurt":false,"completed_at":"2026-08-15T19:37:20.774594+00:00","run_id":"8a30bb35-d659-4979-b14c-a287c5ed9074","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"Apache-2.0","install_count":0,"manifest_hash":"ab94e3b1bbc19c6d97806a201952af259b15ae4a07a02033228e13729406773d","raw_url":"https://app.decimal.ai/s/agentscope-ai-meta-eval/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/agentscope-ai-meta-eval"}