{"slug":"agentscope-ai-metric-design","source_name":"agentscope-ai/metric-design","name":"Agentscope AI/Metric Design","description":"Use when the user has evaluation principles or a dataset but needs help choosing the right graders, designing evaluation metrics, creating LLM-as-judge prompts, combining multiple metrics into a composite score, or building an automated evaluation pipeline. Also use when the user mentions grader selection, metric design, judge prompt engineering, rubric design, evaluation pipeline code, or \"how to evaluate [X] automatically.\" Outputs executable OpenJudge pipeline code.","version":1,"lift":{"pass_rate_delta_pts":54.55,"pass_rate_pct":90.9,"total_cases":22,"passed_cases":20,"tokens_delta_pct":181.8,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-14T18:54:25.877337+00:00"},"skill_score":0.9091,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":54.55,"with_pass_pct":90.9,"without_pass_pct":36.4,"tokens_delta_pct":181.8,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":21,"verdict":"mixed","never_hurt":true,"completed_at":"2026-08-14T18:54:25.877337+00:00","run_id":"31f80bf0-ba8c-4741-a4cd-671eceb0048e","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"Apache-2.0","install_count":0,"manifest_hash":"0fd60a677bcca7943d150ebacb4d323e015245ff703276d34912851b19d57172","raw_url":"https://app.decimal.ai/s/agentscope-ai-metric-design/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/agentscope-ai-metric-design"}