{"slug":"agentscope-ai-eval-design","source_name":"agentscope-ai/eval-design","name":"Agentscope AI/Eval Design","description":"Use when the user needs to design evaluation datasets, create test cases, stratify samples, generate adversarial examples, extract eval dimensions from traces/specs, or build a labeled evaluation set. Also use when the user mentions test data design, eval coverage, difficulty stratification, synthetic data generation for eval, or \"how to create good evaluation data.\" Outputs datasets in OpenJudge-compatible format.","version":1,"lift":{"pass_rate_delta_pts":31.82,"pass_rate_pct":86.4,"total_cases":22,"passed_cases":19,"tokens_delta_pct":70.5,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-14T18:49:04.575100+00:00"},"skill_score":null,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":31.82,"with_pass_pct":86.4,"without_pass_pct":54.5,"tokens_delta_pct":70.5,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"mixed","never_hurt":true,"completed_at":"2026-08-14T18:49:04.575100+00:00","run_id":"b9423b08-c8e9-4676-9162-97d358193939","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"Apache-2.0","install_count":0,"manifest_hash":"a2979d883cf04db0a53237d5e4d26394175dcef9211e5dd8662f0ae82b8b5082","raw_url":"https://app.decimal.ai/s/agentscope-ai-eval-design/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/agentscope-ai-eval-design"}