{"slug":"agentscope-ai-auto-arena","source_name":"agentscope-ai/auto-arena","name":"Agentscope AI/Auto Arena","description":"Automatically evaluate and compare multiple AI models or agents without pre-existing test data. Generates test queries from a task description, collects responses from all target endpoints, auto-generates evaluation rubrics, runs pairwise comparisons via a judge model, and produces win-rate rankings with reports and charts. Supports checkpoint resume, incremental endpoint addition, and judge model hot-swap. Use when the user asks to compare, benchmark, or rank multiple models or agents on a cust","version":1,"lift":{"pass_rate_delta_pts":54.55,"pass_rate_pct":90.9,"total_cases":22,"passed_cases":20,"tokens_delta_pct":81.4,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-14T18:30:02.655039+00:00"},"skill_score":0.9091,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":54.55,"with_pass_pct":90.9,"without_pass_pct":36.4,"tokens_delta_pct":81.4,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"mixed","never_hurt":false,"completed_at":"2026-08-14T18:30:02.655039+00:00","run_id":"d4053764-ca84-4672-89a8-93a99393e3ae","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"Apache-2.0","install_count":0,"manifest_hash":"8eb350c1d7908d6838241ed591bb41288d97b73bd96da2597839230d3df6ba55","raw_url":"https://app.decimal.ai/s/agentscope-ai-auto-arena/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/agentscope-ai-auto-arena"}