{"slug":"bilal140202-tokenwiseab-ab-test-a-task-across-model-tiers","source_name":"bilal140202/tokenwiseab-ab-test-a-task-across-model-tiers","name":"Bilal140202/Tokenwiseab Ab Test A Task Across Model Tiers","description":"Run an A/B test of the same task at multiple model tiers (Haiku, Sonnet, optionally Opus). Captures outputs, computes structural and semantic diffs, scores quality, writes a markdown comparison report. Use when the user wants to validate \"is Haiku good enough for this task class?\" or runs /tokenwise:ab \"<task description>\".","version":1,"lift":{"pass_rate_delta_pts":27.27,"pass_rate_pct":59.1,"total_cases":22,"passed_cases":13,"tokens_delta_pct":-11.4,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-08T16:35:13.554131+00:00"},"skill_score":0.5909,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":27.27,"with_pass_pct":59.1,"without_pass_pct":31.8,"tokens_delta_pct":-11.4,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"mixed","never_hurt":false,"completed_at":"2026-08-08T16:35:13.554131+00:00","run_id":"9e60c8a6-1536-4733-88fc-d8a7fb54eefc","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT","install_count":0,"manifest_hash":"aeb0205e1889219563d1448182783f1cd943ed4c942e3a9c1d2f889a87e3f6b6","raw_url":"https://app.decimal.ai/s/bilal140202-tokenwiseab-ab-test-a-task-across-model-tiers/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/bilal140202-tokenwiseab-ab-test-a-task-across-model-tiers"}