{"slug":"awslabs-model-evaluation","source_name":"awslabs/model-evaluation","name":"Awslabs/Model Evaluation","description":"Generates python code that evaluates SageMaker models. Supports two evaluation types: LLM-as-Judge and Custom Scorer. Use when the user says \"evaluate my model\", \"run a benchmark\", \"test model performance\", \"how did my model perform\", \"compare models\", or other similar requests.","version":1,"lift":{"pass_rate_delta_pts":9.09,"pass_rate_pct":50,"total_cases":22,"passed_cases":11,"tokens_delta_pct":-21.8,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-16T21:33:08.808554+00:00"},"skill_score":null,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":9.09,"with_pass_pct":50,"without_pass_pct":40.9,"tokens_delta_pct":-21.8,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"mixed","never_hurt":false,"completed_at":"2026-08-16T21:33:08.808554+00:00","run_id":"ba8a89eb-2d59-471b-b680-75270cf00c23","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"Apache-2.0","install_count":0,"manifest_hash":"cb1d05f5d0e2cf9b4075bd066b92ad90a3a01e39f4888afbfcc7912c18e1662c","raw_url":"https://app.decimal.ai/s/awslabs-model-evaluation/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/awslabs-model-evaluation"}