{"slug":"aperivue-model-evaluation","source_name":"aperivue/model-evaluation","name":"Aperivue/Model Evaluation","description":"Compute and report task-correct held-out metrics for a trained medical-imaging model — segmentation (Dice plus a boundary metric such as HD95 or NSD, per structure), classification (AUROC plus AUPRC and sensitivity/specificity with bootstrap CIs at the deployment prevalence), detection (FROC or mAP with a stated IoU criterion), interactive/promptable segmentation (the interaction-count, convergence, and per-case-time axes a static Dice omits), or generative/synthesis image evaluation (similarity","version":1,"lift":{"pass_rate_delta_pts":0,"pass_rate_pct":81.8,"total_cases":22,"passed_cases":18,"tokens_delta_pct":32.4,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-24T22:27:11.561126+00:00"},"skill_score":null,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":0,"with_pass_pct":81.8,"without_pass_pct":81.8,"tokens_delta_pct":32.4,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":20,"verdict":"mixed","never_hurt":false,"completed_at":"2026-08-24T22:27:11.561126+00:00","run_id":"e4a270d3-b94a-42ea-8b09-29e143bbb28b","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT","install_count":0,"manifest_hash":"052739377019dae235f5110d8810a4359f5e96244749127a7ec6270aa1a0c224","raw_url":"https://app.decimal.ai/s/aperivue-model-evaluation/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/aperivue-model-evaluation"}