{"slug":"mims-harvard-tooluniverse-diagnostic-test-evaluation","source_name":"mims-harvard/tooluniverse-diagnostic-test-evaluation","name":"Mims Harvard/Tooluniverse Diagnostic Test Evaluation","description":"Diagnostic test / biomarker accuracy — sensitivity, specificity, PPV, NPV, likelihood ratios, accuracy from a 2x2 table; ROC curve, AUC, and the optimal cutoff (Youden) for a continuous biomarker; and post-test probability via Bayes. Use when you have test results vs a gold standard (binary 2x2, or a continuous score + true labels) and need to judge how good the test is, pick a threshold, or compute the probability of disease given a result. Emphasizes the prevalence-dependence of PPV/NPV.","version":1,"lift":{"pass_rate_delta_pts":40.91,"pass_rate_pct":95.5,"total_cases":22,"passed_cases":21,"tokens_delta_pct":49.5,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-06T15:35:23.318032+00:00"},"skill_score":0.9545,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":40.91,"with_pass_pct":95.5,"without_pass_pct":54.5,"tokens_delta_pct":49.5,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"mixed","never_hurt":true,"completed_at":"2026-08-06T15:35:23.318032+00:00","run_id":"c58eb0a9-7f31-4584-9e9b-775e5aa834ce","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"Apache-2.0","install_count":0,"manifest_hash":"ebfd4a3014c8f8d4dec1928e4f5f13b47aec525920866228da03554f56f925ac","raw_url":"https://app.decimal.ai/s/mims-harvard-tooluniverse-diagnostic-test-evaluation/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/mims-harvard-tooluniverse-diagnostic-test-evaluation"}