{"slug":"openmatter-network-ai-model-outputs-audit","source_name":"openmatter-network/ai-model-outputs-audit","name":"Openmatter Network/AI Model Outputs Audit","description":"Use when auditing the scores an AI/ML personnel assessment produces — Component 6 of the Landers & Behrend (2023) framework. Covers evaluating the quality of model predictions: reliability (consistency over time and repeated administrations), validity evidence (do scores reflect the claimed constructs and predict the outcome), appropriateness of the cross-validation given generalizability claims, and subgroup differences across protected classes and their intersections. Triggers: \"evaluate AI as","version":1,"lift":{"pass_rate_delta_pts":4.55,"pass_rate_pct":90.9,"total_cases":22,"passed_cases":20,"tokens_delta_pct":40.5,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-03T14:10:04.925698+00:00"},"skill_score":0.9091,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":4.55,"with_pass_pct":90.9,"without_pass_pct":86.4,"tokens_delta_pct":40.5,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"mixed","never_hurt":true,"completed_at":"2026-08-03T14:10:04.925698+00:00","run_id":"00f08821-12bf-4d4b-a042-5e1f7bd5e175","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT","install_count":0,"manifest_hash":"8354b2875ea48141e577b4b9851e1ab8637ce65536926c8a821523a362244b95","raw_url":"https://app.decimal.ai/s/openmatter-network-ai-model-outputs-audit/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/openmatter-network-ai-model-outputs-audit"}