{"slug":"openlair-evaluating-code-models","source_name":"openlair/evaluating-code-models","name":"Openlair/Evaluating Code Models","description":"Evaluates code generation models across HumanEval, MBPP, MultiPL-E, and 15+ benchmarks with pass@k metrics. Use when benchmarking code models, comparing coding abilities, testing multi-language support, or measuring code generation quality. Industry standard from BigCode Project used by HuggingFace leaderboards.","version":1,"lift":{"pass_rate_delta_pts":31.82,"pass_rate_pct":100,"total_cases":22,"passed_cases":22,"tokens_delta_pct":187.5,"turns_delta_pct":0,"verdict":"pass","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-07T19:12:42.496009+00:00"},"skill_score":1,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":31.82,"with_pass_pct":100,"without_pass_pct":68.2,"tokens_delta_pct":187.5,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":21,"verdict":"pass","never_hurt":true,"completed_at":"2026-08-07T19:12:42.496009+00:00","run_id":"dc81b7b0-e3a5-40d5-a741-38e98da64a90","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT","install_count":0,"manifest_hash":"38f1a45bc43d6ef1479909a6acbe7f2f73c9b195b371ec1d5b2823191afa9fc8","raw_url":"https://app.decimal.ai/s/openlair-evaluating-code-models/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/openlair-evaluating-code-models"}