{"slug":"brycewang-stanford-cav-experiments","source_name":"brycewang-stanford/cav-experiments","name":"Brycewang Stanford/Cav Experiments","description":"Use when designing or auditing a CAV (Computer Aided Verification) empirical evaluation, covering standard benchmark sets (SV-COMP/SMT-COMP/HWMCC/VNN-COMP), fair baseline solvers with pinned versions and equal resource limits, timeout-dominated comparisons, soundness cross-checks and proof witnesses, cactus/scatter reporting, and matching evidence to the shape of each verification claim.","version":1,"lift":{"pass_rate_delta_pts":72.73,"pass_rate_pct":86.4,"total_cases":22,"passed_cases":19,"tokens_delta_pct":-6.8,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-15T20:30:06.058667+00:00"},"skill_score":0.8636,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":72.73,"with_pass_pct":86.4,"without_pass_pct":13.6,"tokens_delta_pct":-6.8,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"mixed","never_hurt":false,"completed_at":"2026-08-15T20:30:06.058667+00:00","run_id":"42b230b2-f2b3-47cc-bf53-24834fbe9e05","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT","install_count":0,"manifest_hash":"4908293b4422d42fcc5ae4067804e7590b61de097c62f4e1740ebe59da73468f","raw_url":"https://app.decimal.ai/s/brycewang-stanford-cav-experiments/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/brycewang-stanford-cav-experiments"}