{"slug":"openlair-awq-quantization","source_name":"openlair/awq-quantization","name":"Openlair/Awq Quantization","description":"Activation-aware weight quantization for 4-bit LLM compression with 3x speedup and minimal accuracy loss. Use when deploying large models (7B-70B) on limited GPU memory, when you need faster inference than GPTQ with better accuracy preservation, or for instruction-tuned and multimodal models. MLSys 2024 Best Paper Award winner.","version":1,"lift":{"pass_rate_delta_pts":45.45,"pass_rate_pct":95.5,"total_cases":22,"passed_cases":21,"tokens_delta_pct":105.2,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-07T19:44:15.842048+00:00"},"skill_score":0.9545,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":45.45,"with_pass_pct":95.5,"without_pass_pct":50,"tokens_delta_pct":105.2,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"mixed","never_hurt":true,"completed_at":"2026-08-07T19:44:15.842048+00:00","run_id":"c33b0adf-758e-4ddf-b0e5-f6ef02462fe8","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT","install_count":0,"manifest_hash":"d2f6cc2324520fe336d3cbea7a458f2250b8ca3c327a9f1094e59e61a58bc845","raw_url":"https://app.decimal.ai/s/openlair-awq-quantization/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/openlair-awq-quantization"}