{"slug":"jpoindexter-quantization-and-model-compression","source_name":"jpoindexter/quantization-and-model-compression","name":"Jpoindexter/Quantization And Model Compression","description":"Reference-grade guide to shrinking and speeding up LLMs without retraining from scratch — numeric formats (FP8/INT8/INT4), PTQ methods (GPTQ, AWQ, SmoothQuant, bitsandbytes NF4, GGUF k-quants), KV-cache quantization, speculative decoding (Medusa/EAGLE/n-gram), and distillation — with concrete numbers, when each fits, and the quality cliffs.","version":1,"lift":{"pass_rate_delta_pts":9.09,"pass_rate_pct":90.9,"total_cases":22,"passed_cases":20,"tokens_delta_pct":195.5,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-03T12:59:18.931947+00:00"},"skill_score":0.9091,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":9.09,"with_pass_pct":90.9,"without_pass_pct":81.8,"tokens_delta_pct":195.5,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"mixed","never_hurt":true,"completed_at":"2026-08-03T12:59:18.931947+00:00","run_id":"f40eb69f-3e73-40f0-bcac-837d6783a33a","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":null,"install_count":0,"manifest_hash":"8aa71d876c9f33c89081b882e91f43f899edf624edafbf2853a0ba1ab29e59df","raw_url":"https://app.decimal.ai/s/jpoindexter-quantization-and-model-compression/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/jpoindexter-quantization-and-model-compression"}