{"slug":"jpoindexter-inference-caching-and-kv","source_name":"jpoindexter/inference-caching-and-kv","name":"Jpoindexter/Inference Caching And Kv","description":"Reference-grade guide to caching in LLM inference — provider prompt caching (Anthropic cache_control breakpoints, OpenAI automatic prefix caching, Gemini implicit/explicit), semantic caching, and KV-cache internals & management (PagedAttention/vLLM, RadixAttention/SGLang, eviction, quantized KV, memory pressure, multi-tenant safety). Concrete numbers, formulas, failure modes.","version":1,"lift":{"pass_rate_delta_pts":4.55,"pass_rate_pct":100,"total_cases":22,"passed_cases":22,"tokens_delta_pct":222.4,"turns_delta_pct":0,"verdict":"pass","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-03T12:39:42.767293+00:00"},"skill_score":1,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":4.55,"with_pass_pct":100,"without_pass_pct":95.5,"tokens_delta_pct":222.4,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"pass","never_hurt":true,"completed_at":"2026-08-03T12:39:42.767293+00:00","run_id":"1a16eddf-7587-4e59-b4eb-461b0751c35f","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":null,"install_count":0,"manifest_hash":"5987337c794379d5a33f7925eeab3eb7bbf91e07497d196709dcd4a09a09141a","raw_url":"https://app.decimal.ai/s/jpoindexter-inference-caching-and-kv/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/jpoindexter-inference-caching-and-kv"}