{"slug":"jpoindexter-inference-performance","source_name":"jpoindexter/inference-performance","name":"Jpoindexter/Inference Performance","description":"Use when serving or optimizing LLM inference in production — diagnosing or improving TTFT/TPOT/throughput, choosing batching strategy, sizing GPUs, picking vLLM/TensorRT-LLM, or debugging low GPU utilization, TTFT spikes, and OOM. Covers prefill vs decode, the roofline, continuous batching, PagedAttention, chunked prefill, disaggregation, and FlashAttention.","version":1,"lift":{"pass_rate_delta_pts":-100,"pass_rate_pct":90.9,"total_cases":22,"passed_cases":20,"tokens_delta_pct":251.3,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-03T12:46:08.325621+00:00"},"skill_score":null,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":-100,"with_pass_pct":0,"without_pass_pct":100,"tokens_delta_pct":251.3,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":21,"verdict":"mixed","never_hurt":false,"completed_at":"2026-08-03T12:46:08.325621+00:00","run_id":"7593519f-7084-44d2-9af2-798dad1c57c5","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":null,"install_count":0,"manifest_hash":"e925f57f7f897faf1693113c21a5b9b0f90de04f3666a4da7bf944601f41db27","raw_url":"https://app.decimal.ai/s/jpoindexter-inference-performance/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/jpoindexter-inference-performance"}