{"slug":"openlair-serving-llms-vllm","source_name":"openlair/serving-llms-vllm","name":"Openlair/Serving Llms Vllm","description":"Serves LLMs with high throughput using vLLM's PagedAttention and continuous batching. Use when deploying production LLM APIs, optimizing inference latency/throughput, or serving models with limited GPU memory. Supports OpenAI-compatible endpoints, quantization (GPTQ/AWQ/FP8), and tensor parallelism.","version":1,"lift":{"pass_rate_delta_pts":21.74,"pass_rate_pct":95.7,"total_cases":23,"passed_cases":22,"tokens_delta_pct":58.5,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-08T00:41:39.081575+00:00"},"skill_score":0.9565,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":21.74,"with_pass_pct":95.7,"without_pass_pct":73.9,"tokens_delta_pct":58.5,"turns_delta_pct":0,"total_cases":23,"cases_aggregated":23,"verdict":"mixed","never_hurt":true,"completed_at":"2026-08-08T00:41:39.081575+00:00","run_id":"0303d7e8-b116-4725-8609-766cb3ee1215","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT","install_count":0,"manifest_hash":"2d363936de5f1e284e12885ea6f833de7b8dc36584881985cfd419d8627f8eeb","raw_url":"https://app.decimal.ai/s/openlair-serving-llms-vllm/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/openlair-serving-llms-vllm"}