{"slug":"mkurman-llama-cpp","source_name":"mkurman/llama-cpp","name":"Mkurman/Llama Cpp","description":"LLM inference in C/C++ with Python bindings. GPU acceleration via CUDA/Metal/Vulkan, 2-8 bit quantization (GGUF), KV cache, and grammar-based sampling. Run Llama, Mistral, Gemma, Phi locally.","version":1,"lift":{"pass_rate_delta_pts":4.55,"pass_rate_pct":100,"total_cases":22,"passed_cases":22,"tokens_delta_pct":-15.1,"turns_delta_pct":0,"verdict":"pass","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-21T09:49:34.145865+00:00"},"skill_score":1,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":4.55,"with_pass_pct":100,"without_pass_pct":95.5,"tokens_delta_pct":-15.1,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"pass","never_hurt":true,"completed_at":"2026-08-21T09:49:34.145865+00:00","run_id":"ad1417a0-87d5-4f52-8276-4a2ae1ed3d82","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT","install_count":0,"manifest_hash":"cf56d149bda0abc390e55953758ab7b9103cb0fc738364d608ee48c7775bf6a2","raw_url":"https://app.decimal.ai/s/mkurman-llama-cpp/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/mkurman-llama-cpp"}