{"slug":"kunanonj-cursor-plugin-posthog-exploring-llm-evaluations","source_name":"kunanonj/cursor-plugin-posthog-exploring-llm-evaluations","name":"Kunanonj/Cursor Plugin Posthog Exploring LLM Evaluations","description":"Investigate AI observability evaluations of both types — `hog` (deterministic code-based) and `llm_judge` (LLM-prompt-based). Find existing evaluations, inspect their configuration, run them against specific generations, query individual pass/fail results, and generate AI-powered summaries of patterns across many runs. Use when the user asks to debug why an evaluation is failing, surface common fa","version":1,"lift":{"pass_rate_delta_pts":65.22,"pass_rate_pct":82.6,"total_cases":23,"passed_cases":19,"tokens_delta_pct":186.4,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-08T09:22:59.492896+00:00"},"skill_score":0.8261,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":65.22,"with_pass_pct":82.6,"without_pass_pct":17.4,"tokens_delta_pct":186.4,"turns_delta_pct":0,"total_cases":23,"cases_aggregated":20,"verdict":"mixed","never_hurt":false,"completed_at":"2026-08-08T09:22:59.492896+00:00","run_id":"af0a0372-3e95-4884-bbc7-422046943b2a","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT","install_count":0,"manifest_hash":"970b96e45049725795e88e2be9eb76b412b76c68fca158811ba7640549634d91","raw_url":"https://app.decimal.ai/s/kunanonj-cursor-plugin-posthog-exploring-llm-evaluations/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/kunanonj-cursor-plugin-posthog-exploring-llm-evaluations"}