{"slug":"agentsope-agentsop-metric-design","source_name":"agentsope/agentsop-metric-design","name":"Agentsope/Agentsop Metric Design","description":"Decomposed, multi-criteria metric design for LLM pipelines. The metric IS the model — change the metric and the optimizer changes behavior. Decompose by default; bool during compile, float during eval; calibrate against human; mitigate judge bias. Search keywords: LLM-as-judge, llm as judge, eval metric, evaluation score, scoring function, rubric, RAGAS, G-Eval, judge bias, verbosity bias, how to evaluate LLM output.","version":1,"lift":{"pass_rate_delta_pts":50,"pass_rate_pct":100,"total_cases":22,"passed_cases":22,"tokens_delta_pct":254.3,"turns_delta_pct":0,"verdict":"pass","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-24T14:09:10.112367+00:00"},"skill_score":1,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":50,"with_pass_pct":100,"without_pass_pct":50,"tokens_delta_pct":254.3,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":20,"verdict":"pass","never_hurt":true,"completed_at":"2026-08-24T14:09:10.112367+00:00","run_id":"74192568-2da4-4637-8464-ee6af26ea5d7","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT","install_count":0,"manifest_hash":"83e45dfac0be83f9ae03ba5c5287c89fb2c8c2429980683028ab4a4cfb86a2b0","raw_url":"https://app.decimal.ai/s/agentsope-agentsop-metric-design/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/agentsope-agentsop-metric-design"}