{"slug":"agentscope-ai-ref-hallucination-arena","source_name":"agentscope-ai/ref-hallucination-arena","name":"Agentscope AI/Ref Hallucination Arena","description":"Benchmark LLM reference recommendation capabilities by verifying every cited paper against Crossref, PubMed, arXiv, and DBLP. Measures hallucination rate, per-field accuracy (title/author/year/DOI), discipline breakdown, and year constraint compliance. Supports tool-augmented (ReAct + web search) mode. Use when the user asks to evaluate, benchmark, or compare models on academic reference hallucination, literature recommendation quality, or citation accuracy.","version":1,"lift":{"pass_rate_delta_pts":78.26,"pass_rate_pct":100,"total_cases":23,"passed_cases":23,"tokens_delta_pct":56.7,"turns_delta_pct":0,"verdict":"pass","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-14T18:32:09.549848+00:00"},"skill_score":1,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":78.26,"with_pass_pct":100,"without_pass_pct":21.7,"tokens_delta_pct":56.7,"turns_delta_pct":0,"total_cases":23,"cases_aggregated":21,"verdict":"pass","never_hurt":true,"completed_at":"2026-08-14T18:32:09.549848+00:00","run_id":"323eb236-5aca-4619-a4c3-801b1e15f2d7","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"Apache-2.0","install_count":0,"manifest_hash":"e84fa4fa80aea77f178806308829521ea9e4e91364648ef4653744bed45dbdc3","raw_url":"https://app.decimal.ai/s/agentscope-ai-ref-hallucination-arena/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/agentscope-ai-ref-hallucination-arena"}