{"slug":"agentscope-ai-rl-reward","source_name":"agentscope-ai/rl-reward","name":"Agentscope AI/Rl Reward","description":"Build RL reward signals using the OpenJudge framework. Covers choosing between pointwise and pairwise reward strategies based on RL algorithm, task type, and cost; aggregating multi-dimensional pointwise scores into a scalar reward; pairwise tournament reward for GRPO on subjective tasks (net win rate across group rollouts); generating preference pairs for DPO/RLAIF; and normalizing scores for training stability. Use when building reward models, scoring rollouts for GRPO/REINFORCE, generating pr","version":1,"lift":{"pass_rate_delta_pts":36.36,"pass_rate_pct":95.5,"total_cases":22,"passed_cases":21,"tokens_delta_pct":28.1,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-14T18:38:49.548720+00:00"},"skill_score":0.9545,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":36.36,"with_pass_pct":95.5,"without_pass_pct":59.1,"tokens_delta_pct":28.1,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":21,"verdict":"mixed","never_hurt":false,"completed_at":"2026-08-14T18:38:49.548720+00:00","run_id":"2db30103-c1a0-4fab-9a4d-e3527f9bd592","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"Apache-2.0","install_count":0,"manifest_hash":"b38134de3c6959b3fe206faa787fa58e75dd908c2dad6a4661ebe40006f2d17a","raw_url":"https://app.decimal.ai/s/agentscope-ai-rl-reward/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/agentscope-ai-rl-reward"}