{"slug":"mkurman-cleanrl","source_name":"mkurman/cleanrl","name":"Mkurman/Cleanrl","description":"Single-file deep reinforcement learning implementations (CleanRL). High-quality standalone implementations of PPO, DQN, C51, SAC, DDPG, TD3 with research-friendly features. Each algorithm is a self-contained file with ~300-500 lines. Includes Atari, MuJoCo, Procgen, PettingZoo multi-agent, and JAX variants. Use for RL algorithm reference, rapid prototyping, and understanding implementation details.","version":1,"lift":{"pass_rate_delta_pts":26.09,"pass_rate_pct":91.3,"total_cases":23,"passed_cases":21,"tokens_delta_pct":90.9,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-21T08:52:58.453382+00:00"},"skill_score":0.913,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":26.09,"with_pass_pct":91.3,"without_pass_pct":65.2,"tokens_delta_pct":90.9,"turns_delta_pct":0,"total_cases":23,"cases_aggregated":23,"verdict":"mixed","never_hurt":false,"completed_at":"2026-08-21T08:52:58.453382+00:00","run_id":"6ada3a39-dc89-4772-bf13-803abf2b8a45","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT license","install_count":0,"manifest_hash":"b9d22878e3edbd725f5fef275f4c74a45d2acbe91b723a49ad3664f6d5044385","raw_url":"https://app.decimal.ai/s/mkurman-cleanrl/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/mkurman-cleanrl"}