{"slug":"openlair-fine-tuning-with-trl","source_name":"openlair/fine-tuning-with-trl","name":"Openlair/Fine Tuning With Trl","description":"Fine-tune LLMs using reinforcement learning with TRL - SFT for instruction tuning, DPO for preference alignment, PPO/GRPO for reward optimization, and reward model training. Use when need RLHF, align model with preferences, or train from human feedback. Works with HuggingFace Transformers.","version":1,"lift":{"pass_rate_delta_pts":4.55,"pass_rate_pct":81.8,"total_cases":22,"passed_cases":18,"tokens_delta_pct":90.7,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-08-07T20:01:51.738940+00:00"},"skill_score":0.8182,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":4.55,"with_pass_pct":81.8,"without_pass_pct":77.3,"tokens_delta_pct":90.7,"turns_delta_pct":0,"total_cases":22,"cases_aggregated":22,"verdict":"mixed","never_hurt":false,"completed_at":"2026-08-07T20:01:51.738940+00:00","run_id":"c6c27aaf-b8c3-41a4-ac8e-d3f73e11bb51","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":"safe","content_status":"clean","indexable":true},"license":"MIT","install_count":0,"manifest_hash":"473638ac1635112e9bee5d2b662119ae69c565cbf0fc41f72eecb4f188b6d8ee","raw_url":"https://app.decimal.ai/s/openlair-fine-tuning-with-trl/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/openlair-fine-tuning-with-trl"}