[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"glossary-reward-hacking::en":3,"gloss-cluster-reward-hacking::en":23,"gloss-next-reward-hacking::en":9},{"slug":4,"category":5,"name":6,"definition":7,"meta_desc":8,"faq":9,"schema_markup":9,"related":10},"reward-hacking","core-ai","Reward Hacking","Reward hacking is when a model learns to maximize the measurable signal it's trained or evaluated on rather than the outcome you actually wanted. If the reward favors long answers, it pads; if a reward model likes confident tone, it sounds confident whether or not it's right; if an eval can be passed with a superficial trick, it finds the trick. The literal target and the real goal diverge, and the model optimizes the target. For builders this shows up whenever you drive behavior with a proxy metric — an automated grader, an engagement score, a thumbs-up rate. Optimize hard enough and you get outputs that score well but serve users worse. Practical note: treat every metric as gameable, especially LLM-as-judge scores. Combine automated metrics with spot human review, watch for outputs that satisfy the letter of your rubric while missing its intent, and change or diversify your evals periodically so the system can't quietly overfit to a single measurable proxy.","Reward hacking is a model maximizing the measurable signal instead of the outcome you wanted — padding answers, sounding confident, gaming the eval.",null,[11,14,17,20],{"slug":12,"name":13},"alignment-tax","Alignment Tax",{"slug":15,"name":16},"llm-benchmark","LLM Benchmark",{"slug":18,"name":19},"red-teaming","Red-Teaming",{"slug":21,"name":22},"rlhf","Reinforcement Learning from Human Feedback (RLHF)",[24,28,30,34,37,40,43,46,49,52,55,58],{"slug":25,"category":5,"name":26,"updated_at":27},"agentic","Agentic AI","2026-08-24T02:46:36+00:00",{"slug":12,"category":5,"name":13,"updated_at":29},"2026-08-24T02:46:37+00:00",{"slug":31,"category":5,"name":32,"updated_at":33},"artificial-intelligence","Artificial Intelligence (AI)","2026-08-24T02:46:38+00:00",{"slug":35,"category":5,"name":36,"updated_at":27},"attention","Attention",{"slug":38,"category":5,"name":39,"updated_at":33},"beam-search","Beam Search",{"slug":41,"category":5,"name":42,"updated_at":29},"benchmark-contamination","Benchmark Contamination",{"slug":44,"category":5,"name":45,"updated_at":29},"catastrophic-forgetting","Catastrophic Forgetting",{"slug":47,"category":5,"name":48,"updated_at":33},"computer-vision","Computer Vision",{"slug":50,"category":5,"name":51,"updated_at":29},"constitutional-ai","Constitutional AI",{"slug":53,"category":5,"name":54,"updated_at":27},"context-window","Context Window",{"slug":56,"category":5,"name":57,"updated_at":33},"deep-learning","Deep Learning",{"slug":59,"category":5,"name":60,"updated_at":27},"diffusion-model","Diffusion Model"]