@misc{indiciae1927b01b7be9, title = {Stackelberg Learning from Human Feedback: Preference Optimization as a Sequential Game}, author = {Barna Pásztor and Thomas Kleine Buening and Andreas Krause}, year = {2025}, url = {https://arxiv.org/abs/2512.16626}, note = {Source identifier: 2512.16626} }