Reference. Direct preference optimization: Your language model is secretly a reward model [rafailov2023direct]

@article{rafailov2023direct,
  title = {Direct Preference Optimization: Your Language Model Is Secretly a Reward Model},
  author = {Rafailov, Rafael and Sharma, Archit and Mitchell, Eric and Ermon, Stefano and Manning, Christopher D. and Finn, Chelsea},
  journal = {Advances in Neural Information Processing Systems},
  year = {2023},
  url = {https://arxiv.org/abs/2305.18290}
}