@article{JMLR:v23:21-054,
  author  = {Alan Chan and Hugo Silva and Sungsu Lim and Tadashi Kozuno and A. Rupam Mahmood and Martha White},
  title   = {Greedification Operators for Policy Optimization: Investigating Forward and Reverse KL Divergences},
  journal = {Journal of Machine Learning Research},
  year    = {2022},
  volume  = {23},
  number  = {253},
  pages   = {1--79},
  url     = {http://jmlr.org/papers/v23/21-054.html}
}