The exact contents of citations.db main_text.text for this paper — one flattened LaTeX string, title through conclusion, appendix excluded, unmodified except for removing email addresses. This is what our citation measures are computed over.
6,241 characters
Optimal Policy Learning for Multi-Action Treatment with Risk Preference using Stata
\inserttype[]{article}
\author{Giovanni Cerulli}{
Giovanni Cerulli\\IRCrES-CNR\\Rome, Italy\\[email removed]
}
\title[Optimal Policy Learning for Multi-Action Treatment]
{Optimal Policy Learning for Multi-Action Treatment with Risk Preference using Stata}
\maketitle
\begin{abstract}
This paper presents the Stata community-distributed command \texttt{opl\_ma\_fb} (and the companion command \texttt{opl\_ma\_vf}), for implementing the \textit{first-best} Optimal Policy Learning (OPL) algorithm to estimate the best treatment assignment given the observation of an outcome, a multi-action (or multi-arm) treatment, and a set of observed covariates (features). It allows for different risk preferences in decision-making (i.e., risk-neutral, linear risk-averse, and quadratic risk-averse), and provides a graphical representation of the optimal policy, along with an estimate of the maximal welfare (i.e., the value-function estimated at optimal policy) using regression adjustment (RA), inverse-probability weighting (IPW), and doubly robust (DR) formulas.
\keywords{\inserttag Optimal Policy Learning, Multi-Action Treatment, Risk Preference}
\end{abstract}
\input sj_dddm.tex
\input stata_dddm.tex
\bibliographystyle{sj}
\nocite*{}
\bibliographystyle{chicago}
\begin{thebibliography}{99}
\bibitem[Auer et al.(2002)]{Auer2002}
Auer, P., N. Cesa-Bianchi, and P. Fischer. 2002. Finite-time analysis of the multiarmed bandit problem. \textit{Machine Learning} 47(2–3): 235–256.
\bibitem[Athey and Wager(2021)]{Athey2021}
Athey, S., and S. Wager. 2021. Policy learning with observational data. \textit{Econometrica} 89(1): 133–161. https://doi.org/10.3982/ECTA15732.
\bibitem[Bertsimas and Kallus(2020)]{Bertsimas2020}
Bertsimas, D., and N. Kallus. 2020. From predictive to prescriptive analytics. \textit{Management Science} 66(3): 1025–1044.
\bibitem[Bhattacharya and Dupas(2012)]{Bhattacharya2012}
Bhattacharya, D., and P. Dupas. 2012. Inferring welfare maximizing treatment assignment under budget constraints. \textit{Journal of Econometrics} 167(1): 168–196. https://doi.org/10.1016/j.jeconom.2011.11.007.
\bibitem[Cattaneo, Drukker, and Holland(2013)]{Catt2023}
Cattaneo, M. D., D. M. Drukker, and A. D. Holland. 2013. Estimation of multivalued treatment effects under conditional independence. Stata Journal 13(3): 407–450. https://doi.org/10.1177/1536867X1301300301
\bibitem[Cerulli(2023)]{Cerulli2023}
Cerulli, G. 2023. Optimal treatment assignment of a threshold-based policy: Empirical protocol and related issues. \textit{Applied Economics Letters} 30(8): 1010–1017. https://doi.org/10.1080/13504851.2022.2032577.
\bibitem[Cerulli(2024)]{Cerulli2024}
Cerulli, G. 2024. Optimal policy learning with observational data in multi-action scenarios: Estimation, risk preference, and potential failures. \textit{arXiv preprint} arXiv:2403.20250 [stat.ML]. https://arxiv.org/abs/2403.20250.
\bibitem[Cerulli(2025)]{Cerulli2025}
Cerulli, G. 2025. Optimal policy learning using Stata. \textit{The Stata Journal} 25(2): 309–343. https://doi.org/10.1177/1536867X251341143.
\bibitem[Cassel, Mannor, and Zeevi(2023)]{Cassel2023}
Cassel, A., S. Mannor, and A. Zeevi. 2023. A general framework for bandit problems beyond cumulative objectives. \textit{Mathematics of Operations Research} 48(4): 2196–2232.
\bibitem[Chandak, Shankar, and Thomas(2021)]{Chandak2021}
Chandak, Y., S. Shankar, and P. S. Thomas. 2021. High confidence off-policy (or counterfactual) variance estimation. In \textit{Proceedings of the Thirty-Fifth AAAI Conference on Artificial Intelligence}.
\bibitem[Dudik, Langford and Li(2011)]{Dudik2011}
Dudik M, Langford J, Li L (2011) Doubly robust policy evaluation and learning. \textit{Proceedings of the 28th International Conference on Machine Learning}, 1097–1104.
\bibitem[Kitagawa and Tetenov(2018)]{Kitagawa2018}
Kitagawa, T., and A. Tetenov. 2018. Who should be treated? Empirical welfare maximization methods for treatment choice. \textit{Econometrica} 86(2): 591–616. https://doi.org/10.3982/ECTA13288.
\bibitem[Sani, Lazaric, and Munos(2012)]{Sani2012}
Sani, A., A. Lazaric, and R. Munos. 2012. Risk-aversion in multi-armed bandits. \textit{Advances in Neural Information Processing Systems} 25.
\bibitem[Silva et al.(2022)]{Silva2022}
Silva, N., H. Werneck, T. Silva, A. C. M. Pereira, and L. Rocha. 2022. Multi-armed bandits in recommendation systems: A survey of the state-of-the-art and future directions. \textit{Expert Systems with Applications} 197: 116669.
\bibitem[Slivkins(2019)]{Slivkins2019}
Slivkins, A. 2019. Introduction to multi-armed bandits. \textit{Foundations and Trends in Machine Learning} 12(1–2): 1–286.
\bibitem[Tschernutter(2022)]{Tschernutter2022}
Tschernutter, D. 2022. \textit{Advances in Data-Driven Decision-Making: A Mathematical Optimization Perspective}. Doctoral Thesis, ETH Zurich, Zürich, Switzerland.
\bibitem[Wen and Li(2023)]{Wen2023}
Wen, R., and S. Li. 2023. Spatial decision support systems with automated machine learning: A review. \textit{ISPRS International Journal of Geo-Information} 12(1): 12.
\bibitem[Xin et al.(2020)]{Xin2020}
Xin, X., A. Karatzoglou, I. Arapakis, and J. M. Jose. 2020. Self-supervised reinforcement learning for recommender systems. In \textit{Proceedings of SIGIR 2020}: 931–940.
\bibitem[Zhou, Athey, and Wager(2023)]{Zhou2023}
Zhou, Z., S. Athey, and S. Wager. 2023. Offline multi-action policy learning: Generalization and optimization. \textit{Operations Research} 71(1): 148–183. https://doi.org/10.1287/opre.2022.2271.
\end{thebibliography}
\begin{aboutauthors}
Giovanni Cerulli is a senior researcher at the CNR-IRCrES, Research Institute on Sustainable
Economic Growth, National Research Council of Italy, Rome. His research interest is in applied
econometrics, with a special focus on causal inference and machine learning. He has developed
original causal inference models and provided several implementations. He is currently editor
in chief of the \textit{International Journal of Computational Economics and Econometrics}.
\end{aboutauthors}
\endinput
\clearpage