The exact contents of citations.db main_text.text for this paper — one flattened LaTeX string, title through conclusion, appendix excluded, unmodified except for removing email addresses. This is what our citation measures are computed over.
96,987 characters
Supplemental to Bootstrap-Based Inference for Cube Root Asymptotics
\title{\vspace{-0.25in}Supplemental to \textquotedblleft Bootstrap-Based Inference
for Cube Root Asymptotics\textquotedblright\thanks{Cattaneo gratefully
acknowledges financial support from the National Science Foundation through
grants SES-1459931 and SES-1947805, and Jansson gratefully acknowledges
financial support from the National Science Foundation through grants
SES-1459967 and SES-1947662 and the research support of CREATES (funded by the
Danish National Research Foundation under grant no. DNRF78).}\bigskip}
\author{Matias D. Cattaneo\thanks{Department of Operations Research and Financial
Engineering, Princeton University.}
\and Michael Jansson\thanks{Department of Economics, University of California at
Berkeley and CREATES.}
\and Kenichi Nagasawa\thanks{Department of Economics, University of Warwick.}}
\maketitle
\begin{abstract}
This supplemental appendix contains proofs and other theoretical results that
may be of independent interest. It also offers more details on the examples
and simulation evidence presented in the paper.
\end{abstract}
\setcounter{section}{0}
\setcounter
{subsection}{0}
\setcounter{equation}{0}
\setcounter{lemma}{0}
\setcounter{corollary}{0}
\setcounter{theorem}{0}
\setcounter{assumption}{0}
\setcounter{CorollaryMS}{0}
\setcounter{LemmaMS}{0}
\setcounter{CorollaryPMS}{0}
\setcounter{LemmaPMS}{0}
\setcounter{CorollaryERM}{0}
\setcounter{LemmaERM}{0}
\setcounter{CorollaryCMS}{0}
\setcounter{LemmaCMS}{0}
\thispagestyle{empty}
\setcounter{page}{0}
\newpage\setcounter{tocdepth}{2}
\tableofcontents\thispagestyle{empty}\setcounter{page}{0}
\setcounter{secnumdepth}{4}
\clearpage\setlength{\abovedisplayskip}{5pt}
\setlength{\belowdisplayskip}{5pt}
\section{Proofs of Main Results\label{[Section] Proofs}}
\subsection{Proof of Theorem 1}
As explained in the paper, Theorem 1 follows from ten technical lemmas. The
remainder of this subsection presents those lemmas and their proofs.
The first lemma can be used to show that $\mathbf{\hat{\theta}}_{n}$ is consistent.
\begin{lemma}
\label{[Lemma] Consistency}Suppose Condition CRA(i) holds. Then $\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0}=o_{\mathbb{P}}(1)$ if
\[
\hat{M}_{n}(\mathbf{\hat{\theta}}_{n})\geq\sup_{\mathbf{\theta}\in\mathbf{\Theta}}\hat{M}_{n}(\mathbf{\theta})-o_{\mathbb{P}}(1).
\]
\end{lemma}
\noindent\textbf{Proof of Lemma }\ref{[Lemma] Consistency}. It suffices to
show that every $\delta>0$ admits a constant $c_{\delta}>0$ such that
\begin{equation}
\mathbb{P}\left[ \hat{M}_{n}(\mathbf{\theta}_{0})-\sup_{\mathbf{\theta}
\in\mathbf{\Theta\setminus\Theta}_{0}^{\delta}}\hat{M}_{n}(\mathbf{\theta
})>c_{\delta}\right] \rightarrow1.\label{Consistency: Sufficient condition}
\end{equation}
By assumption, $\sup_{\mathbf{\theta}\in\mathbf{\Theta}}|M_{n}(\mathbf{\theta})-M_{0}(\mathbf{\theta})|=o(1).$ Also, by \citet*[Theorem 4.2]{Pollard_1989_SS},
\[
\sup_{\mathbf{\theta}\in\mathbf{\Theta}}|\hat{M}_{n}(\mathbf{\theta})-M_{n}(\mathbf{\theta})|=O_{\mathbb{P}}\left( \sqrt{\frac{\mathbb{E}[\bar{m}_{n}(\mathbf{z})^{2}]}{n}}\right) =O_{\mathbb{P}}\left( \frac{1}{\sqrt{nq_{n}}}\right) =o_{\mathbb{P}}(1).
\]
As a consequence, for any $\delta>0,$
\[
\hat{M}_{n}(\mathbf{\theta}_{0})-\sup_{\mathbf{\theta}\in\mathbf{\Theta\setminus\Theta}_{0}^{\delta}}\hat{M}_{n}(\mathbf{\theta})=M_{0}(\mathbf{\theta}_{0})-\sup_{\mathbf{\theta}\in\mathbf{\Theta\setminus\Theta}_{0}^{\delta}}M_{0}(\mathbf{\theta})+o_{\mathbb{P}}(1),
\]
so (\ref{Consistency: Sufficient condition}) is satisfied with $c_{\delta}=[M_{0}(\mathbf{\theta}_{0})-\sup_{\mathbf{\theta}\in\mathbf{\Theta\setminus\Theta}_{0}^{\delta}}M_{0}(\mathbf{\theta})]/2>0.\quad\blacksquare$\bigskip
Assuming the derivatives exist, let $\dot{M}_{n}(\mathbf{\theta})$ and
$\ddot{M}_{n}(\mathbf{\theta})$ denote $\partial M_{n}(\mathbf{\theta}
_{0})/\partial\mathbf{\theta}$ and $\partial^{2}M_{n}(\mathbf{\theta
})/\partial\mathbf{\theta}\partial\mathbf{\theta}^{\prime},$ respectively. If
$M_{n}$ is twice continuously differentiable on a neighborhood $\mathbf{\Theta
}_{n}$ of $\mathbf{\theta}_{0},$ then it follows from Taylor's theorem that
\begin{equation}
\left\vert M_{n}(\mathbf{\theta})-M_{n}(\mathbf{\theta}_{0})+\frac{1}
{2}(\mathbf{\theta}-\mathbf{\theta}_{0})^{\prime}\mathbf{H}_{n}(\mathbf{\theta
}-\mathbf{\theta}_{0})\right\vert \leq\dot{C}_{n}||\mathbf{\theta
}-\mathbf{\theta}_{0}||+\frac{1}{2}\ddot{C}_{n}||\mathbf{\theta}
-\mathbf{\theta}_{0}||^{2},\label{Quadratic approximation: M_n}
\end{equation}
for every $\mathbf{\theta\in\Theta}_{n},$ where $\mathbf{H}_{n}=-\ddot{M}
_{n}(\mathbf{\theta}_{0}),$ $\dot{C}_{n}=||\dot{M}_{n}(\mathbf{\theta}
_{0})||,$ and $\ddot{C}_{n}=\sup_{\mathbf{\theta}\in\mathbf{\Theta}_{n}
}||\ddot{M}_{n}(\mathbf{\theta})-\ddot{M}_{n}(\mathbf{\theta}_{0})||.$
As an immediate consequence of (\ref{Quadratic approximation: M_n}), we have
the following convergence result about $Q_{n}.$
\begin{lemma}
\label{[Lemma] Compact convergence of Q_n}Suppose Condition CRA(ii) holds.
Then $Q_{n}$ converges compactly to $\mathcal{Q}_{0};$ that is,
\[
\sup_{||\mathbf{s}||\leq K}\left\vert Q_{n}(\mathbf{s})-\mathcal{Q}
_{0}(\mathbf{s})\right\vert \rightarrow0
\]
for any $K>0.$
\end{lemma}
\noindent\textbf{Proof of Lemma }\ref{[Lemma] Compact convergence of Q_n}. Let
$K>0$ be given and suppose $n$ is large enough that $Kr_{n}^{-1}\leq\delta,$
where $\delta>0$ is as in Condition CRA(ii). Using
(\ref{Quadratic approximation: M_n}) with $\mathbf{\Theta}_{n}=\mathbf{\Theta
}_{0}^{Kr_{n}^{-1}},$ we have
\begin{align*}
\left\vert Q_{n}(\mathbf{s})-\mathcal{Q}_{0}(\mathbf{s})\right\vert &
=\left\vert r_{n}^{2}[M_{n}(\mathbf{\theta}_{0}+\mathbf{s}r_{n}^{-1}
)-M_{n}(\mathbf{\theta}_{0})]+\frac{1}{2}\mathbf{s}^{\prime}\mathbf{H}
_{0}\mathbf{s}\right\vert \\
& \leq\frac{1}{2}\left\vert \mathbf{s}^{\prime}(\mathbf{H}_{n}-\mathbf{H}
_{0})\mathbf{s}\right\vert +r_{n}\dot{C}_{n}||\mathbf{s}||+\frac{1}{2}\ddot
{C}_{n}||\mathbf{s}||^{2}=(K+K^{2})o(1)
\end{align*}
uniformly in $\mathbf{s}$ with $||\mathbf{s}||\leq K,$ where the last equality
uses $r_{n}\dot{C}_{n}=r_{n}||\dot{M}_{n}(\mathbf{\theta}_{0})||\rightarrow0$
along with the facts that
\[
\mathbf{H}_{n}-\mathbf{H}_{0}=-[\ddot{M}_{n}(\mathbf{\theta}_{0})-\ddot{M}_{0}(\mathbf{\theta}_{0})]\rightarrow0,\qquad\ddot{M}_{0}(\mathbf{\theta}_{0})=\frac{\partial^{2}}{\partial\mathbf{\theta}\partial\mathbf{\theta}^{\prime}}M_{0}(\mathbf{\theta}),
\]
and
\begin{align*}
\ddot{C}_{n} & =\sup_{\mathbf{\theta}\in\mathbf{\Theta}_{0}^{Kr_{n}^{-1}}}||\ddot{M}_{n}(\mathbf{\theta})-\ddot{M}_{n}(\mathbf{\theta}_{0})||\\
& \leq2\sup_{\mathbf{\theta}\in\mathbf{\Theta}_{0}^{Kr_{n}^{-1}}}||\ddot{M}_{n}(\mathbf{\theta})-\ddot{M}_{0}(\mathbf{\theta})||+\sup_{\mathbf{\theta}\in\mathbf{\Theta}_{0}^{Kr_{n}^{-1}}}||\ddot{M}_{0}(\mathbf{\theta})-\ddot{M}_{0}(\mathbf{\theta}_{0})||\rightarrow0.\quad\blacksquare
\end{align*}
The next lemma can be used to obtain the rate of convergence of $\mathbf{\hat
{\theta}}_{n}.$
\begin{lemma}
\label{[Lemma] Rate of convergence}Suppose Conditions CRA(ii)-(iii) hold. Then
$r_{n}(\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0})=O_{\mathbb{P}}(1) $ if
$\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0}=o_{\mathbb{P}}(1)$ and if
\[
\hat{M}_{n}(\mathbf{\hat{\theta}}_{n})\geq\sup_{\mathbf{\theta}\in
\mathbf{\Theta}}\hat{M}_{n}(\mathbf{\theta})-o_{\mathbb{P}}(r_{n}^{-2}).
\]
\end{lemma}
\noindent\textbf{Proof of Lemma }\ref{[Lemma] Rate of convergence}. For any
$\delta>0$ and any $K\in\mathbb{N},$ $\mathbb{P}[r_{n}||\mathbf{\hat{\theta}
}_{n}-\mathbf{\theta}_{0}||>2^{K}]$ is no greater than
\begin{gather*}
\mathbb{P}[\sup_{\mathbf{\theta}\in\mathbf{\Theta}}\hat{M}_{n}(\mathbf{\theta
})-\hat{M}_{n}(\mathbf{\hat{\theta}}_{n})\geq\delta r_{n}^{-2}]+\mathbb{P}
[||\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0}||>\delta/2]\\
+\sum_{j\geq K,2^{j}\leq\delta r_{n}}\mathbb{P}\left[ \sup_{2^{j-1}
<r_{n}||\mathbf{\theta}-\mathbf{\theta}_{0}||\leq2^{j}}\hat{M}_{n}
(\mathbf{\theta})-\hat{M}_{n}(\mathbf{\theta}_{0})\geq-\delta r_{n}
^{-2}\right] .
\end{gather*}
By assumption, the probabilities on the first line go to zero for any
$\delta>0.$ As a consequence, it suffices to show that the sum on the last
line can be made arbitrarily small (for large $n$) by making $\delta>0$ small
and $K$ large.
To do so, let $\delta>0$ be small enough so that Conditions CRA(ii)-(iii) are
satisfied and
\[
c(\delta)=\underset{n\rightarrow\infty}{\lim\inf}\frac{1}{16}[\lambda_{\min
}(\mathbf{H}_{n})-\ddot{C}_{n}^{\delta}]>0,
\]
where $\ddot{C}_{n}^{\delta}=\sup_{\mathbf{\theta}\in\mathbf{\Theta}
_{0}^{\delta}}||\ddot{M}_{n}(\mathbf{\theta})-\ddot{M}_{n}(\mathbf{\theta}
_{0})||$ and where $\lambda_{\min}(\cdot)$ denotes the minimal eigenvalue of
the argument. Then, for all $n$ large enough and for any pair $(j,K)^{\prime
}\in\mathbb{N}^{2}$ with $j\geq K,$ we have
\[
M_{n}(\mathbf{\theta}_{0})-\sup_{2^{j-1}<r_{n}||\mathbf{\theta}-\mathbf{\theta
}_{0}||\leq2^{j}}M_{n}(\mathbf{\theta})-\delta r_{n}^{-2}\geq2^{2j}
c_{n,K}(\delta)r_{n}^{-2},
\]
where $c_{n,K}(\delta)=[\lambda_{\min}(\mathbf{H}_{n})-\ddot{C}_{n}^{\delta
}]/8-2^{-K}r_{n}\dot{C}_{n}-2^{-2K}\delta$ and where the inequality uses the
following implication of (\ref{Quadratic approximation: M_n}): If
$\lambda_{\min}(\mathbf{H}_{n})-\ddot{C}_{n}^{\delta}\geq0$ and if
$\mathbf{\Theta}_{n}^{\prime}$ is a subset of $\mathbf{\Theta}_{n}, $ then
\[
M_{n}(\mathbf{\theta}_{0})-\sup_{\mathbf{\theta}\in\mathbf{\Theta}_{n}
^{\prime}}M_{n}(\mathbf{\theta})\geq\frac{1}{2}[\lambda_{\min}(\mathbf{H}
_{n})-\ddot{C}_{n}^{\delta}]\inf_{\mathbf{\theta}\in\mathbf{\Theta}
_{n}^{\prime}}||\mathbf{\theta}-\mathbf{\theta}_{0}||^{2}-\dot{C}_{n}
\sup_{\mathbf{\theta}\in\mathbf{\Theta}_{n}^{\prime}}||\mathbf{\theta
}-\mathbf{\theta}_{0}||.
\]
Choosing $n$ and $K$ large enough, we may assume that $c_{n,K}(\delta)\geq
c(\delta),$ in which case
\begin{align*}
& \sum_{j\geq K,2^{j}\leq\delta r_{n}}\mathbb{P}\left[ \sup_{2^{j-1}
<r_{n}||\mathbf{\theta}-\mathbf{\theta}_{0}||\leq2^{j}}\hat{M}_{n}
(\mathbf{\theta})-\hat{M}_{n}(\mathbf{\theta}_{0})\geq-\delta r_{n}
^{-2}\right] \\
& \leq\sum_{j\geq K,2^{j}\leq\delta r_{n}}\mathbb{P}\left[ \sup
_{2^{j-1}<r_{n}||\mathbf{\theta}-\mathbf{\theta}_{0}||\leq2^{j}}\{\hat{M}
_{n}(\mathbf{\theta})-\hat{M}_{n}(\mathbf{\theta}_{0})-M_{n}(\mathbf{\theta
})+M_{n}(\mathbf{\theta}_{0})\}\geq2^{2j}c_{n,K}(\delta)r_{n}^{-2}\right] \\
& \leq\sum_{j\geq K,2^{j}\leq\delta r_{n}}\mathbb{P}\left[ \sup
_{r_{n}||\mathbf{\theta}-\mathbf{\theta}_{0}||\leq2^{j}}||\hat{M}
_{n}(\mathbf{\theta})-\hat{M}_{n}(\mathbf{\theta}_{0})-M_{n}(\mathbf{\theta
})+M_{n}(\mathbf{\theta}_{0})||\geq2^{2j}c(\delta)r_{n}^{-2}\right] \\
& \leq\frac{r_{n}^{2}}{c(\delta)}\sum_{j\geq K,2^{j}\leq\delta r_{n}}
2^{-2j}\mathbb{E}\left[ \sup_{r_{n}||\mathbf{\theta}-\mathbf{\theta}
_{0}||\leq2^{j}}||\hat{M}_{n}(\mathbf{\theta})-\hat{M}_{n}(\mathbf{\theta}
_{0})-M_{n}(\mathbf{\theta})+M_{n}(\mathbf{\theta}_{0})||\right] ,
\end{align*}
where the last inequality uses the Markov inequality.
Under Condition CRA(iii), $q_{n}\sup_{0\leq\delta^{\prime}\leq\delta}\mathbb{E}[\bar{d}_{n}^{\delta^{\prime}}(\mathbf{z})^{2}/\delta^{\prime}]=O(1)$ and it follows from \citet*[Theorem 4.2]{Pollard_1989_SS} that the sum
on the last line is bounded by a constant multiple of
\[
r_{n}^{2}\sum_{j\geq K,2^{j}\leq\delta r_{n}}2^{-2j}\sqrt{\frac{\mathbb{E}[\bar{d}_{n}^{2^{j}/r_{n}}(\mathbf{z})^{2}]}{n}}\leq\sqrt{q_{n}\sup_{0\leq\delta^{\prime}\leq\delta}\mathbb{E}[\bar{d}_{n}^{\delta^{\prime}}(\mathbf{z})^{2}/\delta^{\prime}]}\sum_{j\geq K}2^{-3j/2},
\]
which can be made arbitrarily small by making $K$ large.$\quad\blacksquare$\bigskip
In combination, the next two lemmas can be used to show that $\hat{G}
_{n}\rightsquigarrow\mathcal{G}_{0}$ in the topology of uniform convergence on compacta.
\begin{lemma}
\label{[Lemma] FIDI convergence}Suppose Conditions CRA(iii)-(iv) hold and
suppose $Q_{n}(\mathbf{s})=o(\sqrt{n})$ for every $\mathbf{s}\in\mathbb{R}
^{d}.$ Then $\hat{G}_{n}$ converges to $\mathcal{G}_{0}$ in the sense of weak
convergence of finite-dimensional projections.
\end{lemma}
\noindent\textbf{Proof of Lemma }\ref{[Lemma] FIDI convergence}. Because
$\hat{G}_{n}(\mathbf{s})=n^{-1/2}\sum_{i=1}^{n}\psi_{n}(\mathbf{z}
_{i};\mathbf{s}),$ where
\[
\psi_{n}(\mathbf{z;s})=\sqrt{r_{n}q_{n}}[m_{n}(\mathbf{z},\mathbf{\theta}
_{0}+\mathbf{s}r_{n}^{-1})-m_{n}(\mathbf{z},\mathbf{\theta}_{0})-M_{n}
(\mathbf{\theta}_{0}+\mathbf{s}r_{n}^{-1})+M_{n}(\mathbf{\theta}_{0})]
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(\mathbf{\theta}_{0}+\mathbf{s}r_{n}^{-1}\in\mathbf{\Theta})
\]
the result follows from the Cram\'{e}r-Wold device if
\[
\mathbb{E}[\psi_{n}(\mathbf{z};\mathbf{s})\psi_{n}(\mathbf{z};\mathbf{t}
)]\rightarrow\mathcal{C}_{0}(\mathbf{s},\mathbf{t})\qquad\forall
\mathbf{s},\mathbf{t}\in\mathbb{R}^{d},
\]
and if the following Lyapunov condition is satisfied:
\[
\frac{1}{n}\mathbb{E}[\psi_{n}(\mathbf{z};\mathbf{s})^{4}]\rightarrow
0\qquad\forall\mathbf{s}\in\mathbb{R}^{d}.
\]
Let $\mathbf{s},\mathbf{t}\in\mathbb{R}^{d}$ be given and suppose without loss
of generality that $\mathbf{\theta}_{0}+\mathbf{s}r_{n}^{-1},\mathbf{\theta
}_{0}+\mathbf{t}r_{n}^{-1}\in\mathbf{\Theta.}$ Then, using $Q_{n}
(\mathbf{s})=o(\sqrt{n})$ and the representation
\[
\psi_{n}(\mathbf{z};\mathbf{s})=\sqrt{r_{n}q_{n}}[m_{n}(\mathbf{z}
,\mathbf{\theta}_{0}+\mathbf{s}r_{n}^{-1})-m_{n}(\mathbf{z},\mathbf{\theta
}_{0})]-\frac{1}{\sqrt{n}}Q_{n}(\mathbf{s}),
\]
we have
\begin{align*}
& \mathbb{E}[\psi_{n}(\mathbf{z};\mathbf{s})\psi_{n}(\mathbf{z};\mathbf{t})]\\
& =r_{n}q_{n}\mathbb{E}[\{m_{n}(\mathbf{z},\mathbf{\theta}_{0}+\mathbf{s}
r_{n}^{-1})-m_{n}(\mathbf{z},\mathbf{\theta}_{0})\}\{m_{n}(\mathbf{z}
,\mathbf{\theta}_{0}+\mathbf{t}r_{n}^{-1})-m_{n}(\mathbf{z},\mathbf{\theta
}_{0})\}]-\frac{1}{n}Q_{n}(\mathbf{s})Q_{n}(\mathbf{t})\\
& \rightarrow\mathcal{C}_{0}(\mathbf{s},\mathbf{t})
\end{align*}
and, using $\mathbb{E}[\bar{d}_{n}^{\delta_{n}}(\mathbf{z})^{4}]=o(q_{n}
^{-3}r_{n})$ (for $\delta_{n}=O(r_{n}^{-1})$),
\[
\frac{1}{16n}\mathbb{E}[\psi(\mathbf{z};\mathbf{s})^{4}]\leq\frac{r_{n}
^{2}q_{n}^{2}}{n}\mathbb{E}[|m_{n}(\mathbf{z},\mathbf{\theta}_{0}
+\mathbf{s}r_{n}^{-1})-m_{n}(\mathbf{z},\mathbf{\theta}_{0})|^{4}]+\frac
{1}{n^{3}}Q_{n}(\mathbf{s})^{4}=o\left( \frac{r_{n}^{3}}{nq_{n}}+\frac{1}
{n}\right) =o(1),
\]
as was to be shown.$\quad\blacksquare$\bigskip
\begin{lemma}
\label{[Lemma] Stochastic equicontinuity}Suppose Conditions CRA(iii) and
CRA(v) hold. Then $\{\hat{G}_{n}(\mathbf{s}):||\mathbf{s}||\leq K\}$ is
stochastically equicontinuous for every $K>0;$ that is,
\[
\sup_{\substack{||\mathbf{s-t}||\leq\Delta_{n} \\||\mathbf{s}||,||\mathbf{t}
||\leq K}}|\hat{G}_{n}(\mathbf{s})-\hat{G}_{n}(\mathbf{t})|\rightarrow
_{\mathbb{P}}0
\]
for any $K>0$ and for any $\Delta_{n}>0$ with $\Delta_{n}=o(1).$
\end{lemma}
\noindent\textbf{Proof of Lemma }\ref{[Lemma] Stochastic equicontinuity}. Let
$K>0$ be given. As in the proof of \citet*[Lemma 4.6]{Kim-Pollard_1990_AoS} and
using the fact that $q_{n}\delta_{n}^{-1}\mathbb{E}[\bar{d}_{n}^{\delta_{n}
}(\mathbf{z})^{2}]=O(1)$ (for $\delta_{n}=O(r_{n}^{-1})$), it suffices to show
that
\[
r_{n}\sup_{\substack{||\mathbf{s-t}||\leq\Delta_{n} \\||\mathbf{s}
||,||\mathbf{t}||\leq K}}\frac{q_{n}}{n}\sum_{i=1}^{n}d_{n}(\mathbf{z}
_{i};\mathbf{s},\mathbf{t})^{2}\rightarrow_{\mathbb{P}}0,
\]
where $d_{n}(\mathbf{z};\mathbf{s},\mathbf{t})=|m_{n}(\mathbf{z}
,\mathbf{\theta}_{0}+\mathbf{s}r_{n}^{-1})-m_{n}(\mathbf{z},\mathbf{\theta
}_{0}+\mathbf{t}r_{n}^{-1})|/2.$
For any $C>0$ and any $\mathbf{s},\mathbf{t}\in\mathbb{R}^{d}$ with
$||\mathbf{s}||,||\mathbf{t}||\leq K,$
\begin{align*}
\frac{q_{n}}{n}\sum_{i=1}^{n}d_{n}(\mathbf{z}_{i};\mathbf{s},\mathbf{t})^{2}
& \leq\frac{q_{n}}{n}\sum_{i=1}^{n}\bar{d}_{n}^{Kr_{n}^{-1}}(\mathbf{z}
_{i})^{2}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(q_{n}\bar{d}_{n}^{Kr_{n}^{-1}}(\mathbf{z}_{i})>C)\\
& +C\mathbb{E[}d_{n}(\mathbf{z};\mathbf{s},\mathbf{t})]\\
& +C\frac{1}{n}\sum_{i=1}^{n}\{d_{n}(\mathbf{z}_{i};\mathbf{s},\mathbf{t}
)-\mathbb{E[}d_{n}(\mathbf{z};\mathbf{s},\mathbf{t})]\},
\end{align*}
and therefore
\begin{align*}
r_{n}\mathbb{E}\left[ \sup_{\substack{||\mathbf{s-t}||\leq\Delta_{n}
\\||\mathbf{s}||,||\mathbf{t}||\leq K}}\frac{q_{n}}{n}\sum_{i=1}^{n}
d_{n}(\mathbf{z};\mathbf{s},\mathbf{t})^{2}\right] & \leq q_{n}
r_{n}\mathbb{E}\left[ \bar{d}_{n}^{Kr_{n}^{-1}}(\mathbf{z})^{2}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(q_{n}\bar{d}_{n}^{Kr_{n}^{-1}}(\mathbf{z})>C)\right] \\
& +Cr_{n}\sup_{\substack{||\mathbf{s-t}||\leq\Delta_{n} \\||\mathbf{s}
||,||\mathbf{t}||\leq K}}\mathbb{E[}d_{n}(\mathbf{z};\mathbf{s},\mathbf{t})]\\
& +Cr_{n}\mathbb{E}\left[ \sup_{\substack{||\mathbf{s-t}||\leq\Delta_{n}
\\||\mathbf{s}||,||\mathbf{t}||\leq K}}|\frac{1}{n}\sum_{i=1}^{n}
\{d_{n}(\mathbf{z}_{i};\mathbf{s},\mathbf{t})-\mathbb{E[}d_{n}(\mathbf{z}
;\mathbf{s},\mathbf{t})]\}|\right] .
\end{align*}
For large $n,$ the first term on the majorant side can be made arbitrarily
small by making $C$ large. Also, for any fixed $C,$ the second term tends to
zero because $\Delta_{n}\rightarrow0.$ Finally, \citet*[Theorem 4.2]
{Pollard_1989_SS} can be used to show that for fixed $C$ and for large $n,$
the last term is bounded by a constant multiple of
\[
r_{n}\sqrt{\frac{\mathbb{E[}\bar{d}_{n}^{Kr_{n}^{-1}}(\mathbf{z})^{2}]}{n}
}=\frac{\sqrt{K}}{r_{n}}\sqrt{q_{n}\mathbb{E[}\bar{d}_{n}^{Kr_{n}^{-1}
}(\mathbf{z})^{2}/(Kr_{n}^{-1})]}=O\left( \frac{1}{r_{n}}\right)
=o(1).\quad\blacksquare
\]
\bigskip
The analysis of $\mathbf{\tilde{\theta}}_{n}^{\ast}$ also relies on five
lemmas, each of which is a natural bootstrap analog of a lemma used to analyze
$\mathbf{\hat{\theta}}_{n}.$ The following lemma can be used to show that
$\mathbf{\tilde{\theta}}_{n}^{\ast}$ is consistent in the sense that
$\mathbf{\tilde{\theta}}_{n}^{\ast}-\mathbf{\hat{\theta}}_{n}=o_{\mathbb{P}
}(1).$
\begin{lemma}
\label{[Lemma] Consistency (bootstrap)}Suppose Condition CRA(i) holds and
suppose $\mathbf{\tilde{H}}_{n}\rightarrow_{\mathbb{P}}\mathbf{H},$ where
$\mathbf{H}$ is symmetric and positive definite. Then $\mathbf{\tilde{\theta}
}_{n}^{\ast}-\mathbf{\hat{\theta}}_{n}=o_{\mathbb{P}}(1)$ if
\[
\tilde{M}_{n}^{\ast}(\mathbf{\tilde{\theta}}_{n}^{\ast})\geq\sup
_{\mathbf{\theta}\in\mathbf{\Theta}}\tilde{M}_{n}^{\ast}(\mathbf{\theta
})-o_{\mathbb{P}}(1).
\]
\end{lemma}
\noindent\textbf{Proof of Lemma }\ref{[Lemma] Consistency (bootstrap)}. It
suffices to show that every $\delta>0$ admits a constant $c_{\delta}^{\ast}>0$
such that
\begin{equation}
\mathbb{P}\left[ \tilde{M}_{n}^{\ast}(\mathbf{\hat{\theta}}_{n}
)-\sup_{\mathbf{\theta}\in\mathbf{\Theta\setminus\hat{\Theta}}_{n}^{\delta}
}\tilde{M}_{n}^{\ast}(\mathbf{\theta})>c_{\delta}^{\ast}\right]
\rightarrow1,\label{Consistency: Sufficient condition (bootstrap)}
\end{equation}
where $\mathbf{\hat{\Theta}}_{n}^{\delta}=\{\mathbf{\theta}\in\mathbf{\Theta
}:||\mathbf{\theta-\hat{\theta}}_{n}||\leq\delta\}.$ The process $\tilde
{M}_{n}^{\ast}$ satisfies
\[
\tilde{M}_{n}^{\ast}(\mathbf{\theta})=\hat{M}_{n}^{\ast}(\mathbf{\theta}
)-\hat{M}_{n}(\mathbf{\theta})-\frac{1}{2}(\mathbf{\theta}-\mathbf{\hat
{\theta}}_{n})^{\prime}\mathbf{\tilde{H}}_{n}(\mathbf{\theta}-\mathbf{\hat
{\theta}}_{n}),\qquad\hat{M}_{n}^{\ast}(\mathbf{\theta})=\frac{1}{n}\sum
_{i=1}^{n}m_{n}(\mathbf{z}_{i,n}^{\ast},\mathbf{\theta}),
\]
where it follows from \citet*[Theorem 4.2]{Pollard_1989_SS} that
\[
\sup_{\mathbf{\theta}\in\mathbf{\Theta}}|\hat{M}_{n}^{\ast}(\mathbf{\theta
})-\hat{M}_{n}(\mathbf{\theta})|=O_{\mathbb{P}}\left( \sqrt{\frac
{\mathbb{E}[\bar{m}_{n}(\mathbf{z})^{2}]}{n}}\right) =O_{\mathbb{P}}\left(
\frac{1}{\sqrt{nq_{n}}}\right) =o_{\mathbb{P}}(1).
\]
As a consequence, for any $\delta>0,$
\[
\tilde{M}_{n}^{\ast}(\mathbf{\hat{\theta}}_{n})-\sup_{\mathbf{\theta}
\in\mathbf{\Theta\setminus\hat{\Theta}}_{n}^{\delta}}\tilde{M}_{n}^{\ast
}(\mathbf{\theta})=\frac{1}{2}\inf_{\mathbf{\theta}\in\mathbf{\Theta
\setminus\hat{\Theta}}_{n}^{\delta}}(\mathbf{\theta}-\mathbf{\hat{\theta}}
_{n})^{\prime}\mathbf{\tilde{H}}_{n}(\mathbf{\theta}-\mathbf{\hat{\theta}}
_{n})+o_{\mathbb{P}}(1),
\]
so (\ref{Consistency: Sufficient condition (bootstrap)}) is satisfied with
$c_{\delta}^{\ast}=\delta^{2}\lambda_{\min}(\mathbf{H})/4>0.\quad\blacksquare$\bigskip
Next, because
\[
\tilde{M}_{n}(\mathbf{\theta})=\mathbb{E}_{n}^{\ast}[\tilde{M}_{n}^{\ast
}(\mathbf{\theta})]=\frac{1}{n}\sum_{i=1}^{n}\tilde{m}_{n}(\mathbf{z}
_{i},\mathbf{\theta})=-\frac{1}{2}(\mathbf{\theta}-\mathbf{\hat{\theta}}
_{n})^{\prime}\mathbf{\tilde{H}}_{n}(\mathbf{\theta}-\mathbf{\hat{\theta}}
_{n}),
\]
we have the following convergence result about $\tilde{Q}_{n}.$
\begin{lemma}
\label{[Lemma] Convergence of Qtilde_n}Suppose $r_{n}\rightarrow\infty,$
$\mathbf{\tilde{H}}_{n}\rightarrow_{\mathbb{P}}\mathbf{H},$ and suppose
$\mathbf{\hat{\theta}}_{n}\rightarrow_{\mathbb{P}}\mathbf{\theta}_{0},$ where
$\mathbf{\theta}_{0}$ is an interior point of $\mathbf{\Theta.}$ Then
$\tilde{Q}_{n}\rightarrow_{\mathbb{P}}\mathcal{Q}$ in the topology of uniform
convergence on compacta, where $\mathcal{Q}(\mathbf{s})=-\mathbf{s}^{\prime
}\mathbf{Hs}/2;$ that is
\[
\sup_{||\mathbf{s}||\leq K}\left\vert \tilde{Q}_{n}(\mathbf{s})-(-\frac{1}
{2}\mathbf{s}^{\prime}\mathbf{Hs})\right\vert \rightarrow_{\mathbb{P}}0
\]
for any $K>0.$
\end{lemma}
\noindent\textbf{Proof of Lemma }\ref{[Lemma] Convergence of Qtilde_n}.
Uniformly in $\mathbf{s}$ with $||\mathbf{s}||\leq K,$ we have
\[
\left\vert \tilde{Q}_{n}(\mathbf{s})-(-\frac{1}{2}\mathbf{s}^{\prime
}\mathbf{Hs})\right\vert \leq\frac{1}{2}\left\vert \mathbf{s}^{\prime
}(\mathbf{\tilde{H}}_{n}-\mathbf{H})\mathbf{s}\right\vert +\frac{1}
{2}\left\vert \mathbf{s}^{\prime}\mathbf{Hs}\right\vert
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(\mathbf{\hat{\theta}}_{n}+\mathbf{s}r_{n}^{-1}\notin\mathbf{\Theta})\leq
K^{2}o_{\mathbb{P}}(1),
\]
where the last inequality uses $\mathbf{\tilde{H}}_{n}\rightarrow_{\mathbb{P}
}\mathbf{H}$ and $\mathbb{P}(\mathbf{\hat{\theta}}_{n}+\mathbf{s}r_{n}
^{-1}\notin\mathbf{\Theta})\rightarrow0.\quad\blacksquare$\bigskip
The next lemma can be used to obtain the rate of convergence of
$\mathbf{\tilde{\theta}}_{n}^{\ast}.$
\begin{lemma}
\label{[Lemma] Rate of convergence (bootstrap)}Suppose Condition CRA(iii)
holds and suppose $\mathbf{\tilde{H}}_{n}\rightarrow_{\mathbb{P}}\mathbf{H}, $
where $\mathbf{H}$ is symmetric and positive definite. Then $r_{n}
(\mathbf{\tilde{\theta}}_{n}^{\ast}-\mathbf{\hat{\theta}}_{n})=O_{\mathbb{P}
}(1)$ if $r_{n}(\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0})=O_{\mathbb{P}
}(1),$ $\mathbf{\tilde{\theta}}_{n}^{\ast}-\mathbf{\hat{\theta}}
_{n}=o_{\mathbb{P}}(1),$ and if
\[
\tilde{M}_{n}^{\ast}(\mathbf{\tilde{\theta}}_{n}^{\ast})\geq\sup
_{\mathbf{\theta}\in\mathbf{\Theta}}\tilde{M}_{n}^{\ast}(\mathbf{\theta
})-o_{\mathbb{P}}(r_{n}^{-2}).
\]
\end{lemma}
\noindent\textbf{Proof of Lemma }\ref{[Lemma] Rate of convergence (bootstrap)}
. For any $\delta>0$ and any $K\in\mathbb{N},$ $\mathbb{P}[r_{n}
||\mathbf{\tilde{\theta}}_{n}^{\ast}-\mathbf{\hat{\theta}}_{n}||>2^{K+1}]$ is
no greater than
\begin{gather*}
\mathbb{P}[\sup_{\mathbf{\theta}\in\mathbf{\Theta}}\tilde{M}_{n}^{\ast
}(\mathbf{\theta})-\tilde{M}_{n}^{\ast}(\mathbf{\tilde{\theta}}_{n}^{\ast
})\geq\delta r_{n}^{-2}]+\mathbb{P}[||\mathbf{\tilde{H}}_{n}-\mathbf{H}
||>\delta]+\mathbb{P}[||\mathbf{\tilde{\theta}}_{n}^{\ast}-\mathbf{\hat
{\theta}}_{n}||>\delta/4]\\
+\mathbb{P}[r_{n}||\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0}||>2^{K}]\\
+\sum_{j\geq K,2^{j+1}\leq\delta r_{n}}\mathbb{P}\left[ \sup_{2^{j-1}
<r_{n}||\mathbf{\theta}-\mathbf{\hat{\theta}}_{n}||\leq2^{j},r_{n}
||\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0}||\leq2^{K},||\mathbf{\tilde
{H}}_{n}-\mathbf{H}||\leq\delta}\tilde{M}_{n}^{\ast}(\mathbf{\theta}
)-\tilde{M}_{n}^{\ast}(\mathbf{\hat{\theta}}_{n})\geq-\delta r_{n}
^{-2}\right] .
\end{gather*}
By assumption, the probabilities on the first line go to zero for any
$\delta>0$ and the probability on the second line can be made arbitrarily
small by making $K$ large. As a consequence, it suffices to show that the sum
on the last line can be made arbitrarily small (for large $n$) by making
$\delta>0$ small and $K$ large.
To do so, let $\delta>0$ be small enough so that Condition CRA(iii) holds and
\[
\frac{1}{2}\inf_{||\mathbf{\bar{H}}-\mathbf{H}||\leq\delta}\lambda_{\min
}(\mathbf{\bar{H}}+\mathbf{\bar{H}}^{\prime})>\lambda_{\min}(\mathbf{H}).
\]
Then, if $||\mathbf{\tilde{H}}_{n}-\mathbf{H}||\leq\delta,$ we have
\[
\tilde{M}_{n}(\mathbf{\hat{\theta}}_{n})-\sup_{2^{j-1}<r_{n}||\mathbf{\theta
}-\mathbf{\hat{\theta}}_{n}||\leq2^{j}}\tilde{M}_{n}(\mathbf{\theta})-\delta
r_{n}^{-2}\geq2^{2j}c_{K}^{\ast}(\delta)r_{n}^{-2}
\]
for any pair $(j,K)^{\prime}\in\mathbb{N}^{2}$ with $j\geq K,$ where
$c_{K}^{\ast}(\delta)=\lambda_{\min}(\mathbf{H})/16-2^{-2K}\delta.$
Choosing $K$ large enough that $c_{K}^{\ast}(\delta)\geq c^{\ast}
=\lambda_{\min}(\mathbf{H})/32$ and using the fact that
\[
\tilde{M}_{n}^{\ast}(\mathbf{\theta})-\tilde{M}_{n}^{\ast}(\mathbf{\hat
{\theta}}_{n})-\tilde{M}_{n}(\mathbf{\theta})+\tilde{M}_{n}(\mathbf{\hat
{\theta}}_{n})=\hat{M}_{n}^{\ast}(\mathbf{\theta})-\hat{M}_{n}^{\ast
}(\mathbf{\hat{\theta}}_{n})-\hat{M}_{n}(\mathbf{\theta})+\hat{M}
_{n}(\mathbf{\hat{\theta}}_{n}),
\]
we therefore have
\begin{align*}
& \sum_{j\geq K,2^{j+1}\leq\delta r_{n}}\mathbb{P}\left[ \sup_{2^{j-1}
<r_{n}||\mathbf{\theta}-\mathbf{\hat{\theta}}_{n}||\leq2^{j},r_{n}
||\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0}||\leq2^{K},||\mathbf{\tilde
{H}}_{n}-\mathbf{H}||\leq\delta}\tilde{M}_{n}^{\ast}(\mathbf{\theta}
)-\tilde{M}_{n}^{\ast}(\mathbf{\hat{\theta}}_{n})\geq-\delta r_{n}^{-2}\right]
\\
& \leq\sum_{j\geq K,2^{j+1}\leq\delta r_{n}}\mathbb{P}\left[ \sup
_{2^{j-1}<r_{n}||\mathbf{\theta}-\mathbf{\hat{\theta}}_{n}||\leq2^{j}
,r_{n}||\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0}||\leq2^{K}}\{\hat{M}
_{n}^{\ast}(\mathbf{\theta})-\hat{M}_{n}^{\ast}(\mathbf{\hat{\theta}}
_{n})-\hat{M}_{n}(\mathbf{\theta})+\hat{M}_{n}(\mathbf{\hat{\theta}}
_{n})\}\geq2^{2j}c^{\ast}r_{n}^{-2}\right] \\
& \leq\sum_{j\geq K,2^{j+1}\leq\delta r_{n}}\mathbb{P}\left[ \sup
_{r_{n}||\mathbf{\theta}-\mathbf{\hat{\theta}}_{n}||\leq2^{j},r_{n}
||\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0}||\leq2^{K}}||\hat{M}_{n}
^{\ast}(\mathbf{\theta})-\hat{M}_{n}^{\ast}(\mathbf{\hat{\theta}}_{n})-\hat
{M}_{n}(\mathbf{\theta})+\hat{M}_{n}(\mathbf{\hat{\theta}}_{n})||\geq
2^{2j}c^{\ast}r_{n}^{-2}\right] \\
& \leq\frac{r_{n}^{2}}{c^{\ast}}\sum_{j\geq K,2^{j+1}\leq\delta r_{n}}
2^{-2j}\mathbb{E}\left[ \sup_{r_{n}||\mathbf{\theta}-\mathbf{\theta}
_{0}||\leq2^{j+1},r_{n}||\mathbf{\theta}^{\prime}-\mathbf{\theta}_{0}
||\leq2^{K}}||\hat{M}_{n}^{\ast}(\mathbf{\theta})-\hat{M}_{n}^{\ast
}(\mathbf{\theta}^{\prime})-\hat{M}_{n}(\mathbf{\theta})+\hat{M}
_{n}(\mathbf{\theta}^{\prime})||\right] ,
\end{align*}
where the last inequality uses the Markov inequality.
Under Condition CRA(iii), $q_{n}\sup_{0\leq\delta^{\prime}\leq\delta
}\mathbb{E}[\bar{d}_{n}^{\delta^{\prime}}(\mathbf{z})^{2}/\delta^{\prime
}]=O(1)$ and \citet*[Theorem 4.2]{Pollard_1989_SS} can be used to show that the
sum on the last line is bounded by a constant multiple of
\[
r_{n}^{2}\sum_{j\geq K,2^{j+1}\leq\delta r_{n}}2^{-2j}\sqrt{\frac
{\mathbb{E}[\bar{d}_{n}^{2^{j+1}/r_{n}}(\mathbf{z})^{2}]}{n}}\leq\sqrt
{2q_{n}\sup_{0\leq\delta^{\prime}\leq\delta}\mathbb{E}[\bar{d}_{n}
^{\delta^{\prime}}(\mathbf{z})^{2}/\delta^{\prime}]}\sum_{j\geq K}2^{-3j/2},
\]
which can be made arbitrarily small by making $K$ large.$\quad\blacksquare$\bigskip
Finally, the next two lemmas can be combined to show that $\tilde{G}_{n}
^{\ast}\rightsquigarrow_{\mathbb{P}}\mathcal{G}_{0}$ in the topology of
uniform convergence on compacta.
\begin{lemma}
\label{[Lemma] FIDI convergence (bootstrap)}Suppose Conditions CRA(iii)-(iv)
hold, $r_{n}(\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0})=O_{\mathbb{P}
}(1),$ and that, for every $K>0,$ $\sup_{||\mathbf{s}||\leq K}|\hat{G}
_{n}(\mathbf{s})+Q_{n}(\mathbf{s})|=o_{\mathbb{P}}(\sqrt{n}).$ Then $\tilde
{G}_{n}^{\ast}$ converges to $\mathcal{G}_{0}$ in the sense of conditional
weak convergence in probability of finite-dimensional projections.
\end{lemma}
\noindent\textbf{Proof of Lemma }\ref{[Lemma] FIDI convergence (bootstrap)}.
Because $\tilde{G}_{n}^{\ast}(\mathbf{s})=n^{-1/2}\sum_{i=1}^{n}\hat{\psi}
_{n}(\mathbf{z}_{i,n}^{\ast};\mathbf{s}),$ where
\[
\hat{\psi}_{n}(\mathbf{z;s})=\sqrt{r_{n}q_{n}}[m_{n}(\mathbf{z},\mathbf{\hat
{\theta}}_{n}+\mathbf{s}r_{n}^{-1})-m_{n}(\mathbf{z},\mathbf{\hat{\theta}}
_{n})-\hat{M}_{n}(\mathbf{\hat{\theta}}_{n}+\mathbf{s}r_{n}^{-1})+\hat{M}
_{n}(\mathbf{\hat{\theta}}_{n})]
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(\mathbf{\hat{\theta}}_{n}+\mathbf{s}r_{n}^{-1}\in\mathbf{\Theta}),
\]
the result follows from the Cram\'{e}r-Wold device if
\[
\mathbb{E}_{n}^{\ast}[\hat{\psi}_{n}(\mathbf{z}^{\ast};\mathbf{s})\hat{\psi
}_{n}(\mathbf{z}^{\ast};\mathbf{t})]=\frac{1}{n}\sum_{i=1}^{n}\hat{\psi}
_{n}(\mathbf{z}_{i};\mathbf{s})\hat{\psi}_{n}(\mathbf{z}_{i};\mathbf{t}
)\rightarrow_{\mathbb{P}}\mathcal{C}_{0}(\mathbf{s},\mathbf{t})\qquad
\forall\mathbf{s},\mathbf{t}\in\mathbb{R}^{d},
\]
and if the following Lyapunov condition is satisfied:
\[
\frac{1}{n}\mathbb{E}_{n}^{\ast}[\hat{\psi}_{n}(\mathbf{z}^{\ast}
;\mathbf{s})^{4}]=\frac{1}{n^{2}}\sum_{i=1}^{n}\hat{\psi}_{n}(\mathbf{z}
_{i};\mathbf{s})^{4}\rightarrow_{\mathbb{P}}0\qquad\forall\mathbf{s}
\in\mathbb{R}^{d}.
\]
Let $\mathbf{s},\mathbf{t}\in\mathbb{R}^{d}$ be given and suppose without loss
of generality that $\mathbf{\hat{\theta}}_{n}+\mathbf{s}r_{n}^{-1}
,\mathbf{\hat{\theta}}_{n}+\mathbf{t}r_{n}^{-1}\in\mathbf{\Theta.}$ Because
$r_{n}(\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0})=O_{\mathbb{P}}(1),$ we
have
\begin{align*}
\hat{Q}_{n}(\mathbf{s}) & =r_{n}^{2}[\hat{M}_{n}(\mathbf{\hat{\theta}}
_{n}+\mathbf{s}r_{n}^{-1})-\hat{M}_{n}(\mathbf{\hat{\theta}}_{n})]
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(\mathbf{\hat{\theta}}_{n}+\mathbf{s}r_{n}^{-1}\in\mathbf{\Theta})\\
& =\{\hat{G}_{n}[r_{n}(\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}
_{0})+\mathbf{s}]+Q_{n}[r_{n}(\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}
_{0})+\mathbf{s}]\}-\{\hat{G}_{n}[r_{n}(\mathbf{\hat{\theta}}_{n}
-\mathbf{\theta}_{0})]+Q_{n}[r_{n}(\mathbf{\hat{\theta}}_{n}-\mathbf{\theta
}_{0})]\}\\
& =o_{\mathbb{P}}(\sqrt{n})
\end{align*}
and, using $\mathbb{E}[\bar{d}_{n}^{\delta_{n}}(\mathbf{z})^{4}]=o(q_{n}
^{-3}r_{n})$ (for $\delta_{n}=O(r_{n}^{-1})$) and \citet*[Theorem 4.2]
{Pollard_1989_SS},
\begin{align*}
& r_{n}q_{n}\mathbb{E}_{n}^{\ast}[\{m_{n}(\mathbf{z}^{\ast},\mathbf{\hat
{\theta}}_{n}+\mathbf{s}r_{n}^{-1})-m_{n}(\mathbf{z}^{\ast},\mathbf{\hat
{\theta}}_{n})\}\{m_{n}(\mathbf{z}^{\ast},\mathbf{\hat{\theta}}_{n}
+\mathbf{t}r_{n}^{-1})-m_{n}(\mathbf{z}^{\ast},\mathbf{\hat{\theta}}
_{n})\}]-\mathcal{\hat{C}}_{n}(\mathbf{s},\mathbf{t})\\
& =\frac{r_{n}q_{n}}{n}\sum_{i=1}^{n}\{m_{n}(\mathbf{z}_{i},\mathbf{\hat
{\theta}}_{n}+\mathbf{s}r_{n}^{-1})-m_{n}(\mathbf{z}_{i},\mathbf{\hat{\theta}
}_{n})\}\{m_{n}(\mathbf{z}_{i},\mathbf{\hat{\theta}}_{n}+\mathbf{t}r_{n}
^{-1})-m_{n}(\mathbf{z}_{i},\mathbf{\hat{\theta}}_{n})\}-\mathcal{\hat{C}}
_{n}(\mathbf{s},\mathbf{t})\\
& =o_{\mathbb{P}}\left( r_{n}q_{n}\sqrt{\frac{r_{n}}{nq_{n}^{3}}}\right)
=o_{\mathbb{P}}(1),
\end{align*}
where
\begin{align*}
\mathcal{\hat{C}}_{n}(\mathbf{s},\mathbf{t}) & =r_{n}q_{n}\left.
\mathbb{E}[\{m_{n}(\mathbf{z},\mathbf{\theta}+\mathbf{s}r_{n}^{-1}
)-m_{n}(\mathbf{z},\mathbf{\theta})\}\{m_{n}(\mathbf{z},\mathbf{\theta
}+\mathbf{t}r_{n}^{-1})-m_{n}(\mathbf{z},\mathbf{\theta})\}]\right\vert
_{\mathbf{\theta}=\mathbf{\hat{\theta}}_{n}}\\
& =\mathcal{C}_{0}(\mathbf{s},\mathbf{t})+o_{\mathbb{P}}(1).
\end{align*}
Using these facts and the representation
\[
\hat{\psi}_{n}(\mathbf{z};\mathbf{s})=\sqrt{r_{n}q_{n}}[m_{n}(\mathbf{z}
,\mathbf{\hat{\theta}}_{n}+\mathbf{s}r_{n}^{-1})-m_{n}(\mathbf{z}
,\mathbf{\hat{\theta}}_{n})]-\frac{1}{\sqrt{n}}\hat{Q}_{n}(\mathbf{s}),
\]
we have
\begin{align*}
& \mathbb{E}_{n}^{\ast}[\hat{\psi}_{n}(\mathbf{z}^{\ast};\mathbf{s})\hat{\psi
}_{n}(\mathbf{z}^{\ast};\mathbf{t})]\\
& =r_{n}q_{n}\mathbb{E}_{n}^{\ast}[\{m_{n}(\mathbf{z}^{\ast},\mathbf{\hat
{\theta}}_{n}+\mathbf{s}r_{n}^{-1})-m_{n}(\mathbf{z}^{\ast},\mathbf{\hat
{\theta}}_{n})\}\{m_{n}(\mathbf{z}^{\ast},\mathbf{\hat{\theta}}_{n}
+\mathbf{t}r_{n}^{-1})-m_{n}(\mathbf{z}^{\ast},\mathbf{\hat{\theta}}
_{n})\}]-\frac{1}{n}\hat{Q}_{n}(\mathbf{s})\hat{Q}_{n}(\mathbf{t})\\
& =\mathcal{C}_{0}(\mathbf{s},\mathbf{t})+o_{\mathbb{P}}(1)
\end{align*}
and, using $\mathbb{E}[\bar{d}_{n}^{\delta_{n}}(\mathbf{z})^{4}]=o(q_{n}
^{-3}r_{n})$ (for $\delta_{n}=O(r_{n}^{-1})$),
\begin{align*}
\frac{1}{16n}\mathbb{E}_{n}^{\ast}[\hat{\psi}_{n}(\mathbf{z}^{\ast}
;\mathbf{s})^{4}] & =\frac{1}{16n^{2}}\sum_{i=1}^{n}\hat{\psi}_{n}
(\mathbf{z}_{i};\mathbf{s})^{4}\leq\frac{r_{n}^{2}q_{n}^{2}}{n^{2}}\sum
_{i=1}^{n}|m_{n}(\mathbf{z}_{i},\mathbf{\hat{\theta}}_{n}+\mathbf{s}r_{n}
^{-1})-m_{n}(\mathbf{z}_{i},\mathbf{\hat{\theta}}_{n})|^{4}+\frac{1}{n^{3}
}\hat{Q}_{n}(\mathbf{s})^{4}\\
& =o_{\mathbb{P}}\left( \frac{r_{n}^{3}}{nq_{n}}+\frac{1}{n}\right)
=o_{\mathbb{P}}(1).\quad\blacksquare
\end{align*}
\begin{lemma}
\label{[Lemma] Stochastic equicontinuity (bootstrap)}Suppose Conditions
CRA(iii) and CRA(v) hold and suppose $r_{n}(\mathbf{\hat{\theta}}
_{n}-\mathbf{\theta}_{0})=O_{\mathbb{P}}(1).$ Then $\{\tilde{G}_{n}^{\ast
}(\mathbf{s}):||\mathbf{s}||\leq K\}$ is stochastically equicontinuous for
every $K>0;$ that is,
\[
\sup_{\substack{||\mathbf{s-t}||\leq\Delta_{n} \\||\mathbf{s}||,||\mathbf{t}
||\leq K}}|\tilde{G}_{n}^{\ast}(\mathbf{s})-\tilde{G}_{n}^{\ast}
(\mathbf{t})|\rightarrow_{\mathbb{P}}0
\]
for any $K>0$ and for any $\Delta_{n}>0$ with $\Delta_{n}=o(1).$
\end{lemma}
\noindent\textbf{Proof of Lemma }
\ref{[Lemma] Stochastic equicontinuity (bootstrap)}. Let $K>0$ be given.
Proceeding as in the proof of \citet*[Lemma 4.6]{Kim-Pollard_1990_AoS} and using
$q_{n}\delta_{n}^{-1}\mathbb{E}[\bar{d}_{n}^{\delta_{n}}(\mathbf{z}
)^{2}]=O(1)$ (for $\delta_{n}=O(r_{n}^{-1})$) along with the fact that
$r_{n}(\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0})=O_{\mathbb{P}}(1),$ it
suffices to show that, for every finite $k>0,$
\begin{align*}
& r_{n}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
\left( r_{n}||\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0}||\leq k\right)
\sup_{\substack{||\mathbf{s-t}||\leq\Delta_{n} \\||\mathbf{s}||,||\mathbf{t}
||\leq K}}\frac{q_{n}}{n}\sum_{i=1}^{n}\hat{d}_{n}(\mathbf{z}_{i,n}^{\ast
};\mathbf{s},\mathbf{t})^{2}\\
& \leq r_{n}\sup_{\substack{||\mathbf{s-t}||\leq\Delta_{n} \\||\mathbf{s}
||,||\mathbf{t}||\leq K+k}}\frac{q_{n}}{n}\sum_{i=1}^{n}d_{n}(\mathbf{z}
_{i,n}^{\ast};\mathbf{s},\mathbf{t})^{2}\rightarrow_{\mathbb{P}}0,
\end{align*}
where
\[
\hat{d}_{n}(\mathbf{z};\mathbf{s},\mathbf{t})=\frac{1}{2}|m_{n}(\mathbf{z}
,\mathbf{\hat{\theta}}_{n}+\mathbf{s}r_{n}^{-1})-m_{n}(\mathbf{z}
,\mathbf{\hat{\theta}}_{n}+\mathbf{t}r_{n}^{-1})|=d_{n}(\mathbf{z}
;r_{n}(\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0})+\mathbf{s}
,r_{n}(\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0})+\mathbf{t}).
\]
Let $k>0$ be given. For any $C>0$ and any $\mathbf{s},\mathbf{t}\in
\mathbb{R}^{d}$ with $||\mathbf{s}||,||\mathbf{t}||\leq K+k,$
\begin{align*}
\frac{q_{n}}{n}\sum_{i=1}^{n}d_{n}(\mathbf{z}_{i,n}^{\ast};\mathbf{s}
,\mathbf{t})^{2} & \leq\frac{q_{n}}{n}\sum_{i=1}^{n}\bar{d}_{n}
^{(K+k)r_{n}^{-1}}(\mathbf{z}_{i,n}^{\ast})^{2}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(q_{n}\bar{d}_{n}^{(K+k)r_{n}^{-1}}(\mathbf{z}_{i,n}^{\ast})>C)\\
& +C\mathbb{E[}d_{n}(\mathbf{z};\mathbf{s},\mathbf{t})]\\
& +C\frac{1}{n}\sum_{i=1}^{n}\{d_{n}(\mathbf{z}_{i,n};\mathbf{s}
,\mathbf{t})-\mathbb{E[}d_{n}(\mathbf{z};\mathbf{s},\mathbf{t})]\}\\
& +C\frac{1}{n}\sum_{i=1}^{n}\{d_{n}(\mathbf{z}_{i,n}^{\ast};\mathbf{s}
,\mathbf{t})-\mathbb{E}_{n}^{\ast}\mathbb{[}d_{n}(\mathbf{z}^{\ast}
;\mathbf{s},\mathbf{t})]\},
\end{align*}
and therefore
\begin{align*}
& r_{n}\mathbb{E}\left[ \sup_{\substack{||\mathbf{s-t}||\leq\Delta_{n}
\\||\mathbf{s}||,||\mathbf{t}||\leq K+k}}\frac{q_{n}}{n}\sum_{i=1}^{n}
d_{n}(\mathbf{z}_{i,n}^{\ast};\mathbf{s},\mathbf{t})^{2}\right] \\
& \leq r_{n}\mathbb{E}\left[ \frac{q_{n}}{n}\sum_{i=1}^{n}\bar{d}
_{n}^{(K+k)r_{n}^{-1}}(\mathbf{z}_{i,n}^{\ast})^{2}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(q_{n}\bar{d}_{n}^{(K+k)r_{n}^{-1}}(\mathbf{z}_{i,n}^{\ast})>C)\right] \\
& +Cr_{n}\sup_{\substack{||\mathbf{s-t}||\leq\Delta_{n} \\||\mathbf{s}
||,||\mathbf{t}||\leq K+k}}\mathbb{E[}d_{n}(\mathbf{z};\mathbf{s}
,\mathbf{t})]\\
& +Cr_{n}\mathbb{E}\left[ \sup_{\substack{||\mathbf{s-t}||\leq\Delta_{n}
\\||\mathbf{s}||,||\mathbf{t}||\leq K+k}}\left\vert \frac{1}{n}\sum_{i=1}
^{n}\{d_{n}(\mathbf{z}_{i,n};\mathbf{s},\mathbf{t})-\mathbb{E[}d_{n}
(\mathbf{z};\mathbf{s},\mathbf{t})]\}\right\vert \right] \\
& +Cr_{n}\mathbb{E}\left[ \sup_{\substack{||\mathbf{s-t}||\leq\Delta_{n}
\\||\mathbf{s}||,||\mathbf{t}||\leq K+k}}\left\vert \frac{1}{n}\sum_{i=1}
^{n}\{d_{n}(\mathbf{z}_{i,n}^{\ast};\mathbf{s},\mathbf{t})-\mathbb{E}^{\ast
}\mathbb{[}d_{n}(\mathbf{z}^{\ast};\mathbf{s},\mathbf{t})]\}\right\vert
\right] .
\end{align*}
For large $n,$ the first term on the majorant side can be made arbitrarily
small by making $C$ large. Also, for any fixed $C,$ the second term tends to
zero because $\Delta_{n}\rightarrow0.$ Finally, \citet*[Theorem 4.2]
{Pollard_1989_SS} can be used to show that for fixed $C$ and for large $n,$
each of the last two terms is bounded by a constant multiple of
\[
r_{n}\sqrt{\frac{\mathbb{E[}\bar{d}_{n}^{(K+k)r_{n}^{-1}}(\mathbf{z})^{2}]}
{n}}=\frac{\sqrt{K+k}}{r_{n}}\sqrt{q_{n}\mathbb{E[}\bar{d}_{n}^{(K+k)r_{n}
^{-1}}(\mathbf{z})^{2}/\{(K+k)r_{n}^{-1}\}]}=O\left( \frac{1}{r_{n}}\right)
=o(1).\quad\blacksquare
\]
\subsection{Proof of Lemma 1}
Without loss of generality, suppose $r_{n}||\mathbf{\hat{\theta}
}-\mathbf{\theta}_{0}||\leq K\ $for some fixed constant $K.$ Defining
\[
\check{H}_{n,kl}^{\mathtt{ND}}=-\frac{1}{4\epsilon_{n}^{2}}[\hat{M}
_{n}(\mathbf{\theta}_{0}+\epsilon_{n}\mathbf{e}_{k}+\epsilon_{n}\mathbf{e}
_{l})-\hat{M}_{n}(\mathbf{\theta}_{0}-\epsilon_{n}\mathbf{e}_{k}+\epsilon
_{n}\mathbf{e}_{l})-\hat{M}_{n}(\mathbf{\theta}_{0}+\epsilon_{n}\mathbf{e}
_{k}-\epsilon_{n}\mathbf{e}_{l})+\hat{M}_{n}(\mathbf{\theta}_{0}-\epsilon
_{n}\mathbf{e}_{k}-\epsilon_{n}\mathbf{e}_{l})]
\]
and
\[
\bar{H}_{n,kl}^{\mathtt{ND}}(\mathbf{\theta})=-\frac{1}{4\epsilon_{n}^{2}
}[M_{n}(\mathbf{\theta}+\epsilon_{n}\mathbf{e}_{k}+\epsilon_{n}\mathbf{e}
_{l})-M_{n}(\mathbf{\theta}-\epsilon_{n}\mathbf{e}_{k}+\epsilon_{n}
\mathbf{e}_{l})-M_{n}(\mathbf{\theta}+\epsilon_{n}\mathbf{e}_{k}-\epsilon
_{n}\mathbf{e}_{l})+M_{n}(\mathbf{\theta}-\epsilon_{n}\mathbf{e}_{k}
-\epsilon_{n}\mathbf{e}_{l})],
\]
we obtain the decomposition
\[
\tilde{H}_{n,kl}^{\mathtt{ND}}=\check{H}_{n,kl}^{\mathtt{ND}}+R_{n,kl}
^{\mathtt{ND}}+S_{n,kl}^{\mathtt{ND}},
\]
where
\[
R_{n,kl}^{\mathtt{ND}}=\tilde{H}_{n,kl}^{\mathtt{ND}}-\check{H}_{n,kl}
^{\mathtt{ND}}-\bar{H}_{n,kl}^{\mathtt{ND}}(\mathbf{\hat{\theta}}_{n})+\bar
{H}_{n,kl}^{\mathtt{ND}}(\mathbf{\theta}_{0}),\qquad S_{n,kl}^{\mathtt{ND}
}=\bar{H}_{n,kl}^{\mathtt{ND}}(\mathbf{\hat{\theta}}_{n})-\bar{H}
_{n,kl}^{\mathtt{ND}}(\mathbf{\theta}_{0}).
\]
\newline The proof will be completed by showing that $\check{H}_{n,kl}
^{\mathtt{ND}}\rightarrow_{\mathbb{P}}H_{0,kl},$ $R_{n,kl}^{\mathtt{ND}
}=o_{\mathbb{P}}(1),$ and $S_{n,kl}^{\mathtt{ND}}=o_{\mathbb{P}}(1).$
First, using (\ref{Quadratic approximation: M_n}) and the fact that $\dot
{C}_{n}=o(r_{n}^{-1})$ and $\ddot{C}_{n}=o(1)$ under Condition CRA(ii), we
have
\[
M_{n}(\mathbf{\theta}_{0}+\epsilon_{n}\mathbf{e}_{k}+\epsilon_{n}
\mathbf{e}_{l})-M_{n}(\mathbf{\theta}_{0})=-\epsilon_{n}^{2}\frac{1}
{2}(\mathbf{e}_{k}+\mathbf{e}_{l})^{\prime}\mathbf{H}_{n}(\mathbf{e}
_{k}+\mathbf{e}_{l})+o\left( \frac{\epsilon_{n}}{r_{n}}+\epsilon_{n}
^{2}\right) ,
\]
implying in particular that
\[
\bar{H}_{n,kl}^{\mathtt{ND}}(\mathbf{\theta}_{0})=H_{n,kl}+o\left( \frac
{1}{r_{n}\epsilon_{n}}+1\right) ,
\]
where, using $\mathbf{H}_{n}\rightarrow\mathbf{H}_{0},$
\[
H_{n,kl}=\mathbf{e}_{k}^{\prime}\mathbf{H}_{n}\mathbf{e}_{l}\rightarrow
\mathbf{e}_{k}^{\prime}\mathbf{H}_{0}\mathbf{e}_{l}=H_{0,kl}.
\]
Moreover, $\check{H}_{n,kl}^{\mathtt{ND}}-\bar{H}_{n,kl}^{\mathtt{ND}
}(\mathbf{\theta}_{0})$ is $o_{\mathbb{P}}(1)$ because it has mean zero and
its variance is bounded by a constant multiple of
\[
\frac{\mathbb{E}[\bar{d}_{n}^{2\epsilon_{n}}(\mathbf{z})^{2}]}{n\epsilon
_{n}^{4}}=O\left( \frac{1}{nq_{n}\epsilon_{n}^{3}}\right) =O\left( \frac
{1}{r_{n}^{3}\epsilon_{n}^{3}}\right) =o(1).
\]
As a consequence, $\check{H}_{n,kl}^{\mathtt{ND}}\rightarrow_{\mathbb{P}
}H_{0,kl}.$
Next, to show that $R_{n,kl}^{\mathtt{ND}}=o_{\mathbb{P}}(1)$ it suffices to
show that
\[
\frac{1}{\epsilon_{n}^{2}}\sup_{|\mathbf{\theta}-\mathbf{\theta}_{0}|\leq
Kr_{n}^{-1}+2\epsilon_{n}}|\hat{M}_{n}(\mathbf{\theta})-\hat{M}_{n}
(\mathbf{\theta}_{0})-M_{n}(\mathbf{\theta})+M_{n}(\mathbf{\theta}
_{0})|=o_{\mathbb{P}}(1).
\]
The displayed result holds because it follows from \citet*[Theorem
4.2]{Pollard_1989_SS} that
\begin{align*}
\mathbb{E}\left[ \frac{1}{\epsilon_{n}^{2}}\sup_{|\mathbf{\theta
}-\mathbf{\theta}_{0}|\leq Kr_{n}^{-1}+2\epsilon_{n}}|\hat{M}_{n}
(\mathbf{\theta})-\hat{M}_{n}(\mathbf{\theta}_{0})-M_{n}(\mathbf{\theta
})+M_{n}(\mathbf{\theta}_{0})|\right] & =O\left( \sqrt{\frac{\mathbb{E}
[\bar{d}_{n}^{Cr_{n}^{-1}+2\epsilon_{n}}(\mathbf{z})^{2}]}{n\epsilon_{n}^{4}}
}\right) \\
& =O\left( \frac{1}{\sqrt{r_{n}^{3}\epsilon_{n}^{3}}}\right) =o(1).
\end{align*}
Finally, making repeated use of (\ref{Quadratic approximation: M_n}) and the
fact that $r_{n}||\mathbf{\hat{\theta}}-\mathbf{\theta}_{0}||\leq K,$ we have
\[
S_{n,kl}^{\mathtt{ND}}=o_{\mathbb{P}}\left( \frac{1}{r_{n}^{2}\epsilon
_{n}^{2}}+1\right) =o_{\mathbb{P}}(1).
\]
\subsection{Proof of Lemma 2}
Letting $\check{H}_{n,kl}^{\mathtt{ND}},$ $R_{n,kl}^{\mathtt{ND}},$ and
$S_{n,kl}^{\mathtt{ND}}$ be defined as in the proof of Lemma 1, we have
$R_{n,kl}^{\mathtt{ND}}=o_{\mathbb{P}}(1/\sqrt{r_{n}^{3}\epsilon_{n}^{3}})$
because \citet*[Theorem 4.2]{Pollard_1989_SS} can be used to show that for any
$K>0$ and for any $\Delta_{n}>0$ with $\Delta_{n}=o(1),$
\begin{align*}
& \frac{1}{\epsilon_{n}^{2}}\sup_{\substack{\Vert\mathbf{s}-\mathbf{t}
\Vert\leq\Delta_{n} \\\Vert\mathbf{s}\Vert,\Vert\mathbf{t}\Vert\leq K
}}\left\vert \hat{M}_{n}(\mathbf{\theta}_{0}+\epsilon_{n}\mathbf{s})-\hat
{M}_{n}(\mathbf{\theta}_{0}+\epsilon_{n}\mathbf{t})-M_{n}(\mathbf{\theta}
_{0}+\epsilon_{n}\mathbf{s})+M_{n}(\mathbf{\theta}_{0}+\epsilon_{n}
\mathbf{t})\right\vert \\
& =\frac{1}{\epsilon_{n}^{2}}o_{\mathbb{P}}\left( \frac{\sqrt{r_{n}
\epsilon_{n}}}{r_{n}^{2}}\right) =o_{\mathbb{P}}\left( \frac{1}{\sqrt
{r_{n}^{3}\epsilon_{n}^{3}}}\right) .
\end{align*}
Also, Taylor's theorem can be used to show that
\[
S_{n,kl}^{\mathtt{ND}}=-\{\frac{\partial}{\partial\mathbf{\theta}}\ddot
{M}_{n,kl}(\mathbf{\theta}_{0})\mathbf{\}}^{\prime}(\mathbf{\hat{\theta}}
_{n}-\mathbf{\theta}_{0})+o_{\mathbb{P}}(\epsilon_{n}^{2}).
\]
As a consequence, $\tilde{H}_{n,kl}^{\mathtt{ND}}-\check{H}_{n,kl}
^{\mathtt{ND}}=o_{\mathbb{P}}(\epsilon_{n}^{2}+1/\sqrt{r_{n}^{3}\epsilon
_{n}^{3}})+O_{\mathbb{P}}(1/r_{n}),$ where the $O_{\mathbb{P}}(1/r_{n})$ term
\[
-\{\frac{\partial}{\partial\mathbf{\theta}}\ddot{M}_{n,kl}(\mathbf{\theta}
_{0})\mathbf{\}}^{\prime}(\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0})
\]
does not depend on $\epsilon_{n}.$
Next, we approximate the moments of $\check{H}_{n,kl}^{\mathtt{ND}}.$ First,
using Taylor's theorem, it can be shown that
\[
\mathbb{E}[\check{H}_{n,kl}^{\mathtt{ND}}]-H_{n,kl}=-\epsilon_{n}
^{2}\mathsf{B}_{n,kl}+o(\epsilon_{n}^{2}),
\]
where
\[
\mathsf{B}_{n,kl}=-\frac{1}{6}\left[ \frac{\partial^{2}}{\partial\theta
_{k}^{2}}\ddot{M}_{n,kl}(\mathbf{\theta}_{0})+\frac{\partial^{2}}
{\partial\theta_{l}^{2}}\ddot{M}_{n,kl}(\mathbf{\theta}_{0})\right]
\rightarrow-\frac{1}{6}\left[ \frac{\partial^{2}}{\partial\theta_{k}^{2}
}\ddot{M}_{0,kl}(\mathbf{\theta}_{0})+\frac{\partial^{2}}{\partial\theta
_{l}^{2}}\ddot{M}_{0,kl}(\mathbf{\theta}_{0})\right] =\mathsf{B}_{kl}.
\]
Finally, to obtain an expression for the variance of $\check{H}_{n,kl}
^{\mathtt{ND}},$ let $m_{n,kl}^{\Delta}(\mathbf{z})$ denote
\[
m_{n}(\mathbf{z},\mathbf{\theta}_{0}+\epsilon_{n}\mathbf{e}_{k}+\epsilon
_{n}\mathbf{e}_{l})-m_{n}(\mathbf{z},\mathbf{\theta}_{0}+\epsilon
_{n}\mathbf{e}_{k}-\epsilon_{n}\mathbf{e}_{l})-m_{n}(\mathbf{z},\mathbf{\theta
}_{0}-\epsilon_{n}\mathbf{e}_{k}+\epsilon_{n}\mathbf{e}_{l})+m_{n}
(\mathbf{z},\mathbf{\theta}_{0}-\epsilon_{n}\mathbf{e}_{k}-\epsilon
_{n}\mathbf{e}_{l}).
\]
Because
\[
\check{H}_{n,kl}^{\mathtt{ND}}=-\frac{1}{4n\epsilon_{n}^{2}}\sum_{i=1}
^{n}m_{n,kl}^{\Delta}(\mathbf{z}_{i}),
\]
we have
\[
\mathbb{V}[\check{H}_{n,kl}^{\mathtt{ND}}]=\frac{1}{16n\epsilon_{n}^{4}
}\mathbb{V}[m_{n,kl}^{\Delta}(\mathbf{z})]=\frac{1}{16n\epsilon_{n}^{4}
}\mathbb{E}[m_{n,kl}^{\Delta}(\mathbf{z})^{2}]+O\left( \frac{1}{n}\right) .
\]
Also, by condition CRA(iv),
\[
\frac{q_{n}}{\epsilon_{n}}\mathbb{E}[\{m_{n}(\mathbf{z},\mathbf{\theta}
_{0}+\mathbf{s}\epsilon_{n})-m_{n}(\mathbf{z},\mathbf{\theta}_{0}
)\}\{m_{n}(\mathbf{z},\mathbf{\theta}_{0}+\mathbf{t}\epsilon_{n}
)-m_{n}(\mathbf{z},\mathbf{\theta}_{0})\}]\rightarrow\mathcal{C}
_{0}(\mathbf{s},\mathbf{t}).
\]
Therefore,
\[
\mathbb{V}[\check{H}_{n,kl}^{\mathtt{ND}}]=\frac{1}{r_{n}^{3}\epsilon_{n}^{3}
}[\mathsf{V}_{n,kl}+o(1)]+O\left( \frac{1}{n}\right) =\frac{1}{r_{n}
^{3}\epsilon_{n}^{3}}\mathsf{V}_{kl}+o\left( \frac{1}{r_{n}^{3}\epsilon
_{n}^{3}}\right) ,
\]
where, using $\mathcal{C}_{0}(\mathbf{s},-\mathbf{s})=0$ and $\mathcal{C}
_{0}(\mathbf{s},\mathbf{t})=\mathcal{C}_{0}(-\mathbf{s},-\mathbf{t}),$
\begin{align*}
\mathsf{V}_{n,kl} & =\frac{q_{n}}{16\epsilon_{n}}\mathbb{E}[m_{n,kl}^{\Delta
}(\mathbf{z})^{2}]\\
& \rightarrow\frac{1}{8}[\mathcal{C}_{0}(\mathbf{e}_{k}+\mathbf{e}
_{l},\mathbf{e}_{k}+\mathbf{e}_{l})+\mathcal{C}_{0}(\mathbf{e}_{k}
-\mathbf{e}_{l},\mathbf{e}_{k}-\mathbf{e}_{l})-2\mathcal{C}_{0}(\mathbf{e}
_{k}+\mathbf{e}_{l},\mathbf{e}_{k}-\mathbf{e}_{l})-2\mathcal{C}_{0}
(\mathbf{e}_{k}+\mathbf{e}_{l},-\mathbf{e}_{k}+\mathbf{e}_{l})]\\
& =\mathsf{V}_{kl}.
\end{align*}
\subsection{The Benchmark Case}
The remainder of the supplemental appendix verifies Condition CRA for the four
examples in the paper. In three of those examples (namely, maximum score,
panel maximum score, and empirical risk minimization), the function $m_{n}$
does not depend on $n.$ To state a simplified version of Condition CRA
applicable in such cases, let the function $m_{n}$ be denoted by $m_{0}$ and
for any $\delta>0,$ define
\[
\bar{m}_{0}(\mathbf{z})=\sup_{m\in\mathcal{M}_{0}}|m(\mathbf{z})|,\qquad
\mathcal{M}_{0}=\{m_{0}(\cdot,\mathbf{\theta}):\mathbf{\theta}\in
\mathbf{\Theta}\},
\]
and
\[
\bar{d}_{0}^{\delta}(\mathbf{z})=\sup_{d\in\mathcal{D}_{0}^{\delta}
}|d(\mathbf{z})|,\qquad\mathcal{D}_{0}^{\delta}=\{m_{0}(\cdot,\mathbf{\theta
})-m_{0}(\cdot,\mathbf{\theta}_{0}):\mathbf{\theta}\in\mathbf{\Theta}
_{0}^{\delta}\}.
\]
\begin{description}
\item[Condition CRA$_{0}$ (Cube Root Asymptotics, benchmark case)] The
following are satisfied:\newline(i) $\mathcal{M}_{0}$ is manageable for the
envelope $\bar{m}_{0}$ and $\mathbb{E}[\bar{m}_{0}(\mathbf{z})^{2}]<\infty
.$\newline Also, for every $\delta>0,$ $\sup_{\mathbf{\theta}\in
\mathbf{\Theta}\backslash\mathbf{\Theta}_{0}^{\delta}}M_{0}(\mathbf{\theta
})<M_{0}(\mathbf{\theta}_{0}).$\newline(ii) $\mathbf{\theta}_{0}$ is an
interior point of $\mathbf{\Theta}$ and, for some $\delta>0,$ $M_{0}$ is twice
continuously differentiable on $\mathbf{\Theta}_{0}^{\delta}.$ Also,
$\mathbf{H}_{0}=-\partial^{2}M_{0}(\mathbf{\theta}_{0})/\partial
\mathbf{\theta}\partial\mathbf{\theta}^{\prime}$ is positive definite.\newline
(iii) For some $\delta>0,$ $\{\mathcal{D}_{0}^{\delta^{\prime}}:0<\delta
^{\prime}\leq\delta\}$ is uniformly manageable for the envelopes $\bar{d}
_{0}^{\delta^{\prime}}$ and $\sup_{0<\delta^{\prime}\leq\delta}\mathbb{E}
[\bar{d}_{0}^{\delta^{\prime}}(\mathbf{z})^{2}/\delta^{\prime}]<\infty
.$\newline(iv) For every $\delta_{n}>0$ with $\delta_{n}=O(n^{-1/3}),$
$n^{-1/3}\mathbb{E}[\bar{d}_{0}^{\delta_{n}}(\mathbf{z})^{4}]=o(1)$ and, for
all $\mathbf{s},\mathbf{t}\in\mathbb{R}^{d}$ and for some $\mathcal{C}_{0}$
with $\mathcal{C}_{0}(\mathbf{s},\mathbf{s})+\mathcal{C}_{0}(\mathbf{t}
,\mathbf{t})-2\mathcal{C}_{0}(\mathbf{s},\mathbf{t})>0$ for $\mathbf{s\neq
t,}$
\[
\sup_{\mathbf{\theta}\in\mathbf{\Theta}_{0}^{\delta_{n}}}\left\vert \frac
{1}{\delta_{n}}\mathbb{E}[\{m_{0}(\mathbf{z},\mathbf{\theta}+\delta
_{n}\mathbf{s})-m_{0}(\mathbf{z},\mathbf{\theta})\}\{m_{0}(\mathbf{z}
,\mathbf{\theta}+\delta_{n}\mathbf{t})-m_{0}(\mathbf{z},\mathbf{\theta
})\}]-\mathcal{C}_{0}(\mathbf{s},\mathbf{t})\right\vert =o(1).
\]
\newline(v) For every $\delta_{n}>0$ with $\delta_{n}=O(n^{-1/3}),$
\[
\lim_{C\rightarrow\infty}\underset{n\rightarrow\infty}{\lim\sup}\sup
_{0<\delta\leq\delta_{n}}\mathbb{E}[
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(\bar{d}_{0}^{\delta}(\mathbf{z})>C)\bar{d}_{0}^{\delta}(\mathbf{z}
)^{2}/\delta]=0
\]
and $\sup_{\mathbf{\theta,\theta}^{\prime}\in\mathbf{\Theta}_{0}^{\delta_{n}}
}\mathbb{E}[|m_{0}(\mathbf{z},\mathbf{\theta})-m_{0}(\mathbf{z},\mathbf{\theta
}^{\prime})|]/||\mathbf{\theta-\theta}^{\prime}||=O(1).$
\end{description}
\begin{lemma}
\label{[Lemma] Bechmark case}If Condition CRA$_{0}$ is satisfied, then
Condition CRA is satisfied with $q_{n}=1.$
\end{lemma}
\section{Example: Maximum Score}
To state sufficient conditions for Condition CRA$_{0}$ in this example, let
$F_{a|\mathbf{b}}$ denote the conditional distribution function of $a$ given
$\mathbf{b}.$
\begin{description}
\item[Condition MS] For some $\delta>0,$ $S_{F}\geq1,$ and $S_{M}\geq2,$ the
following are satisfied:\newline(i) $0<\mathbb{P}(y=1|\mathbf{x})<1$ almost
surely and $F_{u|x_{1},\mathbf{x}_{2}}(u|x_{1},\mathbf{x}_{2})$ is $S_{F}$
times continuously differentiable in $u$ and $x_{1}$ with bounded
derivatives.\newline(ii) The support of $\mathbf{x}$ is not contained in any
proper linear subspace of $\mathbb{R}^{d+1},$ $\mathbb{E}[\Vert\mathbf{x}
_{2}\Vert^{2}]<\infty,$ and conditional on $\mathbf{x}_{2},$ $x_{1}$ has
everywhere positive Lebesgue density. Also, $F_{x_{1}|\mathbf{x}_{2}}
(x_{1}|\mathbf{x}_{2})$ is $S_{F}$ times continuously differentiable in
$x_{1}$ with bounded derivatives.\newline(iii) $\boldsymbol{\Theta}$ is
compact and $\mathbf{\theta}_{0}$ is an interior point of $\boldsymbol{\Theta
}$.\newline(iv) $M^{\mathtt{MS}}(\mathbf{\theta})=\mathbb{E}[m^{\mathtt{MS}
}(\mathbf{z},\mathbf{\theta})]\ $is $S_{M}$ times continuously differentiable
in $\mathbf{\theta}$ on $\mathbf{\Theta}_{0}^{\delta}$ and
\[
\mathbf{H}^{\mathtt{MS}}=2\mathbb{E}[f_{u|x_{1},\mathbf{x}_{2}}(0|-\mathbf{x}
_{2}^{\prime}\mathbf{\theta}_{0},\mathbf{x}_{2})f_{x_{1}|\mathbf{x}_{2}
}(-\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}|\mathbf{x}_{2})\mathbf{x}
_{2}\mathbf{x}_{2}^{\prime}]
\]
is positive definite.
\end{description}
\begin{CorollaryMS}
\label{[Corollary] MS}Suppose Condition MS is satisfied. Then Condition CRA is
satisfied with $q_{n}=1,$ $\mathbf{H}_{0}=\mathbf{H}^{\mathtt{MS}},$ and
$\mathcal{C}_{0}=\mathcal{C}^{\mathtt{MS}},$ where
\[
\mathcal{C}^{\mathtt{MS}}(\mathbf{s},\mathbf{t})=\mathbb{E}[f_{x_{1}
|\mathbf{x}_{2}}(-\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}|\mathbf{x}
_{2})\min\{|\mathbf{x}_{2}^{\prime}\mathbf{s}|,|\mathbf{x}_{2}^{\prime
}\mathbf{t}|\}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(\operatorname*{sgn}(\mathbf{x}_{2}^{\prime}\mathbf{s})=\operatorname*{sgn}
(\mathbf{x}_{2}^{\prime}\mathbf{t}))].
\]
\end{CorollaryMS}
Alternative representations of $\mathbf{H}^{\mathtt{MS}}$ and $\mathcal{C}
^{\mathtt{MS}}$ are available. In particular, defining
\begin{align*}
\eta^{\mathtt{MS}}(\mathbf{x}_{2}) & =\left. \left\{ \frac{\partial
}{\partial x_{1}}\mathbb{E}(2y-1|x_{1},\mathbf{x}_{2})\right\} f_{x_{1}
|\mathbf{x}_{2}}(x_{1}|\mathbf{x}_{2})\right\vert _{x_{1}=-\mathbf{x}
_{2}^{\prime}\mathbf{\theta}_{0}}\\
& =2f_{u|x_{1},\mathbf{x}_{2}}(0|-\mathbf{x}_{2}^{\prime}\mathbf{\theta}
_{0},\mathbf{x}_{2})f_{x_{1}|\mathbf{x}_{2}}(-\mathbf{x}_{2}^{\prime
}\mathbf{\theta}_{0}|\mathbf{x}_{2})
\end{align*}
and
\begin{align*}
\psi^{\mathtt{MS}}(\mathbf{x}_{2}) & =\left. \mathbb{E[}(2y-1)^{2}
|x_{1},\mathbf{x}_{2}]f_{x_{1}|\mathbf{x}_{2}}(x_{1}|\mathbf{x}_{2}
)\right\vert _{x_{1}=-\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}}\\
& =f_{x_{1}|\mathbf{x}_{2}}(-\mathbf{x}_{2}^{\prime}\mathbf{\theta}
_{0}|\mathbf{x}_{2}),
\end{align*}
we have
\[
\mathbf{H}^{\mathtt{MS}}=\mathbb{E}[\eta^{\mathtt{MS}}(\mathbf{x}
_{2})\mathbf{x}_{2}\mathbf{x}_{2}^{\prime}]
\]
and
\[
\mathcal{C}^{\mathtt{MS}}(\mathbf{s},\mathbf{t})=\mathbb{E[}\psi^{\mathtt{MS}
}(\mathbf{x}_{2})\min\{|\mathbf{x}_{2}^{\prime}\mathbf{s}|,|\mathbf{x}
_{2}^{\prime}\mathbf{t}|\}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(\operatorname*{sgn}(\mathbf{x}_{2}^{\prime}\mathbf{s})=\operatorname*{sgn}
(\mathbf{x}_{2}^{\prime}\mathbf{t}))].
\]
Similar representations will be obtained for the other two maximum score examples.
As an estimator of $\mathbf{H}^{\mathtt{MS}},$ the generic numerical
derivative estimator can be used directly. Another option is to employ a
\textquotedblleft plug-in\textquotedblright\ estimator, where the conditional
densities are replaced by nonparametric estimators thereof. As a third
alternative, consider the example-specific construction $\mathbf{\tilde{H}
}_{n}^{\mathtt{MS}}$ discussed in the paper. To obtain results for that
estimator, we impose some standard conditions on the (derivative of the)
kernel function.
\begin{description}
\item[Condition K] The following are satisfied:\newline(i) $\int_{\mathbb{R}
}\dot{K}(u)^{2}du+\int_{\mathbb{R}}(1+|u|^{3})|\dot{K}(u)|du<\infty.$
\newline(ii) $\int_{\mathbb{R}}\dot{K}(u)du=0,$ $\int_{\mathbb{R}}u\dot
{K}(u)du=-1,$ and $\int_{\mathbb{R}}u^{2}\dot{K}(u)du=0.$\newline(iii)
$\int_{\mathbb{R}}\bar{K}(u)^{2}du<\infty,$ where $\bar{K}(u)=\sup_{v\neq
u}|\dot{K}(v)-\dot{K}(u)|/|v-u|.$
\end{description}
Under Condition K, $\mathbf{\tilde{H}}_{n}^{\mathtt{MS}}$ admits counterparts
of Lemmas 1 and 2 in the paper. To state these, we let $\tilde{H}
_{n,kl}^{\mathtt{MS}}$ and $H_{kl}^{\mathtt{MS}}$ denote element $(k,l)$ of
$\mathbf{\tilde{H}}_{n}^{\mathtt{MS}}$ and $\mathbf{H}^{\mathtt{MS}},$
respectively, and define
\[
\mathsf{B}_{kl}=\mathbb{E}[\{F_{0}^{(1,3)}(\mathbf{x}_{2})+F_{0}
^{(2,2)}(\mathbf{x}_{2})+F_{0}^{(3,1)}(\mathbf{x}_{2})/3\}x_{2,k}x_{2,l}
]\int_{\mathbb{R}}u^{3}\dot{K}(u)du
\]
and
\[
\mathsf{V}_{kl}=2\mathbb{E}[F_{0}^{(0,1)}(\mathbf{x}_{2})x_{2,k}^{2}
x_{2,l}^{2}]\int_{\mathbb{R}}\dot{K}(u)^{2}du,
\]
where $x_{2,k}=\mathbf{e}_{k}^{\prime}\mathbf{x}_{2}$ and
\[
F_{0}^{(i,j)}(\mathbf{x}_{2})=\left. \frac{\partial^{i}}{\partial u^{i}
}F_{u|x_{1},\mathbf{x}_{2}}(-u|x_{1}+u,\mathbf{x}_{2})\frac{\partial^{j}
}{\partial x_{1}^{j}}F_{x_{1}|\mathbf{x}_{2}}(x_{1}|\mathbf{x}_{2})\right\vert
_{u=0,x_{1}=-\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}}.
\]
\begin{LemmaMS}
\label{[Lemma] MS}Suppose Conditions MS and K hold.\newline(i) If
$h_{n}\rightarrow0,$ $nh_{n}^{3}\rightarrow\infty,$ and if $\mathbb{E}
[\Vert\mathbf{x}_{2}\Vert^{6}]<\infty,$ then $\mathbf{\tilde{H}}
_{n}^{\mathtt{MS}}\rightarrow_{\mathbb{P}}\mathbf{H}^{\mathtt{MS}}.$
\newline(ii) If also $S_{F}\geq3$ and $S_{M}\geq4,$ then $\tilde{H}
_{n,kl}^{\mathtt{MS}}$ admits an approximation $\check{H}_{n,kl}^{\mathtt{MS}
}$ satisfying
\[
\tilde{H}_{n,kl}^{\mathtt{MS}}=\check{H}_{n,kl}^{\mathtt{MS}}+o_{\mathbb{P}
}\left( h_{n}^{2}+\frac{1}{\sqrt{nh_{n}^{3}}}\right) +O_{\mathbb{P}}\left(
\frac{1}{\sqrt[3]{n}}\right)
\]
where the $O_{\mathbb{P}}(1/\sqrt[3]{n})$ term does not depend on $h_{n},$ and
where
\[
\mathbb{E}[(\check{H}_{n,kl}^{\mathtt{MS}}-H_{kl}^{\mathtt{MS}})^{2}
]=h_{n}^{4}\mathsf{B}_{kl}^{2}+\frac{1}{nh_{n}^{3}}\mathsf{V}_{kl}+o\left(
h_{n}^{4}+\frac{1}{nh_{n}^{3}}\right) .
\]
\end{LemmaMS}
\subsection{Proof of Corollary \ref{[Corollary] MS}}
By Lemma \ref{[Lemma] Bechmark case}, it suffices to verify that Condition
CRA$_{0}$ is satisfied.\medskip
\emph{Condition CRA}$_{0}$\emph{(i)}. The manageability assumption can be
verified using the same argument as in \citet*{Kim-Pollard_1990_AoS}. Note that
the function $|m^{\mathtt{MS}}(\mathbf{z},\mathbf{\theta})|$ is bounded by
unity in this example, and thus finite second moment condition holds. It is
easy to show that $\mathbf{\theta}_{0}$ uniquely maximizes $M_{0}
(\mathbf{\theta})$ over the parameter set. Well-separatedness follows from
unique maximum, compactness of the parameter space, and continuity of the
function $M_{0}(\mathbf{\theta}).$\medskip
\emph{Condition CRA}$_{0}$\emph{(ii)}. Conditions MS(iii)-(iv) imply this
condition with $\mathbf{H}_{0}=\mathbf{H}^{\mathtt{MS}}.$\medskip
\emph{Condition CRA}$_{0}$\emph{(iii)}. Uniform manageability can be verified
using the same argument as in \citet*{Kim-Pollard_1990_AoS}. Note $d_{0}
^{\delta}(\mathbf{z})=\sup_{\Vert\mathbf{\theta}-\mathbf{\theta}_{0}\Vert
\leq\delta}|
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta}\geq0)-
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}\geq0)|.$ The condition
$\sup_{0<\delta^{\prime}\leq\delta}\mathbb{E}[\bar{d}_{0}^{\delta^{\prime}
}(\mathbf{z})]/\delta^{\prime}<\infty$ is verified in
\citet*{Abrevaya-Huang_2005_ECMA}.\medskip
\emph{Condition CRA}$_{0}$\emph{(iv)}. Since $d_{0}^{\delta}(\mathbf{z}
)^{4}=d_{0}^{\delta}(\mathbf{z}),$ $\mathbb{E}[d_{0}^{\delta_{n}}
(\mathbf{z})^{4}]=O(\delta_{n}),$ which implies the first condition. Also,
\[
\mathcal{C}^{\mathtt{MS}}(\mathbf{s},\mathbf{t})=\mathbb{E}[f_{x_{1}
|\mathbf{x}_{2}}(-\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}|\mathbf{x}
_{2})\min\{|\mathbf{x}_{2}^{\prime}\mathbf{s}|,|\mathbf{x}_{2}^{\prime
}\mathbf{t}|\}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(\operatorname*{sgn}(\mathbf{x}_{2}^{\prime}\mathbf{s})=\operatorname*{sgn}
(\mathbf{x}_{2}^{\prime}\mathbf{t}))]
\]
satisfies $\mathcal{C}^{\mathtt{MS}}(\mathbf{s},\mathbf{s})+\mathcal{C}
^{\mathtt{MS}}(\mathbf{t},\mathbf{t})-2\mathcal{C}^{\mathtt{MS}}
(\mathbf{s},\mathbf{t})>0$ for $\mathbf{s}\neq\mathbf{t.}$ Finally,
$\mathcal{C}^{\mathtt{MS}}$ admits the representation
\[
\mathcal{C}^{\mathtt{MS}}(\mathbf{s},\mathbf{t})=\frac{1}{2}[\mathcal{B}
^{\mathtt{MS}}(\mathbf{s})+\mathcal{B}^{\mathtt{MS}}(\mathbf{t})-\mathcal{B}
^{\mathtt{MS}}(\mathbf{s}-\mathbf{t})],\qquad\mathcal{B}^{\mathtt{MS}
}(\mathbf{s})=\mathbb{E}\left[ f_{x_{1}|\mathbf{x}_{2}}(-\mathbf{x}
_{2}^{\prime}\mathbf{\theta}_{0}|\mathbf{x}_{2})|\mathbf{x}_{2}^{\prime
}\mathbf{s}|\right] .
\]
Using this representation and the fact that $2xy=x^{2}+y^{2}-(x-y)^{2},$ the
displayed part of Condition CRA$_{0}$(iv) can be verified with $\mathcal{C}
_{0}=\mathcal{C}^{\mathtt{MS}}$ by showing that for $\delta_{n}=O(n^{-1/3}), $
\[
\sup_{\mathbf{\theta}\in\boldsymbol{\Theta}_{0}^{\delta_{n}}}\left\vert
\frac{1}{\delta_{n}}\mathbb{E}|m^{\mathtt{MS}}(\mathbf{z},\mathbf{\theta
}+\delta_{n}\mathbf{s})-m^{\mathtt{MS}}(\mathbf{z},\mathbf{\theta}+\delta
_{n}\mathbf{t})|^{2}-\mathcal{B}^{\mathtt{MS}}(\mathbf{s}-\mathbf{t}
)\right\vert =o(1).
\]
Defining $\mathbf{\theta}_{\mathbf{s},n}=\mathbf{\theta}+\delta_{n}\mathbf{s}$
and $\mathbf{\theta}_{\mathbf{t},n}=\mathbf{\theta}+\delta_{n}\mathbf{t,}$ we
have, uniformly in $\mathbf{\theta}\in\boldsymbol{\Theta}_{0}^{\delta_{n}},$
\begin{align*}
& \frac{1}{\delta_{n}}\mathbb{E}|m^{\mathtt{MS}}(\mathbf{z},\mathbf{\theta
}_{\mathbf{s},n})-m^{\mathtt{MS}}(\mathbf{z},\mathbf{\theta}_{\mathbf{t}
,n})|^{2}\\
& =\frac{1}{\delta_{n}}\mathbb{E}\left[
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{\mathbf{s},n}\geq
0>x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{\mathbf{t},n})\right]
+\frac{1}{\delta_{n}}\mathbb{E}\left[
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
\left( x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{\mathbf{t},n}\geq
0>x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{\mathbf{s},n}\right) \right]
\\
& =\frac{1}{\delta_{n}}\mathbb{E}\left[ \int_{-\mathbf{x}_{2}^{\prime
}\mathbf{\theta}_{\mathbf{s},n}}^{-\mathbf{x}_{2}^{\prime}\mathbf{\theta
}_{\mathbf{t},n}}f_{x_{1}|\mathbf{x}_{2}}(x_{1}|\mathbf{x}_{2})dx_{1}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
\left( \mathbf{x}_{2}^{\prime}\mathbf{t}<\mathbf{x}_{2}^{\prime}
\mathbf{s}\right) \right] +\frac{1}{\delta_{n}}\mathbb{E}\left[
\int_{-\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{\mathbf{t},n}}^{-\mathbf{x}
_{2}^{\prime}\mathbf{\theta}_{\mathbf{s},n}}f_{x_{1}|\mathbf{x}_{2}}
(x_{1}|\mathbf{x}_{2})dx_{1}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
\left( \mathbf{x}_{2}^{\prime}\mathbf{s}<\mathbf{x}_{2}^{\prime}
\mathbf{t}\right) \right] \\
& =\mathbb{E}\left[ f_{x_{1}|\mathbf{x}_{2}}(-\mathbf{x}_{2}^{\prime
}\mathbf{\theta}|\mathbf{x}_{2})\mathbf{x}_{2}^{\prime}(\mathbf{s}-\mathbf{t})
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
\left( \mathbf{x}_{2}^{\prime}\mathbf{t}<\mathbf{x}_{2}^{\prime}
\mathbf{s}\right) \right] +\mathbb{E}\left[ f_{x_{1}|\mathbf{x}_{2}
}(-\mathbf{x}_{2}^{\prime}\mathbf{\theta}|\mathbf{x}_{2})\mathbf{x}
_{2}^{\prime}(\mathbf{t}-\mathbf{s})
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
\left( \mathbf{x}_{2}^{\prime}\mathbf{s}<\mathbf{x}_{2}^{\prime}
\mathbf{t}\right) \right] +o(1)\\
& =\mathbb{E}[f_{x_{1}|\mathbf{x}_{2}}(-\mathbf{x}_{2}^{\prime}
\mathbf{\theta}_{0}|\mathbf{x}_{2})|\mathbf{x}_{2}^{\prime}\mathbf{s}
-\mathbf{x}_{2}^{\prime}\mathbf{t}|]+o(1),
\end{align*}
from which the desired result follows.\medskip
\emph{Condition CRA}$_{0}$\emph{(v)}. The first part easily follows from
$\bar{d}_{0}^{\delta}(\mathbf{z})\leq1,$ while the second part follows from
the verification of Condition CRA$_{0}$(iv).
\subsection{Proof of Lemma \ref{[Lemma] MS}}
\subsubsection{Part (i) [Consistency]}
Defining
\[
\mathbf{\check{H}}_{n}^{\mathtt{MS}}=-\frac{1}{n}\sum_{i=1}^{n}(2y_{i}
-1)\dot{K}_{n}(x_{1i}+\mathbf{x}_{2i}^{\prime}\mathbf{\theta}_{0}
)\mathbf{x}_{2i}\mathbf{x}_{2i}^{\prime},\qquad\mathbf{\bar{H}}_{n}
^{\mathtt{MS}}(\mathbf{\theta})=-\mathbb{E}[(2y-1)\dot{K}_{n}(x_{1}
+\mathbf{x}_{2}^{\prime}\mathbf{\theta})\mathbf{x}_{2}\mathbf{x}_{2}^{\prime
}],
\]
we obtain the decomposition
\[
\mathbf{\tilde{H}}_{n}^{\mathtt{MS}}=\mathbf{\check{H}}_{n}^{\mathtt{MS}
}+\mathbf{R}_{n}^{\mathtt{MS}}+\mathbf{S}_{n}^{\mathtt{MS}},
\]
where
\[
\mathbf{R}_{n}^{\mathtt{MS}}=\mathbf{\tilde{H}}_{n}^{\mathtt{MS}
}-\mathbf{\check{H}}_{n}^{\mathtt{MS}}-\mathbf{\bar{H}}_{n}^{\mathtt{MS}
}(\mathbf{\hat{\theta}}_{n}^{\mathtt{MS}})+\mathbf{\bar{H}}_{n}^{\mathtt{MS}
}(\mathbf{\theta}_{0}),\qquad\mathbf{S}_{n}^{\mathtt{MS}}=\mathbf{\bar{H}}
_{n}^{\mathtt{MS}}(\mathbf{\hat{\theta}}_{n}^{\mathtt{MS}})-\mathbf{\bar{H}
}_{n}^{\mathtt{MS}}(\mathbf{\theta}_{0}).
\]
\newline The proof will be completed by showing that $\mathbf{\check{H}}
_{n}^{\mathtt{MS}}\rightarrow_{\mathbb{P}}\mathbf{H}^{\mathtt{MS}}
,\mathbf{R}_{n}^{\mathtt{MS}}=o_{\mathbb{P}}(1),$ and $\mathbf{S}
_{n}^{\mathtt{MS}}=o_{\mathbb{P}}(1).$
First, using the dominated convergence theorem and $\int_{\mathbb{R}}u\dot
{K}(u)du=-1,$ we have
\begin{align*}
\mathbf{\bar{H}}_{n}^{\mathtt{MS}}(\mathbf{\theta}_{0}) & =-\mathbb{E}
[(2y-1)\dot{K}_{n}\left( x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta}
_{0}\right) \mathbf{x}_{2}\mathbf{x}_{2}^{\prime}]\\
& =-\mathbb{E}\left[ \int_{\mathbb{R}}\frac{1-2F_{u|x_{1},\mathbf{x}_{2}
}(-uh_{n}|uh_{n}-\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0},\mathbf{x}_{2}
)}{h_{n}}f_{x_{1}|\mathbf{x}_{2}}(uh_{n}-\mathbf{x}_{2}^{\prime}
\mathbf{\theta}_{0}|\mathbf{x}_{2})\dot{K}(u)du\mathbf{x}_{2}\mathbf{x}
_{2}^{\prime}\right] \\
& \rightarrow2\mathbb{E}[F_{0}^{(1,1)}(\mathbf{x}_{2})\mathbf{x}
_{2}\mathbf{x}_{2}^{\prime}]\int_{\mathbb{R}}u\dot{K}(u)du\\
& =2\mathbb{E}[f_{u|x_{1},\mathbf{x}_{2}}(0|-\mathbf{x}_{2}^{\prime
}\mathbf{\theta}_{0},\mathbf{x}_{2})f_{x_{1}|\mathbf{x}_{2}}(-\mathbf{x}
_{2}^{\prime}\mathbf{\theta}_{0}|\mathbf{x}_{2})\mathbf{x}_{2}\mathbf{x}
_{2}^{\prime}]=\mathbf{H}^{\mathtt{MS}}.
\end{align*}
Moreover, $\mathbf{\check{H}}_{n}^{\mathtt{MS}}-\mathbf{\bar{H}}
_{n}^{\mathtt{MS}}(\mathbf{\theta}_{0})=o_{\mathbb{P}}(1)$ because each
element has mean zero and a variance that is bounded by a constant multiple
of
\[
\frac{\mathbb{E}[\dot{K}_{n}(x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta}
_{0})^{2}]}{n}=O\left( \frac{1}{nh_{n}^{3}}\right) =o(1).
\]
As a consequence, $\mathbf{\check{H}}_{n}^{\mathtt{MS}}\rightarrow
_{\mathbb{P}}\mathbf{H}^{\mathtt{MS}}.$
Next, $\mathbf{R}_{n}^{\mathtt{MS}}=o_{\mathbb{P}}(1/\sqrt{nh_{n}^{3}
})=o_{\mathbb{P}}(1)$ follows from \citet*[Theorem 4.2]{Pollard_1989_SS} if it
can be shown that, for every $C>0,$
\[
h_{n}^{3}\mathbb{E}\left[ \sup_{||\mathbf{\theta}-\mathbf{\theta}_{0}||\leq
Cn^{-1/3}}|\dot{K}_{n}(x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta})-\dot
{K}_{n}(x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0})|^{2}\Vert
\mathbf{x}_{2}\Vert^{4}\right] =o(1).
\]
Defining $\bar{K}_{n}(u)=\bar{K}(u/h_{n})/h_{n},$ we have, by Condition
K(iii),
\[
\sup_{||\mathbf{\theta}-\mathbf{\theta}_{0}||\leq Cn^{-1/3}}|\dot{K}_{n}
(x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta})-\dot{K}_{n}(x_{1}
+\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0})|\leq\frac{C}{n^{1/3}h_{n}^{2}
}\bar{K}_{n}(x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0})\Vert
\mathbf{x}_{2}\Vert,
\]
and therefore, using $nh_{n}^{3}\rightarrow\infty,$
\begin{align*}
& h_{n}^{3}\mathbb{E}\left[ \sup_{||\mathbf{\theta}-\mathbf{\theta}_{0}||\leq
Cn^{-1/3}}|\dot{K}_{n}(x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta})-\dot
{K}_{n}(x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0})|^{2}\Vert
\mathbf{x}_{2}\Vert^{4}\right] \\
& \leq\frac{C^{2}}{n^{2/3}h_{n}}\mathbb{E}[\bar{K}_{n}(x_{1}+\mathbf{x}
_{2}^{\prime}\mathbf{\theta}_{0})^{2}\Vert\mathbf{x}_{2}\Vert^{6}]=O\left(
\frac{1}{n^{2/3}h_{n}^{2}}\right) =o(1).
\end{align*}
Finally, defining
\begin{align*}
\xi_{n}(u,\mathbf{\delta},\mathbf{x}_{2}) & =\frac{1-2F_{u|x_{1}
,\mathbf{x}_{2}}(-uh_{n}+\mathbf{x}_{2}^{\prime}\mathbf{\delta}|uh_{n}
-\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}-\mathbf{x}_{2}^{\prime
}\mathbf{\delta},\mathbf{x}_{2})}{h_{n}}f_{x_{1}|\mathbf{x}_{2}}
(uh_{n}-\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}-\mathbf{x}_{2}^{\prime
}\mathbf{\delta}|\mathbf{x}_{2})\\
& -\frac{1-2F_{u|x_{1},\mathbf{x}_{2}}(-uh_{n}|uh_{n}-\mathbf{x}_{2}^{\prime
}\mathbf{\theta}_{0},\mathbf{x}_{2})}{h_{n}}f_{x_{1}|\mathbf{x}_{2}}
(uh_{n}-\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}|\mathbf{x}_{2}),
\end{align*}
we have
\begin{align*}
\sup_{||\mathbf{\theta}-\mathbf{\theta}_{0}||\leq Cn^{-1/3}}\left\Vert
\mathbf{\bar{H}}_{n}^{\mathtt{MS}}(\mathbf{\theta})-\mathbf{\bar{H}}
_{n}^{\mathtt{MS}}(\mathbf{\theta}_{0})\right\Vert & =\sup_{||\mathbf{\delta
}||\leq Cn^{-1/3}}\left\Vert \mathbb{E}\left[ \int_{\mathbb{R}}\xi
_{n}(u,\mathbf{\delta},\mathbf{x}_{2})\dot{K}(u)du\mathbf{x}_{2}\mathbf{x}
_{2}^{\prime}\right] \right\Vert \\
& \leq\mathbb{E}\left[ \left\{ \int_{\mathbb{R}}\sup_{||\mathbf{\delta
}||\leq Cn^{-1/3}}|\xi_{n}(u,\mathbf{\delta},\mathbf{x}_{2})||\dot
{K}(u)|du\right\} ||\mathbf{x}_{2}||^{2}\right] \\
& \rightarrow0
\end{align*}
for any $C>0,$ where the last line uses the dominated convergence theorem.
\subsubsection{Part (ii) [Approximate MSE]}
It was shown in the proof of part (i) that $R_{n,kl}^{\mathtt{MS}
}=o_{\mathbb{P}}(1/\sqrt{nh_{n}^{3}}).$ Also, Taylor's theorem and Condition
K(ii) can be used to show that for any $C>0,$ we have, uniformly in
$||\mathbf{\delta}_{n}||\leq C/\sqrt[3]{n},$
\begin{align*}
\bar{H}_{n,kl}^{\mathtt{MS}}(\mathbf{\theta}_{0}+\mathbf{\delta}_{n}) &
=H_{kl}^{\mathtt{MS}}+h_{n}^{2}\mathbb{E}[\{F_{0}^{(1,3)}(\mathbf{x}
_{2})+F_{0}^{(2,2)}(\mathbf{x}_{2})+F_{0}^{(3,1)}(\mathbf{x}_{2}
)/3\}x_{2,k}x_{2,l}]\int_{\mathbb{R}}u^{3}\dot{K}(u)du\\
& +\{4\mathbb{E}[F_{0}^{(1,2)}(\mathbf{x}_{2})x_{2,k}x_{2,l}\mathbf{x}
_{2}]+2\mathbb{E}[F_{0}^{(2,1)}(\mathbf{x}_{2})x_{2,k}x_{2,l}\mathbf{x}
_{2}]\}^{\prime}\mathbf{\delta}_{n}+o(h_{n}^{2}),
\end{align*}
implying in particular that
\[
S_{n,kl}^{\mathtt{MS}}=\{4\mathbb{E}[F_{0}^{(1,2)}(\mathbf{x}_{2}
)x_{2,k}x_{2,l}\mathbf{x}_{2}]+2\mathbb{E}[F_{0}^{(2,1)}(\mathbf{x}
_{2})x_{2,k}x_{2,l}\mathbf{x}_{2}]\}^{\prime}(\mathbf{\hat{\theta}}
_{n}-\mathbf{\theta}_{0})+o_{\mathbb{P}}(h_{n}^{2}).
\]
As a consequence, $\tilde{H}_{n,kl}^{\mathtt{ND}}-\check{H}_{n,kl}
^{\mathtt{ND}}=o_{\mathbb{P}}(h_{n}^{2}+1/\sqrt{nh_{n}^{3}})+O_{\mathbb{P}
}(1/\sqrt[3]{n}),$ where the $O_{\mathbb{P}}(1/\sqrt[3]{n})$ term
\[
\{4\mathbb{E}[F_{0}^{(1,2)}(\mathbf{x}_{2})x_{2,k}x_{2,l}\mathbf{x}
_{2}]+2\mathbb{E}[F_{0}^{(2,1)}(\mathbf{x}_{2})x_{2,k}x_{2,l}\mathbf{x}
_{2}]\}^{\prime}(\mathbf{\hat{\theta}}_{n}-\mathbf{\theta}_{0})
\]
does not depend on $h_{n}.$
Next, we approximate the moments of $\check{H}_{n,kl}^{\mathtt{MS}}.$ By the
previous paragraph,
\[
\mathbb{E}[\check{H}_{n,kl}^{\mathtt{MS}}]-H_{kl}^{\mathtt{MS}}=\bar{H}
_{n,kl}^{\mathtt{MS}}(\mathbf{\theta}_{0})-H_{kl}^{\mathtt{MS}}=h_{n}
^{2}\mathsf{B}_{kl}+o(\epsilon_{n}^{2}),
\]
where
\[
\mathsf{B}_{kl}=\mathbb{E}[\{F_{0}^{(1,3)}(\mathbf{x}_{2})+F_{0}
^{(2,2)}(\mathbf{x}_{2})+F_{0}^{(3,1)}(\mathbf{x}_{2})/3\}x_{2,k}x_{2,l}
]\int_{\mathbb{R}}u^{3}\dot{K}(u)du.
\]
Also,
\begin{align*}
\mathbb{V}[\check{H}_{n,kl}^{\mathtt{MS}}] & =\frac{1}{n}\mathbb{V}
[(2y-1)\dot{K}_{n}(x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}
)x_{2,k}x_{2,l}]=\frac{1}{n}\mathbb{V}[\dot{K}_{n}(x_{1}+\mathbf{x}
_{2}^{\prime}\mathbf{\theta}_{0})x_{2,k}x_{2,l}]\\
& =\frac{1}{n}\mathbb{E}[\dot{K}_{n}(x_{1}+\mathbf{x}_{2}^{\prime
}\mathbf{\theta}_{0})^{2}x_{2,k}^{2}x_{2,l}^{2}]+O\left( \frac{1}{n}\right)
\\
& =\frac{1}{nh_{n}^{3}}\mathsf{V}_{kl}+o\left( \frac{1}{nh_{n}^{3}}\right) ,
\end{align*}
where
\begin{align*}
\mathsf{V}_{kl} & =\lim_{n\rightarrow\infty}h_{n}^{3}\mathbb{E}[\dot{K}
_{n}(x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0})^{2}x_{2,k}^{2}
x_{2,l}^{2}]=\mathbb{E}[f_{x_{1}|\mathbf{x}_{2}}(-\mathbf{x}_{2}^{\prime
}\mathbf{\theta}_{0}|\mathbf{x}_{2})x_{2,k}^{2}x_{2,l}^{2}]\int_{\mathbb{R}
}\dot{K}(u)^{2}du\\
& =2\mathbb{E}[F_{0}^{(0,1)}(\mathbf{x}_{2})x_{2,k}^{2}x_{2,l}^{2}
]\int_{\mathbb{R}}\dot{K}(u)^{2}du.
\end{align*}
\subsection{Rule-of-Thumb Bandwidth Selection}
We provide details on the rule-of-thumb (ROT) bandwidth selection rules used
in the simulations reported below. To construct ROT bandwidths, we choose a
reference model involving finite dimensional parameters and
calculate/approximate the corresponding leading constants entering the
approximate MSE of $\mathbf{\tilde{H}}_{n}^{\mathtt{MS}}$ and $\mathbf{\tilde
{H}}_{n}^{\mathtt{ND}}.$
Specifically, we assume $u|\mathbf{x}\thicksim\mathcal{N}(0,\sigma_{u}
^{2}(\mathbf{x}))$ and $x_{1}|\mathbf{x}_{2}\thicksim\mathcal{N}(\mu
_{1},\sigma_{1}^{2}),\,$where we will specify some parametric specification on
$\sigma_{u}^{2}(\mathbf{x})=\sigma_{u}^{2}(x_{1},\mathbf{x}_{2}).$ Then, in
this reference model, $F_{0}^{(2,2)}(\mathbf{x}_{2})=0,$
\[
F_{0}^{(1,3)}(\mathbf{x}_{2})=-\frac{\phi(0)}{\sigma_{u}(-\mathbf{x}
_{2}^{\prime}\mathbf{\theta}_{0},\mathbf{x}_{2})\sigma_{1}^{3}}\phi\left(
\frac{\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}+\mu_{1}}{\sigma_{1}}\right)
\left[ \left( \frac{\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}+\mu_{1}
}{\sigma_{1}}\right) ^{2}-1\right] ,
\]
and
\[
F_{0}^{(3,1)}(\mathbf{x}_{2})=\left. \frac{\phi(0)}{\sigma_{u}^{3}
(\mathbf{x})\sigma_{1}}\phi\left( \frac{\mathbf{x}_{2}^{\prime}
\mathbf{\theta}_{0}+\mu_{1}}{\sigma_{1}}\right) [1-\ddot{\sigma}
_{u}(\mathbf{x})\sigma_{u}(\mathbf{x})+2\dot{\sigma}_{u}(\mathbf{x}
)^{2}]\right\vert _{x_{1}=-\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}},
\]
where $\phi$ is the standard normal density and where $\dot{\sigma}
_{u}(\mathbf{x})=\partial\sigma_{u}(\mathbf{x})/\partial x_{1}$ and
$\ddot{\sigma}_{u}(\mathbf{x})=\partial^{2}\sigma_{u}(\mathbf{x})/\partial
x_{1}^{2}.$
\subsubsection{Plug-in Estimator $\mathbf{\tilde{H}}_{n}^{\mathtt{MS}}$}
Given our reference model, natural estimators of the bias constants
\[
\mathsf{B}_{kl}=\mathbb{E}[\{F_{0}^{(1,3)}(\mathbf{x}_{2})+F_{0}
^{(3,1)}(\mathbf{x}_{2})/3\}x_{2,k}x_{2,l}]\int_{\mathbb{R}}u^{3}\dot{K}(u)du
\]
are
\[
\left[ \frac{1}{n}\sum_{i=1}^{n}\{\hat{F}_{n}^{(1,3)}(\mathbf{x}_{2i}
)+\hat{F}_{n}^{(3,1)}(\mathbf{x}_{2i})/3\}\mathbf{e}_{k}^{\prime}
\mathbf{x}_{2i}\mathbf{e}_{l}^{\prime}\mathbf{x}_{2i}\right] \int
_{\mathbb{R}}u^{3}\dot{K}(u)du,
\]
where $\hat{F}_{n}^{(1,3)}$ and $\hat{F}_{n}^{(3,1)}$ are constructed using
maximum likelihood for the parametric reference model (i.e., heteroskedastic
Probit) together with a flexible parametric specification $\sigma_{u}
^{2}(\mathbf{x})=\mathbf{\gamma}^{\prime}\mathbf{p}(\mathbf{x})$ for
$\sigma_{u}^{2}(\mathbf{x}),\ $with $\mathbf{p}(\mathbf{x})$ denoting a
polynomial expansion.
Similarly, natural estimators of the variance constants
\[
\mathsf{V}=2\mathbb{E}[F_{0}^{(0,1)}(\mathbf{x}_{2})x_{2,k}^{2}x_{2,l}
^{2}]\int_{\mathbb{R}}\dot{K}(u)^{2}du,
\]
are given by
\[
\mathsf{\hat{V}}_{n}=2\left[ \frac{1}{n}\sum_{i=1}^{n}\hat{F}_{n}
^{(0,1)}(\mathbf{x}_{2i})(\mathbf{e}_{k}^{\prime}\mathbf{x}_{2i}
)^{2}(\mathbf{e}_{l}^{\prime}\mathbf{x}_{2i})^{2}\right] \int_{\mathbb{R}
}\dot{K}(u)^{2}du.
\]
\subsubsection{Numerical Differentiation Estimator $\mathbf{\tilde{H}}
_{n}^{\mathtt{ND}}$}
In our reference model, the bias constants are of the form
\[
\mathsf{B}_{kl}=-\mathbb{E}[\{F_{0}^{(1,3)}(\mathbf{x}_{2})+F_{0}
^{(3,1)}(\mathbf{x}_{2})/3\}\{x_{2,k}^{3}x_{2,l}+x_{2,k}x_{2,l}^{3}\}],
\]
natural estimators of which are given by
\[
-\frac{1}{n}\sum_{i=1}^{n}\{\hat{F}_{n}^{(1,3)}(\mathbf{x}_{2i})+\hat{F}
_{n}^{(3,1)}(\mathbf{x}_{2i})/3\}\{(\mathbf{e}_{k}^{\prime}\mathbf{x}
_{2i})^{3}(\mathbf{e}_{l}^{\prime}\mathbf{x}_{2i})+(\mathbf{e}_{k}^{\prime
}\mathbf{x}_{2i})(\mathbf{e}_{l}^{\prime}\mathbf{x}_{2i})^{3}\}.
\]
Similarly, natural estimators of the variance constants
\[
\mathsf{V}_{kl}=\{2\mathcal{B}_{0}(\mathbf{e}_{k})+2\mathcal{B}_{0}
(\mathbf{e}_{l})-\mathcal{B}_{0}(\mathbf{e}_{k}+\mathbf{e}_{l})-\mathcal{B}
_{0}(\mathbf{e}_{k}-\mathbf{e}_{l})\}/16,\qquad\mathcal{B}_{0}(\mathbf{s}
)=2\mathbb{E}[F_{0}^{(0,1)}(\mathbf{x}_{2})|\mathbf{x}_{2}^{\prime}
\mathbf{s}|],
\]
are given by
\[
\frac{1}{8n}\sum_{i=1}^{n}\{2|\mathbf{e}_{k}^{\prime}\mathbf{x}_{2i}
|+2|\mathbf{e}_{l}^{\prime}\mathbf{x}_{2i}|-|(\mathbf{e}_{k}+\mathbf{e}
_{l})^{\prime}\mathbf{x}_{2i}|-|(\mathbf{e}_{k}-\mathbf{e}_{l})^{\prime
}\mathbf{x}_{2i}|\}\hat{F}_{n}^{(0,1)}(\mathbf{x}_{2i}).
\]
\section{Example: Panel Maximum Score}
To state sufficient conditions for Condition CRA$_{0}$ in this example, define
\[
\eta^{\mathtt{PMS}}(\mathbf{x}_{2})=\left. \left\{ \frac{\partial}{\partial
x_{1}}\mathbb{E}(y|x_{1},\mathbf{x}_{2})\right\} f_{x_{1}|\mathbf{x}_{2}
}(x_{1}|\mathbf{x}_{2})\right\vert _{x_{1}=-\mathbf{x}_{2}^{\prime
}\mathbf{\theta}_{0}}.
\]
\begin{description}
\item[Condition PMS] For some $\delta>0,$ the following are satisfied:\newline
(i) For every $u\in\mathbb{R},$ $0<F_{u_{1}|\mathbf{X}_{1},\mathbf{X}
_{2},\alpha}(u_{1}|\mathbf{X}_{1},\mathbf{X}_{2},\alpha)=F_{u_{2}
|\mathbf{X}_{1},\mathbf{X}_{2},\alpha}(u_{2}|\mathbf{X}_{1},\mathbf{X}
_{2},\alpha)<1$ almost surely. Also, $\mathbb{E}[y|x_{1},\mathbf{x}_{2}]$ is
continuously differentiable in $x_{1}$ with bounded derivative, and
$\mathbb{E}[y^{2}|x_{1},\mathbf{x}_{2}]$ is continuous in $x_{1}.$\newline(ii)
The support of $\mathbf{x}$ is not contained in any proper linear subspace of
$\mathbb{R}^{d+1},$ $\mathbb{E}[\Vert\mathbf{x}_{2}\Vert^{2}]<\infty,$ and
conditional on $\mathbf{x}_{2},$ $x_{1}$ has everywhere positive Lebesgue
density. Also, $F_{x_{1}|\mathbf{x}_{2}}(x_{1}|\mathbf{x}_{2})$ is
continuously differentiable in $x_{1}$ with bounded derivative.\newline(iii)
$\boldsymbol{\Theta}$ is compact and $\mathbf{\theta}_{0}$ is an interior
point of $\boldsymbol{\Theta}.$\newline(iv) $M^{\mathtt{PMS}}(\mathbf{\theta
})=\mathbb{E}[m^{\mathtt{PMS}}(\mathbf{z},\mathbf{\theta})]\ $is twice
continuously differentiable in $\mathbf{\theta}$ on $\mathbf{\Theta}
_{0}^{\delta}$ and
\[
\mathbf{H}^{\mathtt{PMS}}=\mathbb{E}[\eta^{\mathtt{PMS}}(\mathbf{x}
_{2})\mathbf{x}_{2}\mathbf{x}_{2}^{\prime}]
\]
is positive definite.
\end{description}
Letting
\[
\psi^{\mathtt{PMS}}(\mathbf{x}_{2})=\left. \mathbb{E}(y^{2}|x_{1}
,\mathbf{x}_{2})f_{x_{1}|\mathbf{x}_{2}}(x_{1}|\mathbf{x}_{2})\right\vert
_{x_{1}=-\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}}
\]
and proceeding as in the proof of Corollary \ref{[Corollary] MS}, the
following result is obtained.
\begin{CorollaryPMS}
\label{[Corollary] PMS}Suppose Condition PMS is satisfied. Then Condition CRA
is satisfied with $q_{n}=1,$ $\mathbf{H}_{0}=\mathbf{H}^{\mathtt{PMS}},$ and
$\mathcal{C}_{0}=\mathcal{C}^{\mathtt{PMS}},$ where
\[
\mathcal{C}^{\mathtt{PMS}}(\mathbf{s},\mathbf{t})=\mathbb{E[}\psi
^{\mathtt{PMS}}(\mathbf{x}_{2})\min\{|\mathbf{x}_{2}^{\prime}\mathbf{s}
|,|\mathbf{x}_{2}^{\prime}\mathbf{t}|\}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(\operatorname*{sgn}(\mathbf{x}_{2}^{\prime}\mathbf{s})=\operatorname*{sgn}
(\mathbf{x}_{2}^{\prime}\mathbf{t}))].
\]
\end{CorollaryPMS}
The case-specific estimator $\mathbf{\tilde{H}}_{n}^{\mathtt{PMS}}$ of
$\mathbf{H}^{\mathtt{PMS}}$ admits a counterpart of Lemma \ref{[Lemma] MS},
but for brevity we omit a precise statement.
\section{Example: Conditional Maximum Score}
To state sufficient conditions for Condition CRA in this example, let
$\mathcal{X}$ denote the support of $\mathbf{x}=(x_{1},\mathbf{x}_{2}^{\prime
})^{\prime}$ and for $\delta>0,$ let $\mathcal{W}^{\delta}=\{\mathbf{w\in
}\mathbb{R}^{d}:||\mathbf{w}||\leq\delta\}.$ Also, define
\[
\mu^{\mathtt{CMS}}(\mathbf{w};\mathbf{\theta})=\mathbb{E}\left[ \left. y
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(x_{1}+\mathbf{x}_{2}^{\prime}\mathbf{\theta}\geq0)\right\vert \mathbf{w}
\right] f_{\mathbf{w}}(\mathbf{w),}
\]
\[
\dot{\mu}^{\mathtt{CMS}}(\mathbf{w};\mathbf{\theta})=\frac{\partial}
{\partial\mathbf{\theta}}\mu^{\mathtt{CMS}}(\mathbf{w};\mathbf{\theta
})=\mathbb{E}\left[ \left. \left. \{\mathbb{E}(y|x_{1},\mathbf{x}
_{2},\mathbf{w})f_{x_{1}|\mathbf{x}_{2},\mathbf{w}}(x_{1}|\mathbf{x}
_{2},\mathbf{w})\}\right\vert _{x_{1}=-\mathbf{x}_{2}^{\prime}\mathbf{\theta}
}\mathbf{x}_{2}\right\vert \mathbf{w}\right] f_{\mathbf{w}}(\mathbf{w),}
\]
\[
\ddot{\mu}^{\mathtt{CMS}}(\mathbf{w};\mathbf{\theta})=\frac{\partial^{2}
}{\partial\mathbf{\theta}\partial\mathbf{\theta}^{\prime}}\mu^{\mathtt{CMS}
}(\mathbf{w};\mathbf{\theta})\mathbf{,}
\]
and
\[
\eta^{\mathtt{CMS}}(\mathbf{x}_{2})=\left. \left\{ \frac{\partial}{\partial
x_{1}}\mathbb{E}(y|x_{1},\mathbf{x}_{2},\mathbf{w})\right\} f_{x_{1}
|\mathbf{x}_{2},\mathbf{w}}(x_{1}|\mathbf{x}_{2},\mathbf{w})\right\vert
_{x_{1}=-\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0},\mathbf{w=0}}.
\]
\begin{description}
\item[Condition CMS] For some $\delta>0$ and $P\geq1,$ the following are
satisfied:\newline(i) For some strictly increasing $F,$
\[
\mathbb{P}(Y_{t}=1|\mathbf{X}_{1},\mathbf{X}_{2},\mathbf{X}_{3},\alpha
,Y_{0},\dots,Y_{t-1})=F[X_{1t}+(\mathbf{X}_{2t}^{\prime},Y_{t-1}
)\mathbf{\theta}_{0}+\alpha],\qquad t=1,2,3.
\]
Also, on $\mathcal{X}\times\mathcal{W}^{\delta},$ $\mathbb{E}(y|x_{1}
,\mathbf{x}_{2},\mathbf{w})$ is differentiable in $x_{1},$ $\partial
\mathbb{E}(y|x_{1},\mathbf{x}_{2},\mathbf{w})/\partial x_{1}$ is bounded and
continuous in $(x_{1},\mathbf{w}),$ and $\mathbb{E}(y^{2}|x_{1},\mathbf{x}
_{2},\mathbf{w})$ is positive and continuous in $(x_{1},\mathbf{w}).$
\newline(ii) $\mathbb{E}[\Vert\mathbf{x}_{2}\Vert^{2}|\mathbf{w}]$ is bounded
on $\mathcal{W}^{\delta}$ and for every $\mathbf{w}\in\mathcal{W}^{\delta},$
the support of $\mathbf{x}$ given $\mathbf{w}$ is not contained in any proper
linear subspace of $\mathbb{R}^{d+1}.$ Also, on $\mathcal{X}\times
\mathcal{W}^{\delta},$ $f_{x_{1}|\mathbf{x}_{2},\mathbf{w}}(x_{1}
|\mathbf{x}_{2},\mathbf{w})$ is positive, bounded, and continuous in
$(x_{1},\mathbf{w}) $ and $f_{\mathbf{w}}(\mathbf{w})$ is positive and
continuous in $\mathbf{w}. $\newline(iii) $\boldsymbol{\Theta}$ is compact and
$\mathbf{\theta}_{0}$ is an interior point of $\boldsymbol{\Theta}.$
\newline(iv) $\mu^{\mathtt{CMS}}(\mathbf{w};\mathbf{\theta})$ is twice
continuously differentiable in $\mathbf{\theta}$ on $\mathbf{\Theta}
_{0}^{\delta}$ with bounded derivatives, $\mu^{\mathtt{CMS}}(\mathbf{w}
;\mathbf{\theta})$ is uniformly (in $\mathbf{\theta}\in\mathbf{\Theta}$)
continuous in $\mathbf{w}$ at $\mathbf{0,}$ $\dot{\mu}^{\mathtt{CMS}
}(\mathbf{w};\mathbf{\theta}_{0})$ is $P$ times continuously differentiable in
$\mathbf{w}$ on $\mathcal{W}^{\delta},$ $\ddot{\mu}^{\mathtt{CMS}}
(\mathbf{w};\mathbf{\theta})$ is uniformly (in $\mathbf{\theta}\in
\mathbf{\Theta}_{0}^{\delta}$) continuous in $\mathbf{w}$ at $\mathbf{0,}$ and
\[
\mathbf{H}^{\mathtt{CMS}}=\left. \mathbb{E}\left[ \left. \eta
^{\mathtt{CMS}}(\mathbf{x}_{2})\mathbf{x}_{2}\mathbf{x}_{2}^{\prime
}\right\vert \mathbf{w}\right] f_{\mathbf{w}}(\mathbf{w)}\right\vert
_{\mathbf{w=0}}
\]
is positive definite.\newline(v) $\kappa$ is bounded, of order $P,$ and
supported on $[-1,1]^{d}.$ Also, $nb_{n}^{d}\rightarrow\infty$ and
$nb_{n}^{d+3P}\rightarrow0.$
\end{description}
Let
\[
\psi^{\mathtt{CMS}}(\mathbf{x}_{2})=\left. \mathbb{E}(y^{2}|x_{1}
,\mathbf{x}_{2},\mathbf{w})f_{x_{1}|\mathbf{x}_{2},\mathbf{w}}(x_{1}
|\mathbf{x}_{2},\mathbf{w})\right\vert _{x_{1}=-\mathbf{x}_{2}^{\prime
}\mathbf{\theta}_{0},\mathbf{w=0}}.
\]
\begin{CorollaryCMS}
\label{[Corollary] CMS}Suppose Condition CMS is satisfied. Then Condition CRA
is satisfied with $q_{n}=b_{n}^{d},$ $\mathbf{H}_{0}=\mathbf{H}^{\mathtt{CMS}
},$ and $\mathcal{C}_{0}=\mathcal{C}^{\mathtt{CMS}},$ where
\[
\mathcal{C}^{\mathtt{CMS}}(\mathbf{s},\mathbf{t})=\left. \mathbb{E}\left[
\left. \psi^{\mathtt{CMS}}(\mathbf{x}_{2})\min\{|\mathbf{x}_{2}^{\prime
}\mathbf{s}|,|\mathbf{x}_{2}^{\prime}\mathbf{t}|\}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
\{\operatorname*{sgn}(\mathbf{x}_{2}^{\prime}\mathbf{s})=\operatorname*{sgn}
(\mathbf{x}_{2}^{\prime}\mathbf{t})\}\right\vert \mathbf{w}\right]
f_{\mathbf{w}}(\mathbf{w)}\right\vert _{\mathbf{w=0}}\cdot\int_{\mathbb{R}
^{d}}\kappa(\mathbf{v})^{2}d\mathbf{v,}
\]
\end{CorollaryCMS}
The case-specific estimator $\mathbf{\tilde{H}}_{n}^{\mathtt{CMS}}$ of
$\mathbf{H}^{\mathtt{CMS}}$ admits a counterpart of Lemma \ref{[Lemma] MS},
but for brevity we omit a precise statement.
\subsection{Proof of Corollary \ref{[Corollary] CMS}}
\emph{Condition CRA(i)}. Because $\kappa_{n}$ does not depend on
$\mathbf{\theta,}$ uniform manageability can be established by proceeding as
in the maximum score example.
Also, $\bar{m}_{n}(\mathbf{z})=|\kappa_{n}(\mathbf{w})|$ satisfies
$q_{n}\mathbb{E}[\bar{m}_{n}(\mathbf{z})^{2}]\mathbb{=}b_{n}^{d}O(1/b_{n}
^{d})=O(1).$
Next, using the representations
\[
M_{n}(\mathbf{\theta})=\int_{\mathbb{R}^{d}}\mu(\mathbf{v}b_{n};\mathbf{\theta
})\kappa(\mathbf{v})d\mathbf{v}\qquad\text{and}\qquad M_{0}(\mathbf{\theta
})=\int_{\mathbb{R}^{d}}\mu(\mathbf{0};\mathbf{\theta})\kappa(\mathbf{v}
)d\mathbf{v,}
\]
we have
\[
\sup_{\mathbf{\theta}\in\mathbf{\Theta}}|M_{n}(\mathbf{\theta})-M_{0}
(\mathbf{\theta})|\leq\left\{ \sup_{\mathbf{\theta}\in\mathbf{\Theta
},||\mathbf{w}||\leq b_{n}}|\mu(\mathbf{w};\mathbf{\theta})-\mu(\mathbf{0}
;\mathbf{\theta})|\right\} \int_{\mathbb{R}^{d}}|\kappa(\mathbf{v}
)|d\mathbf{v}=o(1),
\]
where the equality uses uniform (in $\mathbf{\theta}\in\mathbf{\Theta}$)
continuity of $\mu(\mathbf{w};\mathbf{\theta})$ at $\mathbf{w}=\mathbf{0.}$
Finally, well-separatedness follows from compactness of $\mathbf{\Theta},$
continuity of $M_{0}(\mathbf{\theta})=\mu(0;\mathbf{\theta})$ in
$\mathbf{\theta},$ and the fact (shown by \citet*[Lemmas 6 and 7]
{Honore-Kyriazidou_2000_ECMA}) that $\mathbf{\theta}_{0}$ is the unique
maximizer of $M_{0}(\mathbf{\theta}).$\bigskip
\emph{Condition CRA(ii)}. We have
\[
\frac{\partial}{\partial\mathbf{\theta}}M_{n}(\mathbf{\theta})=\int
_{\mathbb{R}^{d}}\dot{\mu}(\mathbf{v}b_{n};\mathbf{\theta})\kappa
(\mathbf{v})d\mathbf{v,}\qquad\frac{\partial}{\partial\mathbf{\theta}}
M_{0}(\mathbf{\theta})=\int_{\mathbb{R}^{d}}\dot{\mu}(\mathbf{0}
;\mathbf{\theta})\kappa(\mathbf{v})d\mathbf{v,}
\]
and
\[
\frac{\partial^{2}}{\partial\mathbf{\theta}\partial\mathbf{\theta}^{\prime}
}M_{n}(\mathbf{\theta})=\int_{\mathbb{R}^{d}}\ddot{\mu}(\mathbf{v}
b_{n};\mathbf{\theta})\kappa(\mathbf{v})d\mathbf{v,}\qquad\frac{\partial^{2}
}{\partial\mathbf{\theta}\partial\mathbf{\theta}^{\prime}}M_{0}(\mathbf{\theta
})=\int_{\mathbb{R}^{d}}\ddot{\mu}(\mathbf{0};\mathbf{\theta})\kappa
(\mathbf{v})d\mathbf{v,}
\]
where, using uniform (in $\mathbf{\theta}\in\mathbf{\Theta}_{0}^{\delta}$)
continuity of $\ddot{\mu}(\mathbf{w};\mathbf{\theta})$ at $\mathbf{w}
=\mathbf{0},$
\[
\sup_{\mathbf{\theta}\in\mathbf{\Theta}_{0}^{\delta}}\left\Vert \frac
{\partial^{2}}{\partial\mathbf{\theta}\partial\mathbf{\theta}^{\prime}}
[M_{n}(\mathbf{\theta})-M_{0}(\mathbf{\theta})]\right\Vert \leq\left\{
\sup_{\mathbf{\theta}\in\mathbf{\Theta}_{0}^{\delta},||\mathbf{w}||\leq b_{n}
}|\ddot{\mu}(\mathbf{w};\mathbf{\theta})-\ddot{\mu}(\mathbf{0};\mathbf{\theta
})|\right\} \int_{\mathbb{R}^{d}}|\kappa(\mathbf{v})|d\mathbf{v}
=o(1)\mathbf{.}
\]
Also, by a standard bias calculation for kernel estimators,
\[
\frac{\partial}{\partial\mathbf{\theta}}M_{n}(\mathbf{\theta}_{0}
)=\int_{\mathbb{R}^{d}}\dot{\mu}(\mathbf{v}b_{n};\mathbf{\theta}_{0}
)\kappa(\mathbf{v})d\mathbf{v}=\dot{\mu}(\mathbf{0};\mathbf{\theta}
_{0})+O(b_{n}^{P}),
\]
where it follows from \citet*{Honore-Kyriazidou_2000_ECMA} that $\dot{\mu
}(\mathbf{0};\mathbf{\theta}_{0})=\partial M_{0}(\mathbf{\theta}_{0}
)/\partial\mathbf{\theta}=\mathbf{0.}$ As a consequence, $r_{n}||\partial
M_{n}(\mathbf{\theta}_{0})/\partial\mathbf{\theta}||=O(\sqrt[3]{nb_{n}^{d+3P}
})=o(1).$
Finally, $\mathbf{H}_{0}=-\ddot{\mu}(\mathbf{0};\mathbf{\theta}_{0}
)=\mathbf{H}^{\mathtt{CMS}}$ is positive definite by assumption.\bigskip
\emph{Condition CRA(iii)}. Because $\kappa_{n}$ does not depend on
$\mathbf{\theta,}$ uniform manageability can be established by proceeding as
in the maximum score example. For this example,
\[
\bar{d}_{n}^{\delta}(\mathbf{z})=\left[ \sup_{\Vert\mathbf{\theta
}-\mathbf{\theta}_{0}\Vert\leq\delta}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(\mathbf{x}_{2}^{\prime}\mathbf{\theta}_{0}<-x_{1}\leq\mathbf{x}_{2}^{\prime
}\mathbf{\theta})+\sup_{\Vert\mathbf{\theta}-\mathbf{\theta}_{0}\Vert
\leq\delta}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(\mathbf{x}_{2}^{\prime}\mathbf{\theta}<-x_{1}\leq\mathbf{x}_{2}^{\prime
}\mathbf{\theta}_{0})\right] |\kappa_{n}(\mathbf{w})|.
\]
By change of variables and using boundedness of $f_{x_{1}|\mathbf{x}
_{2},\mathbf{w}}(x_{1}|\mathbf{x}_{2},\mathbf{w}),$ we have, uniformly in
$\delta,$
\[
b_{n}^{d}\mathbb{E}[\bar{d}_{n}^{\delta}(\mathbf{z})^{2}/\delta]=b_{n}
^{d}O(\mathbb{E}[\Vert\mathbf{x}_{2}\Vert\kappa_{n}(\mathbf{w})^{2}])=O(1).
\]
As a consequence, $q_{n}\sup_{0<\delta^{\prime}\leq\delta}\mathbb{E}[\bar
{d}_{n}^{\delta^{\prime}}(\mathbf{z})^{2}/\delta^{\prime}]=O(1).$\bigskip
\emph{Condition CRA(iv)}. Using $\bar{d}_{n}^{\delta}(\mathbf{z})^{4}
\leq8d_{n}^{\delta}(\mathbf{z})|\kappa_{n}(\mathbf{w})|^{3},$ it follows from
calculations similar to those above that $q_{n}^{3}r_{n}^{-1}\mathbb{E}
[\bar{d}_{n}^{\delta_{n}}(\mathbf{z})^{4}]=O(r_{n}^{-1}\delta_{n})=o(1).$
As in the maximum score example, $\mathcal{C}^{\mathtt{CMS}}(\mathbf{s}
,\mathbf{s})+\mathcal{C}^{\mathtt{CMS}}(\mathbf{t},\mathbf{t})-2\mathcal{C}
^{\mathtt{CMS}}(\mathbf{s},\mathbf{t})>0.$
Finally, the representation
\begin{align*}
& \{m_{n}^{\mathtt{CMS}}(\mathbf{z},\mathbf{\theta}+\delta_{n}\mathbf{s}
)-m_{n}^{\mathtt{CMS}}(\mathbf{z},\mathbf{\theta})\}\{m_{n}^{\mathtt{CMS}
}(\mathbf{z},\mathbf{\theta}+\delta_{n}\mathbf{t})-m_{n}^{\mathtt{CMS}
}(\mathbf{z},\mathbf{\theta})\}\\
& =y^{2}[
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
\left\{ \delta_{n}\min(\mathbf{x}_{2}^{\prime}\mathbf{s},\mathbf{x}
_{2}^{\prime}\mathbf{t)}\geq-x_{1}-\mathbf{x}_{2}^{\prime}\mathbf{\theta
}>0\right\} +
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
\left\{ \delta_{n}\max(\mathbf{x}_{2}^{\prime}\mathbf{s},\mathbf{x}
_{2}^{\prime}\mathbf{t)<}-x_{1}-\mathbf{x}_{2}^{\prime}\mathbf{\theta\leq
}0\right\} ]\kappa_{n}(\mathbf{w})^{2}
\end{align*}
can be used to show that, uniformly in $\mathbf{\theta}\in\mathbf{\Theta}
_{0}^{\delta_{n}},$
\[
\frac{q_{n}}{\delta_{n}}\mathbb{E}[\{m_{n}^{\mathtt{CMS}}(\mathbf{z}
,\mathbf{\theta}+\delta_{n}\mathbf{s})-m_{n}^{\mathtt{CMS}}(\mathbf{z}
,\mathbf{\theta})\}\{m_{n}^{\mathtt{CMS}}(\mathbf{z},\mathbf{\theta}
+\delta_{n}\mathbf{t})-m_{n}^{\mathtt{CMS}}(\mathbf{z},\mathbf{\theta
})\}]=\mathcal{C}^{\mathtt{CMS}}(\mathbf{s},\mathbf{t})+o(1).
\]
\emph{Condition CRA(v)}. The first condition follows from $q_{n}\bar{d}
_{n}^{\delta}(\mathbf{z})\leq\sup_{\mathbf{v\in}\mathbb{R}^{d}}|\kappa
(\mathbf{v})|.$ The second condition follows from the calculation similar to
the covariance kernel calculation.
\section{Example: Empirical Risk Minimization}
In this example, we follow \citet*[Theorem 1]{Mohammadi-vandeGeer_2005_JMLR}
when stating primitive conditions. Let $F$ denote the distribution function of
$x$ and let $P(x)=\mathbb{P}[y=1|x].$
\begin{description}
\item[Condition ERM] The following are satisfied:\newline(i) $P(0)<1/2$ and
$P$ admits a continuous derivative $p$ in a neighborhood of each element of
$\mathbf{\theta}_{0}.$\newline(ii) $F$ is absolutely continuous and its
Lebesgue density $f\ $is continuously differentiable in a neighborhood of each
element of $\mathbf{\theta}_{0}.$\newline(iii) $\mathbf{\theta}_{0}$ is an
interior point of $\boldsymbol{\Theta}.$\newline(iv) $\mathbf{\theta}
_{0}=(\theta_{0,1},\theta_{0,2},\cdots,\theta_{0,d})^{\prime}$ is the unique
minimizer of $\mathbb{P}[h_{\mathbf{\theta}}(x)\neq y]$ and $p(\theta_{0,\ell
})f(\theta_{0,\ell})\neq0$ for $\ell\in\{1,\cdots,d\}.$
\end{description}
\begin{CorollaryERM}
\label{[Corollary] ERM}Suppose Condition ERM is satisfied. Then Condition CRA
is satisfied with $q_{n}=1,$ $\mathbf{H}_{0}=\mathbf{H}^{\mathtt{ERM}},$ and
$\mathcal{C}_{0}=\mathcal{C}^{\mathtt{ERM}},$ where
\[
\mathbf{H}^{\mathtt{ERM}}=2\left(
\begin{array}
[c]{cccc}
p(\theta_{0,1})f(\theta_{0,1}) & 0 & \cdots & 0\\
0 & -p(\theta_{0,2})f(\theta_{0,2}) & \cdots & 0\\
\cdots & \cdots & \cdots & \cdots\\
0 & 0 & \cdots & (-1)^{d+1}p(\theta_{0,d})f(\theta_{0,d})
\end{array}
\right)
\]
and, for $\mathbf{s}=(s_{1},\cdots,s_{d})^{\prime}$ and $\mathbf{t}
=(t_{1},\cdots,t_{d})^{\prime},$
\[
\mathcal{C}^{\mathtt{ERM}}(\mathbf{s},\mathbf{t})=\sum_{\ell=1}^{d}
f(\theta_{0,\ell})\min\{|s_{\ell}|,|t_{\ell}|\}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
\{\operatorname*{sgn}(s_{\ell})=\operatorname*{sgn}(t_{\ell})\}.
\]
\end{CorollaryERM}
A case-specific (plug-in) estimator of $\mathbf{H}^{\mathtt{ERM}}$ is given by
the diagonal matrix $\mathbf{\tilde{H}}_{n}^{\mathtt{ERM}}$ with diagonal
elements
\[
\tilde{H}_{n,\ell\ell}^{\mathtt{ERM}}=(-1)^{\ell+1}2\hat{p}_{n}(\hat{\theta
}_{n,\ell}^{\mathtt{ERM}})\hat{f}_{n}(\hat{\theta}_{n,\ell}^{\mathtt{ERM}
}),\qquad\ell=1,\dots,d,
\]
where $\hat{p}_{n}$ and $\hat{f}_{n}$ are some nonparametric estimators of $p
$ and $f.$ This estimator is consistent whenever its ingredients $\hat{p}_{n}
$ and $\hat{f}_{n}$ are.
\subsection{Proof of Corollary \ref{[Corollary] ERM}}
By Lemma \ref{[Lemma] Bechmark case}, it suffices to verify that Condition
CRA$_{0}$ is satisfied.\medskip
\emph{Condition CRA}$_{0}$\emph{(i)}. Manageability of $\mathcal{M}_{0}$
follows from $\{
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(h_{\mathbf{\theta}}(x)\neq y):\mathbf{\theta}\in\boldsymbol{\Theta}\}$
forming a VC subgraph class. Also, the envelope is bounded by $1.$ Finally,
$\sup_{\mathbf{\theta}\in\boldsymbol{\Theta}\backslash\boldsymbol{\Theta}
_{0}^{\delta}}M_{0}(\mathbf{\theta})<M_{0}(\mathbf{\theta}_{0})$ for every
$\delta>0$ because $\mathbf{\Theta}$ is compact, $M_{0}$ is continuous, and
$\mathbf{\theta}_{0}$ is the unique maximizer of $M_{0}(\mathbf{\theta}
).$\medskip
\emph{Condition CRA}$_{0}$\emph{(ii)}. By assumption, $\mathbf{\theta}_{0}$
belongs to the interior of $\mathbf{\Theta}_{0}.$
\citet*{Mohammadi-vandeGeer_2005_JMLR} show that, for odd $\ell,$
\[
\frac{\partial^{2}}{\partial\theta_{\ell}^{2}}\mathbb{P}(h_{\mathbf{\theta}
}(x)\neq y)=2p(\theta_{\ell})f(\theta_{\ell})+(2P(\theta_{\ell})-1)\frac
{d}{d\theta}f(\theta_{\ell}),
\]
and that a similar formula holds for even $\ell$ as well. In particular,
$M_{0}$ is twice continuous differentiability on $\mathbf{\Theta}_{0}^{\delta
}.$ Finally, positive definiteness of $\mathbf{H}_{0}=\mathbf{H}
^{\mathtt{ERM}}$ is established in \citet*{Mohammadi-vandeGeer_2005_JMLR}
.\medskip
\emph{Condition CRA}$_{0}$\emph{(iii)}. This condition corresponds to the
first part of (vii) in Theorem 7 of \citet*{Mohammadi-vandeGeer_2005_JMLR}
.\medskip
\emph{Condition CRA}$_{0}$\emph{(iv)}. Since $d_{0}^{\delta}(\mathbf{z}
)^{4}=d_{0}^{\delta}(\mathbf{z}),$ $\mathbb{E}[d_{0}^{\delta_{n}}
(\mathbf{z})^{4}]=O(\delta_{n}),$ which implies the first condition. For the
second part, \citet*{Mohammadi-vandeGeer_2005_JMLR} show that
\[
\mathcal{C}_{0}(\mathbf{s},\mathbf{t})=\sum_{\ell=1}^{d}f(\theta_{0,\ell
})[\min\{s_{\ell},t_{\ell}\}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(s_{\ell}>0,t_{\ell}>0)-\max\{s_{\ell},t_{\ell}\}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(s_{\ell}<0,t_{\ell}<0)].
\]
Using the representations
\[
m_{0}(1,x,\mathbf{\theta})=-\sum_{\ell=0}^{\lfloor d/2\rfloor}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
\left( x\in\lbrack\theta_{2\ell},\theta_{2\ell+1})\right) ,\qquad
m_{0}(-1,x,\mathbf{\theta})=-\sum_{\ell=1}^{\lfloor(d+1)/2\rfloor}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
\left( x\in\lbrack\theta_{2\ell-1},\theta_{2\ell})\right) ,
\]
it can be shown that, for $\mathbf{\theta}$ in the interior of
$\boldsymbol{\Theta}$ and for $\delta_{n}$ small enough,
\begin{align*}
& \{m_{0}(\mathbf{z},\mathbf{\theta}+\delta_{n}\mathbf{s})-m_{0}
(\mathbf{z},\mathbf{\theta})\}\{m_{0}(\mathbf{z},\mathbf{\theta}+\delta
_{n}\mathbf{t})-m_{0}(\mathbf{z},\mathbf{\theta})\}\\
& =\sum_{\ell=1}^{d}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(x\in\lbrack\theta_{\ell}+\delta_{n}\max\{s_{\ell},t_{\ell}\},\theta_{\ell}))
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(s_{\ell}<0,t_{\ell}<0)+
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(x\in\lbrack\theta_{\ell},\theta_{\ell}+\delta_{n}\min\{s_{\ell},t_{\ell}\}))
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(s_{\ell}>0,t_{\ell}>0).
\end{align*}
As a consequence,
\begin{align*}
& \frac{1}{\delta_{n}}\mathbb{E}[\{m_{0}(\mathbf{z},\mathbf{\theta}+\delta
_{n}\mathbf{s})-m_{0}(\mathbf{z},\mathbf{\theta})\}\{m_{0}(\mathbf{z}
,\mathbf{\theta}+\delta_{n}\mathbf{t})-m_{0}(\mathbf{z},\mathbf{\theta})\}]\\
& =\sum_{\ell=1}^{d}f(\theta_{\ell})[-\max\{s_{\ell},t_{\ell}\}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(s_{\ell}<0,t_{\ell}<0)+\min\{s_{\ell},t_{\ell}\}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(s_{\ell}>0,t_{\ell}>0)]+o(1)\\
& =\sum_{\ell=1}^{d}f(\theta_{0,\ell})[-\max\{s_{\ell},t_{\ell}\}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(s_{\ell}<0,t_{\ell}<0)+\min\{s_{\ell},t_{\ell}\}
{\rm 1\hspace*{-0.4ex}\rule{0.1ex}{1.52ex}\hspace*{0.2ex}}
(s_{\ell}>0,t_{\ell}>0)]+o(1)
\end{align*}
uniformly in $\mathbf{\theta}\in\mathbf{\Theta}_{0}^{\delta_{n}}.$\medskip
\emph{Condition CRA}$_{0}$\emph{(v)}. The condition in display is identical to
the second part of (vii) in Theorem 7 of \citet*{Mohammadi-vandeGeer_2005_JMLR}.
The second assumption corresponds to (vi) in Theorem 7 of
\citet*{Mohammadi-vandeGeer_2005_JMLR}.
\bibliographystyle{econometrica}
\bibliography{BootstrapCubeRoot}