EconBase
← Back to paper

Estimation and Inference in Boundary Discontinuity Designs: Distance-Based Methods

The exact contents of citations.db main_text.text for this paper — one flattened LaTeX string, title through conclusion, appendix excluded, unmodified except for removing email addresses. This is what our citation measures are computed over.

150,930 characters

Estimation and Inference in Boundary Discontinuity Designs: Distance-Based Methods Supplemental Appendix


\title{Estimation and Inference in Boundary Discontinuity Designs: Distance-Based Methods\\ Supplemental Appendix\bigskip}
\author{Matias D. Cattaneo\thanks{Department of Operations Research and Financial Engineering, Princeton University.} \and
	    Rocio Titiunik\thanks{Department of Politics, Princeton University.} \and
	    Ruiqi (Rae) Yu\thanks{Department of Operations Research and Financial Engineering, Princeton University.}
	    }
\maketitle


\begin{abstract}
    This supplemental appendix presents more general theoretical results encompassing those reported in the paper, their theoretical proofs, and other technical results. In particular, it presents a new strong approximation result for multiplicative-separable empirical processes leveraging and extending ideas from \cite{Cattaneo-Yu_2025_AOS}.
\end{abstract}


\thispagestyle{empty}
\clearpage

\onehalfspacing
\setcounter{page}{1}
\pagestyle{plain}

\setcounter{tocdepth}{2}
\setcounter{secnumdepth}{4}
\tableofcontents

\clearpage


\section{Setup}\label{section: Setup}

This supplemental appendix considers a generalized version of the problems studied in the main paper. Specifically, the underlying bivariate location variable $\mathbf{X}_i$ is $d$-dimensional ($d\geq1$) with support $\mathcal{X}\subseteq\mathbb{R}^d$, and the boundary region $\mathcal{B}$ is a low dimensional manifold with ``effective dimension'' $d-1$. The results in the paper correspond to $d = 2$, that is, $\mathbf{X}_i$ is bivariate and $\mathcal{B}$ is a one-dimensional (boundary assignment) curve.

Assumption 1 in the paper generalizes as follows.

\begin{assumption}[Data Generating Process]\label{sa-assump: DGP}
    Let $t\in\{0,1\}$.
    \begin{enumerate}[label=\normalfont(\roman*),noitemsep,leftmargin=*]

    \item $(Y_1(t), \mathbf{X}_1^\top)^\top,\ldots, (Y_n(t), \mathbf{X}_n^\top)^\top$ are independent and identically distributed random vectors with $\mathcal{X} = \prod_{l = 1}^d [a_l, b_l]$ for $-\infty < a_l < b_l < \infty$ for $l = 1,\cdots,d$.

    \item The distribution of $\mathbf{X}_i$ has a Lebesgue density $f_X(\mathbf{x})$ that is continuous and bounded away from zero on $\mathcal{X}$.

    \item $\mu_t(\mathbf{x}) = \mathbb{E}[Y_i(t)| \mathbf{X}_i = \mathbf{x}]$ is $(p+1)$-times continuously differentiable on $\mathcal{X}$.

    \item $\sigma^2_t(\mathbf{x}) = \mathbb{V}[Y_i(t)|\mathbf{X}_i = \mathbf{x}]$ is bounded away from zero and continuous on $\mathcal{X}$.

    \item $\sup_{\mathbf{x} \in \mathcal{X}}\mathbb{E}[|Y_i(t)|^{2+v}|\mathbf{X}_i = \mathbf{x}] < \infty$ for some $v \geq 2$.
    \end{enumerate}
\end{assumption}

The support $\mathcal{X}$ is partitioned into two (assignment) areas, $\mathcal{A}_0\subset\mathbb{R}^d$ and $\mathcal{A}_1\subset\mathbb{R}^d$, representing the control and treatment regions, respectively. Thus, $\mathcal{X} = \mathcal{A}_0 \cup \mathcal{A}_1$ with $\mathcal{A}_0$ and $\mathcal{A}_1$ disjoint regions in $\mathbb{R}^d$. The observed outcome is $Y_i = \mathds{1}(\mathbf{X}_i \in \mathcal{A}_0) Y_i(0) + \mathds{1}(\mathbf{X}_i \in \mathcal{A}_1) Y_i(1)$, and $\mathcal{B} = \mathtt{bd}(\mathcal{A}_0) \cap  \mathtt{bd}(\mathcal{A}_1)$ is the boundary determined by the assignment regions, where $\mathtt{bd}(\mathcal{A}_t)$ denotes the topological boundary of $\mathcal{A}_t$.

The \emph{conditional treatment effect curve at the boundary} is
\begin{align*}
    \tau(\mathbf{x}) = \mathbb{E}[Y_i(1) - Y_i(0)|\mathbf{X}_i = \mathbf{x}], \qquad \mathbf{x}\in\mathcal{B}.
\end{align*}

The univariate distance score induced by the bivariate location variable is
\begin{align*}
    D_i(\mathbf{x})
    = [ \mathds{1}(\mathbf{X}_i \in \mathcal{A}_1) - \mathds{1}(\mathbf{X}_i \in \mathcal{A}_0)] \mathcal{d}(\mathbf{X}_i,\mathbf{x}),
    \qquad \mathbf{x} \in \mathcal{B},
\end{align*}
where $\mathcal{d}(\cdot,\cdot)$ denotes a distance function. The distance-based treatment effect estimator process along the boundary based is $(\tau(\mathbf{x}): \mathbf{x} \in \mathcal{B})$ is
\begin{align*}
    \Big(\widehat{\vartheta}(\mathbf{x})
    = \widehat{\theta}_{1,\mathbf{x}}(0) - \widehat{\theta}_{0,\mathbf{x}}(0) : \mathbf{x}\in\mathcal{B} \Big),
\end{align*}
where, for $t \in \{0,1\}$,
\begin{align*}
    \widehat{\theta}_{t,\mathbf{x}}(0) = \mathbf{e}_1^{\top} \widehat{\boldsymbol{\gamma}}_t(\mathbf{x}),
    \qquad
    \widehat{\boldsymbol{\gamma}}_{t}(\mathbf{x}) = \operatorname*{arg\,min}_{\boldsymbol{\gamma} \in \mathbb{R}^{p+1}} \mathbb{E}_n \Big[ \big(Y_i - \mathbf{r}_p(D_i(\mathbf{x}))^{\top} \boldsymbol{\gamma} \big)^2 K_h(D_i(\mathbf{x})) \mathds{1}_{\mathcal{I}_t}(D_i(\mathbf{x})) \Big],
\end{align*}
$\mathbf{r}_p(u)=(1,u,\cdots,u^p)^\top$ and $K_h(u)=K(u/h)/h^2$ with $K(\cdot)$ a univariate kernel and $h$ a bandwidth parameter, and $\mathcal{I}_0 = (-\infty,0)$ and $\mathcal{I}_1 = [0,\infty)$. More generally, the least squares projection is
\begin{align*}
    \widehat{\theta}_{t,\mathbf{x}}(D_i(\mathbf{x})) = \mathbf{r}_p(D_i(\mathbf{x}))^{\top}\widehat{\boldsymbol{\gamma}}_{t}(\mathbf{x}),
    \qquad t\in\{0,1\}, \quad \mathbf{x} \in \mathcal{B}.
\end{align*}

We impose the following assumptions on the kernel function, distance function, and assignment boundary manifold. Let
\begin{align*}
    \boldsymbol{\Psi}_{t,\mathbf{x}}
    = \mathbb{E} \left[\mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big) \mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big)^{\top} K_h(D_i(\mathbf{x})) \mathds{1}(D_i(\mathbf{x}) \in \mathcal{I}_t)\right],
\end{align*}
for $t\in\{0,1\}$.

\begin{assumption}[Kernel, Distance, and Boundary]\label{sa-assump: Kernel, Distance and Boundary}
Let $t \in \{0,1\}$.
\vspace{-.1in}
\begin{enumerate}[label=\normalfont(\roman*),noitemsep,leftmargin=*]
    \item $\mathcal{B}$ is compact $(d-1)$-rectifiable, with $\mathfrak{H}^{d-1}(\mathcal{B})$ positive and finite.
    \item $\mathcal{d}: \mathbb{R}^d \times \mathbb{R}^d \to \mathbb{R}_+$ is a metric on $\mathbb{R}^d$ equivalent to the Euclidean distance, that is, there exists positive constants $C_u$ and $C_l$ such that $C_l \left\lVert\mathbf{x} - \mathbf{x}'\right\rVert \leq \mathcal{d}(\mathbf{x},\mathbf{x}') \leq C_u\left\lVert\mathbf{x} - \mathbf{x}'\right\rVert$ for all $\mathbf{x},\mathbf{x}' \in \mathcal{X}$.
    \item $K: \mathbb{R} \to [0,\infty)$ is compact supported and Lipschitz continuous, or $K(u)=\mathds{1}(u \in [-1,1])$.
    \item $\liminf_{h\downarrow0}\inf_{\mathbf{x} \in \mathcal{B}} \lambda_{\min}(\boldsymbol{\Psi}_{t,\mathbf{x}}) \gtrsim 1$.
\end{enumerate}
\end{assumption}

For each $t \in \{0,1\}$, the induced conditional expectation based on univariate distance is
\begin{align*}
    \theta_{t,\mathbf{x}}(r)
    = \mathbb{E}[Y_i| D_i(\mathbf{x}) = r ]
    = \mathbb{E}[Y_i|\mathcal{d}(\mathbf{X}_i,\mathbf{x}) = |r|, \mathbf{X}_i \in \mathcal{A}_t],
    \qquad r \in \mathcal{I}_t, \quad \mathbf{x} \in \mathcal{B}.
\end{align*}
More rigorously, for each $t \in \{0,1\}$, and letting $S_{t,\mathbf{x}}(r) = \{\mathbf{v} \in \mathcal{X}: \mathcal{d}(\mathbf{v},\mathbf{x}) = r, \mathbf{v} \in \mathcal{A}_t\}$ for $r \geq 0$ and $\mathbf{x} \in \mathcal{B}$,
\begin{align*}
    \theta_{t,\mathbf{x}}(r)
    = \frac{\int_{S_{t,\mathbf{x}}(|r|)} \mu_t(\mathbf{v}) f_X(\mathbf{v}) \mathfrak{H}^{d-1}(d \mathbf{v})}
           {\int_{S_{t,\mathbf{x}}(|r|)} f_X(\mathbf{v}) \mathfrak{H}^{d-1}(d \mathbf{v})},
\end{align*}
for $|r| > 0, \mathbf{x} \in \mathcal{B}, t \in \{0,1\}$, and therefore (under our assumptions)
\begin{align*}
    \theta_{t,\mathbf{x}}(0)
    = \lim_{r \to 0}\mathbb{E}[Y_i|\mathcal{d}(\mathbf{X}_i,\mathbf{x}) = |r|, \mathbf{X}_i \in \mathcal{A}_t]
    = \lim_{r \to 0}\frac{\int_{S_{t,\mathbf{x}}(|r|)} \mu_t(\mathbf{v}) f_X(\mathbf{v}) \mathfrak{H}^{d-1}(d \mathbf{v})}
                         {\int_{S_{t,\mathbf{x}}(|r|)} f_X(\mathbf{v}) \mathfrak{H}^{d-1}(d \mathbf{v})}.
\end{align*}
Thus, the population limit based on the induced conditional expectations is $\theta_{\mathbf{x}}(0) = \theta_{1,\mathbf{x}}(0) - \theta_{0,\mathbf{x}}(0)$. Theorem \ref{sa-thm: Identification} shows that $\theta_{\mathbf{x}}(0) = \tau(\mathbf{x})$ under Assumptions \ref{sa-assump: DGP} and \ref{sa-assump: Kernel, Distance and Boundary}.

The best mean square approximation is
\begin{align*}
    \theta_{t,\mathbf{x}}^{\ast}(D_i(\mathbf{x})) =  \mathbf{r}_p(D_i(\mathbf{x}))^{\top}\boldsymbol{\gamma}^{\ast}_t(\mathbf{x}),
\end{align*}
where
\begin{align*}
     \boldsymbol{\gamma}^{\ast}_{t}(\mathbf{x}) = \operatorname*{arg\,min}_{\boldsymbol{\gamma} \in \mathbb{R}^{p+1}} \mathbb{E} \Big[ \left(Y_i - \mathbf{r}_p(D_i(\mathbf{x}))^{\top} \boldsymbol{\gamma} \right)^2 K_h(D_i(\mathbf{x})) \mathds{1}(D_i(\mathbf{x}) \in \mathcal{I}_t) \Big],
\end{align*}
and uniqueness will follow from the results below. The estimation error decomposes into \emph{linear error}, \emph{approximation error}, and \emph{non-linear error}: for all $t\in\{0,1\}$ and $\mathbf{x} \in \mathcal{B}$,
\begin{align}\label{sa-eq: distance error decomposition}
    \widehat{\theta}_{t,\mathbf{x}}(0) - \theta_{t,\mathbf{x}}(0)
    & = \mathbf{e}_1^{\top}  \widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}^{-1} \mathbb{E}_n \Big[\mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big) K_h(D_i(\mathbf{x})) Y_i\Big] - \theta_{t,\mathbf{x}}(0) \nonumber \\
    & = \mathbf{e}_1^{\top}  \widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}^{-1} \mathbb{E}_n \Big[\mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big) K_h(D_i(\mathbf{x})) (Y_i - \theta_{t,\mathbf{x}}^{\ast}(D_i(\mathbf{x})))\Big] + \theta_{t,\mathbf{x}}^{\ast}(0)  - \theta_{t,\mathbf{x}}(0) \nonumber \\
    & = \underbrace{\theta_{t,\mathbf{x}}^{\ast}(0)  - \theta_{t,\mathbf{x}}(0)}_{\text{approximation error}}
      + \underbrace{\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{O}_{t,\mathbf{x}}}_{\text{linear error}}
      + \underbrace{\mathbf{e}_1^{\top} (\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}^{-1} - \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1}) \mathbf{O}_{t,\mathbf{x}}}_{\text{non-linear error}},
\end{align}
where
\begin{align*}
    \mathbf{O}_{t,\mathbf{x}}
    = \mathbb{E}_n \left[\mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big) K_h(D_i(\mathbf{x})) (Y_i - \theta_{t,\mathbf{x}}^{\ast}(D_i(\mathbf{x})))\mathds{1}(D_i(\mathbf{x}) \in \mathcal{I}_t)\right],
\end{align*}
\begin{align*}
    &\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}
    = \mathbb{E}_n \Big[\mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big) \mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big)^{\top} K_h(D_i(\mathbf{x})) \mathds{1}(D_i(\mathbf{x}) \in \mathcal{I}_t)\Big],
\end{align*}
and the misspecification bias is
\begin{align*}
    \mathfrak{B}_{t}(\mathbf{x}) = \theta_{t,\mathbf{x}}^{\ast}(0) - \theta_{t,\mathbf{x}}(0).
\end{align*}

Finally, we define the following for quantities for future analysis: for $t \in \{0,1\}$, $\mathbf{x}_1, \mathbf{x}_2 \in \mathcal{B}$,
\begin{align*}
    \widehat{\boldsymbol{\Upsilon}}_{t,\mathbf{x}_1,\mathbf{x}_2}
    &= h^d \mathbb{E}_n \Big[\mathbf{r}_p \Big(\frac{D_i(\mathbf{x}_1)}{h}\Big) \mathbf{r}_p \Big( \frac{D_i(\mathbf{x}_2)}{h}\Big)^{\top}
                   K_h(D_i(\mathbf{x}_1)) K_h(D_i(\mathbf{x}_2))\\
    & \hspace{1in} (Y_i - \widehat{\theta}_{t,\mathbf{x}_1}(D_i(\mathbf{x}_1)))(Y_i - \widehat{\theta}_{t,\mathbf{x}_2}(D_i(\mathbf{x}_2))) \mathds{1}_{\mathcal{I}_t}(D_i(\mathbf{x}_1))\bigg], \\
    \boldsymbol{\Upsilon}_{t, \mathbf{x}_1,\mathbf{x}_2}
    &= h^d \mathbb{E}  \Big[\mathbf{r}_p \Big(\frac{D_i(\mathbf{x}_1)}{h}\Big) \mathbf{r}_p \Big( \frac{D_i(\mathbf{x}_2)}{h}\Big)^{\top}
                   K_h(D_i(\mathbf{x}_1)) K_h(D_i(\mathbf{x}_2))\\
    & \hspace{1in} (Y_i - \theta^*_{t,\mathbf{x}_1}(D_i(\mathbf{x}_1)))(Y_i - \theta^*_{t,\mathbf{x}_2}(D_i(\mathbf{x}_2))),
\end{align*}
\begin{align*}
    \widehat{\Xi}_{\mathbf{x}_1,\mathbf{x}_2} = \widehat{\Xi}_{0,\mathbf{x}_1,\mathbf{x}_2} + \widehat{\Xi}_{1,\mathbf{x}_1,\mathbf{x}_2},
    \qquad
    \widehat{\Xi}_{t,\mathbf{x}_1,\mathbf{x}_2}
    = \frac{1}{n h^d} \mathbf{e}_1^{\top} \widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}_1}^{-1} \widehat{\boldsymbol{\Upsilon}}_{t,\mathbf{x}_1,\mathbf{x}_2} \widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}_2}^{-1} \mathbf{e}_1
\end{align*}
and
\begin{align*}
    \Xi_{\mathbf{x}_1,\mathbf{x}_2} = \Xi_{0,\mathbf{x}_1,\mathbf{x}_2} + \Xi_{1,\mathbf{x}_1,\mathbf{x}_2}.
    \qquad
    \Xi_{t,\mathbf{x}_1,\mathbf{x}_2} = \frac{1}{n h^d} \mathbf{e}_{1}^{\top}  \boldsymbol{\Psi}_{t,\mathbf{x}_1}^{-1} \boldsymbol{\Upsilon}_{t,\mathbf{x}_1,\mathbf{x}_2} \boldsymbol{\Psi}_{t,\mathbf{x}_2}^{-1} \mathbf{e}_{1}.
\end{align*}
In particular, $\widehat{\Xi}_{\mathbf{x}} = \widehat{\Xi}_{\mathbf{x},\mathbf{x}}$, $\Xi_{\mathbf{x}} = \Xi_{\mathbf{x},\mathbf{x}}$, $\mathfrak{B}(\mathbf{x}) = \mathfrak{B}_{1}(\mathbf{x}) - \mathfrak{B}_{0}(\mathbf{x})$, etc.


\subsection{Notation and Definitions}\label{sa-sec: notations and defns}

For textbook references on empirical process, see \cite{van-der-Vaart-Wellner_1996_Book}, \cite{dudley2014uniform}, and \cite{Gine-Nickl_2016_Book}. For textbook reference on geometric measure theory, see \cite{simon1984lectures}, \cite{federer2014geometric}, and \cite{folland2002advanced}.

\begin{enumerate}[label=(\roman*)]
    \item \textit{Multi-index Notations}. For a multi-index $\mathbf{u} = (u_1, \ldots, u_d) \in \mathbb{N}^d$, denote $|\mathbf{u}| = \sum_{i = 1}^d u_d$, $\mathbf{u}! = \Pi_{i = 1}^d u_d$. Denote $\mathbf{r}_p(\mathbf{u}) = (1, u_1, \ldots, u_d, u_1^2, \ldots, u_d^2, \ldots, u_1^p, \ldots, u_d^p)$, that is, all monomials $u_1^{\alpha_1} \cdots u_d^{\alpha_d}$ such that $\alpha_i \in \mathbb{N}$ and $\sum_{i = 1}^d \alpha_i \leq p$. Define $\mathbf{e}_{1 + \boldsymbol{\nu}}$ to be the $p_d = \frac{(d + p)!}{d! p!}$-dimensional vector such that $\mathbf{e}_{1 + \boldsymbol{\nu}}^{\top} \mathbf{r}_p(\mathbf{u}) = \mathbf{u}^{\boldsymbol{\nu}}$ for all $\mathbf{u} \in \mathbb{R}^d$.

    \item \textit{Norms}. For a vector $\mathbf{v} \in \mathbb{R}^k$, $\left\lVert\mathbf{v}\right\rVert = (\sum_{i = 1}^k \mathbf{v}_i^2)^{1/2}$, $\lVert \mathbf{v} \rVert_{\infty} = \max_{1 \leq i \leq k}|\mathbf{v}_i|$. For a matrix $A \in \mathbb{R}^{m \times n}$, $\left\lVertA\right\rVert_p = \sup_{\left\lVert\mathbf{x}\right\rVert_p = 1} \left\lVertA\mathbf{x}\right\rVert_p$, $p \in \mathbb{N} \cup \{\infty\}$, and $\lambda_{\min}(A)$ denotes its minimum eigenvalue. For a function $f$ on a metric space $(S, d)$, $\lVert f \rVert_{\infty} = \sup_{\mathbf{x} \in \mathcal{X}} |f(\mathbf{x})|$. For a probability measure $Q$ on $(\mathcal{S}, \mathscr{S})$ and $p \geq 1$, define $\left\lVertf\right\rVert_{Q,p} = (\int_{\mathcal{S}} |f|^p d Q)^{1/p}$, and $Q(f) = \int f d Q$. For a set $E \subseteq \mathbb{R}^d$, denote by $\mathfrak{m}(E)$ the Lebesgue measure of $E$.

    \item \textit{Empirical Process}. We use standard empirical process notations: $\mathbb{E}_n[g(\mathbf{v}_i)] = \frac{1}{n} \sum_{i = 1}^n g(\mathbf{v}_i)$ and $\mathbb{G}_n[g(\mathbf{v}_i)] = \frac{1}{\sqrt{n}} \sum_{i = 1}^n (g(\mathbf{v}_i) - \mathbb{E}[g(\mathbf{v}_i)])$. Let $(\mathcal{S},d)$ be a semi-metric space. The covering number $N(\mathcal{S}, d, \varepsilon)$ is the minimal number of balls $B_s(\varepsilon) =\{t: d(t,s) < \varepsilon\}$ needed to cover $\mathcal{S}$. A \emph{$\mathbb{P}$-Brownian bridge} is a mean-zero Gaussian random function $W_n(f), f \in L_2(\mathcal{X}, \mathbb{P})$ with the covariance $\mathbb{E}[W_{\mathbb{P}}(f)W_{\mathbb{P}}(g)] = \mathbb{P}(fg) - \mathbb{P}(f)\mathbb{P}(g)$, for $f,g \in L_2(\mathcal{X},\mathbb{P})$. A class $\mathcal{F} \subseteq L_2(\mathcal{X}, \mathbb{P})$ is \emph{$\mathbb{P}$-pregaussian} if there is a version of $\mathbb{P}$-Brownian bridge $W_{\mathbb{P}}$ such that $W_{\mathbb{P}} \in C(\mathcal{F}; \rho_{\mathbb{P}})$ almost surely, where $\rho_{\mathbb{P}}$ is the semi-metric on $L_2(\mathcal{X},\mathbb{P})$ is defined by $\rho_{\mathbb{P}}(f, g) = (\|f - g\|_{\mathbb{P},2}^2 - (\int f \, d\mathbb{P} - \int g \, d\mathbb{P})^2)^{1/2}$, for $f, g \in L_2(\mathcal{X},\mathbb{P})$.

    \item \textit{Geometric Measure Theory}. For a set $E \subseteq \mathcal{X}$, the \emph{De Giorgi perimeter of $E$ related to $\mathcal{X}$} is $\mathcal{L}(E) = \mathtt{TV}_{\{\mathds{1}_{E}\},\mathcal{X}}$. For $d \in \mathbb{N}$ and $0 \leq m \leq d$, the $m$-dimensional Hausdorff (outer) measure is given by $\mathfrak{H}^m(A) = \lim_{\delta \downarrow 0}\mathfrak{H}^m_{\delta}(A)$, $A \subseteq \mathbb{R}^d$, where for each $\delta > 0$, $\mathfrak{H}^m_{\delta}(A)$ is defined by taking $\mathfrak{H}^m_{\delta}(\emptyset) = 0$, and for any non-empty $A \subseteq \mathbb{R}^d$, $\mathfrak{H}^m_{\delta}(A) = \frac{\pi^{m/2}}{\Gamma(m/2+1)} \inf \sum_{j = 1}^{\infty} (\operatorname{diam}(C_j)/2)^m$, and the infimum is taken over all countable collections $C_1, C_2, \cdots$ of subsets of $\mathbb{R}^d$ such that $\operatorname{diam}(C_j) < \delta$ and $A \subseteq \cup_{j = 1}^{\infty}C_j$. Integration against $\mathfrak{H}^m$ is defined via Carathéodory's Theorem following the classical measure-theoretic literature. The Hausdorff dimension $\dim_{\mathfrak{H}}(A)$ of $A$ is defined by $\dim_{\mathfrak{H}}(A) = \inf\{t \geq 0: \mathfrak{H}^t(A) = 0\}$. A set $A \subseteq \mathbb{R}^d$ is said to be $k$-rectifiable if $A$ is of Hausdorff dimension $k$, and there exist a countable collection $\{f_i\}$ of continuously differentiable maps $f_i: \mathbb{R}^k \to \mathbb{R}^d$ such that $\mathfrak{H}^{k}(E \setminus \cup_{i = 0}^{\infty} f_i(\mathbb{R}^k)) = 0$. $B$ is a \emph{rectifiable curve} if there exists a Lipschitz continuous function $\gamma:[0,1] \to \mathbb{R}$ such that $B=\gamma([0,1])$. We define the curve length function of $B$ to be $\mathfrak{L}({B}) = \sup_{\pi \in \Pi} s(\pi, \gamma)$, where $\Pi = \left\{(t_0, t_1, \ldots, t_N): N \in \mathbb{N}, 0 \leq t_0 < t_1 < \ldots \leq t_N \leq 1\right\}$ and $s(\pi,\gamma) = \sum_{i = 0}^{N}\left\lVert\gamma(t_{i}) - \gamma(t_{i+1})\right\rVert_2$ for $\pi = (t_0, t_1, \ldots, t_N)$.

    \item \textit{Bounds and Asymptotics}. For reals sequences $|a_n| = o(|b_n|)$ if $\limsup \frac{a_n}{b_n} = 0$,  $|a_n| \lesssim |b_n|$ if there exists some constant $C$ and $N > 0$ such that $n > N$ implies $|a_n| \leq C |b_n|$. For sequences of random variables $a_n = o_{\mathbf{b}{P}}(b_n)$ if $\operatorname{plim}_{n \to \infty}\frac{a_n}{b_n} = 0, |a_n| \lesssim_{\mathbb{P}} |b_n|$ if $\limsup_{M \to \infty} \limsup_{n \to \infty} \mathbb{P}[|\frac{a_n}{b_n}| \geq M] = 0$.

    \item \textit{Distributions and Statistical Distances}. For $\boldsymbol{\mu} \in \mathbb{R}^k$ and $\boldsymbol{\Sigma}$ a $k \times k$ positive definite matrix, $\mathsf{Normal}(\boldsymbol{\mu}, \boldsymbol{\Sigma})$ denotes the Gaussian distribution with mean $\boldsymbol{\mu}$ and covariance $\boldsymbol{\Sigma}$. For $-\infty < a < b < \infty$, $\mathsf{Uniform}([a,b])$ denotes the uniform distribution on $[a,b]$. $\mathsf{Bernoulli}(p)$ denotes the Bernoulli distribution with success probability $p$. $\Phi(\cdot)$ denotes the standard Gaussian cumulative distribution function. For two distributions $P$ and $Q$, $d_{\operatorname{KL}}(P,Q)$ denotes the KL-distance between $P$ and $Q$, and $d_{\chi^2}(P,Q)$ denotes the $\chi^2$ distance between $P$ and $Q$.
\end{enumerate}

\subsection{Mapping between Main Paper and Supplement}

The results in the main paper are special cases of the results in this supplemental appendix as follows.

\begin{itemize}
    \item Theorem 1 in the paper corresponds to Theorem \ref{sa-thm: Identification} with $d=2$.
    \item Theorem 2 in the paper is proven in Section \ref{sa-sec: Proof of Theorem 2}.
    \item Theorem 3 in the paper is proven in Section \ref{sa-sec: Proof of Theorem 3}.
    \item Theorem 4(i) in the paper corresponds in Theorem \ref{sa-thm: Pointwise Convergence Rate} with $d=2$.
    \item Theorem 4(ii) in the paper corresponds in Theorem \ref{sa-thm: Uniform Convergence Rate} with $d=2$.
    \item Theorem 5(i) in the paper corresponds in Theorem \ref{sa-thm: Confidence Intervals} with $d=2$.
    \item Theorem 5(ii) in the paper corresponds in Theorem \ref{sa-thm: Confidence Bands} with $d=2$.
    \item Theorem 6 in the paper is proven in Section \ref{sa-sec: Proof of Theorem 6}.
\end{itemize}


\section{Preliminary Lemmas}

Recall that $t \in \{0,1\}$.

The following lemma gives a sufficient condition for Assumption~\ref{sa-assump: Kernel, Distance and Boundary}.

\begin{lem}[Gram Invertibility]\label{sa-lem: Gram invert}
    Suppose the following conditions hold:
    \begin{enumerate}
        \item  Assumptions \ref{sa-assump: DGP}(i)(ii) and Assumption~\ref{sa-assump: Kernel, Distance and Boundary} (iii) hold.
        \item $\mathcal{d}(\cdot,\cdot)$ is the Euclidean distance.
        \item There exists a set $U \subseteq \mathbb{R}^d$, such that $K(\lVert \mathbf{u} \rVert) \geq \kappa > 0$ for all $\mathbf{u} \in U$, $\lambda_{\min} (\int_U \mathbf{r}_p(\lVert \mathbf{z} \rVert) \mathbf{r}_p(\lVert \mathbf{z} \rVert)^{\top} d \mathbf{z}) > 0$, and $\liminf_{h \downarrow 0}\inf_{\mathbf{x} \in \mathcal{B}} \int_{U} K(\lVert \mathbf{u} \rVert) \mathds{1}(\mathbf{x} + h \mathbf{u} \in \mathcal{A}_t) d \mathbf{u} \gtrsim 1$.
    \end{enumerate}
    Then Assumption~\ref{sa-assump: Kernel, Distance and Boundary} (iv) holds.
\end{lem}

\begin{lem}[Gram]\label{sa-lem: Gram}
    Suppose Assumptions \ref{sa-assump: DGP}(i)(ii) and \ref{sa-assump: Kernel, Distance and Boundary} hold. If $\frac{n h^d}{\log (1/h)} \to \infty$, then
    \begin{align*}
        \sup_{\mathbf{x} \in \mathcal{B}} \big\|\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}} - \boldsymbol{\Psi}_{t,\mathbf{x}}\big\|
        & \lesssim_{\mathbb{P}} \sqrt{\frac{\log (1/h)}{n h^d}},
        \qquad
        1 \lesssim_{\mathbb{P}} \inf_{\mathbf{x} \in \mathcal{B}} \big\|\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}\big\| \leq \sup_{\mathbf{x} \in \mathcal{B}} \big\|\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}\big\| \lesssim_{\mathbb{P}} 1, \\
        \sup_{\mathbf{x} \in \mathcal{B}} \big\|\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}^{-1} - \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1}\big\|
        & \lesssim_{\mathbb{P}} \sqrt{\frac{\log (1/h)}{n h^d}}.
    \end{align*}
\end{lem}

\begin{lem}[Stochastic Linear Approximation]\label{sa-lem: Stochastic Linear Approximation}
    Suppose Assumptions \ref{sa-assump: DGP}(i)(ii)(iii)(v) and \ref{sa-assump: Kernel, Distance and Boundary} hold. If $\frac{n h^d}{\log (1/h)} \to \infty$, then
    \begin{align*}
        \sup_{\mathbf{x} \in \mathcal{B}} \big\|\mathbf{O}_{t,\mathbf{x}}\big\|
        & \lesssim_{\mathbb{P}} \sqrt{\frac{\log (1/h)}{n h^d}} + \frac{\log (1/h)}{n^{\frac{1+v}{2+v}}h^d},\\
        \sup_{\mathbf{x} \in \mathcal{B}} \big|\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{O}_{t,\mathbf{x}}\big|
        & \lesssim_{\mathbb{P}} \sqrt{\frac{\log (1/h)}{n h^d}} + \frac{\log (1/h)}{n^{\frac{1+v}{2+v}}h^d}, \\
        \sup_{\mathbf{x} \in \mathcal{B}} \big|\mathbf{e}_1^{\top} (\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}^{-1} - \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1}) \mathbf{O}_{t,\mathbf{x}}\big|
        & \lesssim_{\mathbb{P}} \sqrt{\frac{\log (1/h)}{n h^d}} \bigg(\sqrt{\frac{\log (1/h)}{n h^d}} + \frac{\log (1/h)}{n^{\frac{1+v}{2+v}}h^d} \bigg).
    \end{align*}
 \end{lem}

\begin{lem}[Covariance]\label{sa-lem: Covariance}
    Suppose Assumptions \ref{sa-assump: DGP} and \ref{sa-assump: Kernel, Distance and Boundary} hold. If $\frac{n h^d}{\log (1/h)} \to \infty$, then
    \begin{align*}
        \sup_{\mathbf{x}_1,\mathbf{x}_2 \in \mathcal{B}}
        \big\|\widehat{\boldsymbol{\Upsilon}}_{t, \mathbf{x}_1,\mathbf{x}_2} - \boldsymbol{\Upsilon}_{t, \mathbf{x}_1,\mathbf{x}_2}\big\|
        &\lesssim_{\mathbb{P}} \sqrt{\frac{\log(1/h)}{n h^d}} + \frac{\log(1/h)}{n^{\frac{v}{2+v}}h^d},\\
        \sup_{\mathbf{x}_1,\mathbf{x}_2 \in \mathcal{B}} n h^d \big|\widehat{\Xi}_{t, \mathbf{x}_1,\mathbf{x}_2} - \Xi_{t, \mathbf{x}_1,\mathbf{x}_2}\big|
        &\lesssim_{\mathbb{P}} \sqrt{\frac{\log(1/h)}{n h^d}} + \frac{\log(1/h)}{n^{\frac{v}{2+v}}h^d}.
    \end{align*}

    If, in addition, $\frac{n^{\frac{v}{2+v}} h^d}{\log(1/h)} \to \infty$, then
    \begin{align*}
        \inf_{\mathbf{x} \in \mathcal{B}} \lambda_{\min}(\widehat{\boldsymbol{\Upsilon}}_{t,\mathbf{x},\mathbf{x}})
        \gtrsim_{\mathbb{P}} 1,
        \qquad
        \inf_{\mathbf{x} \in \mathcal{B}} \widehat{\Xi}_{t,\mathbf{x},\mathbf{x}}
        \gtrsim_{\mathbb{P}} (n h^d)^{-1},
    \end{align*}
    and
    \begin{align*}
        \sup_{\mathbf{x}_1,\mathbf{x}_2 \in \mathcal{B}}
        \bigg|\frac{\widehat{\Xi}_{t,\mathbf{x}_1,\mathbf{x}_2}}{\sqrt{\widehat{\Xi}_{t,\mathbf{x}_1,\mathbf{x}_2} \widehat{\Xi}_{t,\mathbf{x}_2,\mathbf{x}_2}}} - \frac{\Xi_{t,\mathbf{x}_1,\mathbf{x}_2}}{\sqrt{\Xi_{t,\mathbf{x}_2,\mathbf{x}_2} \Xi_{t,\mathbf{x}_2,\mathbf{x}_2}}} \bigg|
        \lesssim_{\mathbb{P}} \sqrt{\frac{\log(1/h)}{n h^d}} + \frac{\log(1/h)}{n^{\frac{v}{2+v}}h^d}.
    \end{align*}
\end{lem}

\begin{lem}[Uniform Bias: Minimal Guarantee]\label{sa-lem: Uniform Bias: Minimal Guarantee}
    Suppose Assumptions \ref{sa-assump: DGP} (i)(ii)(iii) and \ref{sa-assump: Kernel, Distance and Boundary} hold. If $h\to0$, then
    \begin{align*}
        \sup_{\mathbf{x} \in \mathcal{B}}|\mathfrak{B}(\mathbf{x})| \lesssim h.
    \end{align*}
\end{lem}



\section{Identification and Point Estimation}

\begin{thm}[Distance-Based Identification]\label{sa-thm: Identification}
    Suppose Assumptions \ref{sa-assump: DGP}(i)-(iii) and \ref{sa-assump: Kernel, Distance and Boundary} hold. Then, $\tau(\mathbf{x}) = \lim_{r\downarrow0} \theta_{1,\mathbf{x}}(r) - \lim_{r\uparrow0} \theta_{0,\mathbf{x}}(r)$ for all $\mathbf{x} \in \mathcal{B}$.
\end{thm}

\begin{thm}[Pointwise Convergence Rate]\label{sa-thm: Pointwise Convergence Rate}
    Suppose Assumptions \ref{sa-assump: DGP} and \ref{sa-assump: Kernel, Distance and Boundary} hold. If $n h^d \to \infty$, then
    \begin{align*}
        \big|\widehat{\vartheta}(\mathbf{x}) - \tau(\mathbf{x}) \big|
        \lesssim_{\mathbb{P}} \frac{1}{\sqrt{n h^d}} + \frac{1}{n^{\frac{1+v}{2+v}}h^d} + \big|\mathfrak{B}(\mathbf{x})\big|.
    \end{align*}
\end{thm}

\begin{thm}[Uniform Convergence Rate]\label{sa-thm: Uniform Convergence Rate}
    Suppose Assumptions \ref{sa-assump: DGP} and \ref{sa-assump: Kernel, Distance and Boundary} hold. If $\frac{n h^d}{\log (1/h)} \to \infty$, then
    \begin{align*}
        \sup_{\mathbf{x} \in \mathcal{B}} \big|\widehat{\vartheta}(\mathbf{x}) - \tau(\mathbf{x})\big|
        \lesssim_{\mathbb{P}} \sqrt{\frac{\log(1/h)}{ n h^d}}
             + \frac{\log(1/h)}{n^{\frac{1+v}{2+v}}h^d}
             + \sup_{\mathbf{x} \in \mathcal{B}} \big|\mathfrak{B}(\mathbf{x})\big|.
    \end{align*}
\end{thm}

\section{Distributional Approximation and Inference}

Let $\mathbf{W} = ((\mathbf{X}_1^{\top},Y_1), \cdots, (\mathbf{X}_n^{\top},Y_n))$, and recall that $t \in \{0,1\}$. The feasible t-statistics is
\begin{align*}
    \widehat{\operatorname{T}}(\mathbf{x})
    = \frac{\widehat{\vartheta}(\mathbf{x}) - \tau(\mathbf{x})}
           {\sqrt{\widehat{\Xi}_{\mathbf{x},\mathbf{x}}}},
    \qquad \mathbf{x} \in \mathcal{B}.
\end{align*}
The associated $100(1-\alpha)\%$ confidence interval estimator is
\begin{align*}
    \widehat{\operatorname{I}}_{\alpha}(\mathbf{x})
    = \bigg[\;\widehat{\vartheta}(\mathbf{x}) - \mathfrak{q}_{\alpha} \sqrt{\widehat{\Xi}_{\mathbf{x},\mathbf{x}}}
            \; , \;
              \widehat{\vartheta}(\mathbf{x}) + \mathfrak{q}_{\alpha} \sqrt{\widehat{\Xi}_{\mathbf{x},\mathbf{x}}} \;\bigg],
\end{align*}
where $\mathfrak{q}_{\alpha}$ denotes an appropriate quantile depending on the desired confidence level $\alpha\in(0,1)$, and coverage objective (pointwise vs. uniform over $\mathcal{B}$). The following theorem establishes pointwise asymptotic normality and validity of confidence intervals. Let $\Phi(\cdot)$ be the cumulative distribution function of a standard univariate Gaussian random variable.

\begin{thm}[Confidence Intervals]\label{sa-thm: Confidence Intervals}
    Suppose Assumptions \ref{sa-assump: DGP} and \ref{sa-assump: Kernel, Distance and Boundary} hold. If $n^{\frac{v}{2+v}}h^d \to \infty$ and $\sqrt{n h^d} |\mathfrak{B}(\mathbf{x})| \to 0$, then
    \begin{align*}
        \sup_{u \in \mathbb{R}} \Big|\mathbb{P} \big(\widehat{\operatorname{T}}(\mathbf{x}) \leq u \big) - \Phi(u)\Big| = o(1),
        \qquad \mathbf{x} \in \mathcal{B},
    \end{align*}
    and
    \begin{align*}
        \mathbb{P} \big(\tau(\mathbf{x}) \in \widehat{\operatorname{I}}_{\alpha}(\mathbf{x}) \big) = 1 - \alpha + o(1),
        \qquad \mathbf{x} \in \mathcal{B},
    \end{align*}
    provided that $\mathfrak{q}_{\alpha} = \inf \{c > 0: \mathbb{P}( |\widehat{Z}| \geq c | \mathbf{W}) \leq \alpha \}$ with $\widehat{Z}|\mathbf{W} \thicksim \mathsf{Normal}(0, \widehat{\Xi}_{\mathbf{x},\mathbf{x}})$.
\end{thm}

To conduct uniform inference, and in particular construct confidence bands, we rely on a new strong approximation result established in Section \ref{sa-sec: Gaussian Strong Approximation}. First, we approximate (uniformly over $\mathbf{x}\in\mathcal{B}$) the feasible statistic $\widehat{\operatorname{T}}^{(\boldsymbol{\nu})}$ by the following linear statistic (which is a sum of independent random variables):
\begin{align*}
    \overline{\operatorname{T}}_{\operatorname{dis}}(\mathbf{x})
    = \Xi_{\mathbf{x},\mathbf{x}}^{-1/2} \Big(\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{1,\mathbf{x}}^{-1} \mathbf{O}_{1,\mathbf{x}}
                              - \mathbf{e}_1^{\top} \boldsymbol{\Psi}_{0,\mathbf{x}}^{-1} \mathbf{O}_{0,\mathbf{x}}\Big),
    \qquad \mathbf{x} \in \mathcal{B}.
\end{align*}


\begin{thm}[Stochastic Linearization]\label{sa-thm: Stochastic Linearization}
    Suppose Assumptions \ref{sa-assump: DGP} and \ref{sa-assump: Kernel, Distance and Boundary} hold. If $\frac{n h^d}{\log (1/h)} \to \infty$, then
    \begin{align*}
        \sup_{\mathbf{x} \in \mathcal{B}} \big|\widehat{\operatorname{T}}(\mathbf{x}) - \overline{\operatorname{T}}(\mathbf{x}) \big|
        \lesssim_{\mathbb{P}} \sqrt{\log(1/h)} \bigg(\sqrt{\frac{\log(1/h)}{n h^d}}
             + \frac{\log(1/h)}{n^{\frac{v}{2+v}}h^d} \bigg)
             + \sqrt{n h^d} \sup_{\mathbf{x} \in \mathcal{B}} |\mathfrak{B}(\mathbf{x})|.
    \end{align*}
\end{thm}

The pointwise (in $\mathcal{B}$) analogue of this result removes the $\log(1/h)$ penalty. See the proof of Theorem \ref{sa-thm: Confidence Intervals} for more details. To establish a Gaussian strong approximation for $\overline{\operatorname{T}}(\mathbf{x})$, define the class of functions $\mathcal{G} = \{g_{\mathbf{x}}: \mathbf{x} \in \mathcal{B}\}$ and $\mathscr{M} = \{m_{\mathbf{x}}: \mathbf{x} \in \mathcal{B}\}$, where
\begin{align}\label{sa-eq: def G and M}
    \nonumber g_{\mathbf{x}}(\mathbf{u})
    &= \mathds{1}(\mathbf{u} \in \mathcal{A}_1) \mathfrak{K}_1(\mathbf{u};\mathbf{x})
     - \mathds{1}(\mathbf{u} \in \mathcal{A}_0) \mathfrak{K}_0(\mathbf{u};\mathbf{x}),\\
    m_{\mathbf{x}}(\mathbf{u})
    &= - \mathds{1}(\mathbf{u} \in \mathcal{A}_1) \mathfrak{K}_1(\mathbf{u};\mathbf{x}) \theta_{1,\mathbf{x}}^{\ast}(\mathcal{d}(\mathbf{u},\mathbf{x}))
       + \mathds{1}(\mathbf{u} \in \mathcal{A}_0) \mathfrak{K}_0(\mathbf{u};\mathbf{x}) \theta_{0,\mathbf{x}}^{\ast}(\mathcal{d}(\mathbf{u},\mathbf{x})),
\end{align}
with
\begin{align*}
    \mathfrak{K}_t(\mathbf{u};\mathbf{x})
    = \frac{1}{\sqrt{n \Xi_{\mathbf{x},\mathbf{x}}}}\mathbf{e}_{1}^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{r}_p \Big(\frac{\mathcal{d}(\mathbf{u},\mathbf{x})}{h}\Big) K_h(\mathcal{d}(\mathbf{u},\mathbf{x})),
\end{align*}
for all $\mathbf{u} \in \mathcal{X}$, $\mathbf{x} \in \mathcal{B}$, and $t\in\{0,1\}$. In addition, let $\mathcal{R}$ be the class of functions containing the singleton identity function $\operatorname{Id}: \mathbb{R} \mapsto \mathbb{R}$, $\operatorname{Id}(x) = x$. Then, $\overline{\operatorname{T}}(\mathbf{x})$ can be represented as
\begin{align*}
    \overline{\operatorname{T}}(\mathbf{x})
    & = \frac{1}{\sqrt{n}} \sum_{i=1}^n
        \Big[g_{\mathbf{x}}(\mathbf{X}_i) \operatorname{Id}(y_i) + m_{\mathbf{x}}(\mathbf{X}_i)
        - \mathbb{E}\big[g_{\mathbf{x}}(\mathbf{X}_i) \operatorname{Id}(y_i) + m_{\mathbf{x}}(\mathbf{X}_i)\big]\Big].
\end{align*}
Following \cite{Cattaneo-Yu_2025_AOS}, we define the multiplicative separable empirical processes by
\begin{align*}
    M_n(g,r) = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \big[g(\mathbf{x}_i) r(y_i) - \mathbb{E}[g(\mathbf{x}_i) r(y_i)]\big],
    \qquad g \in \mathcal{G}, r \in \mathcal{R},
\end{align*}
which implies that
\begin{align*}
    \overline{\operatorname{T}}(\mathbf{x}) = M_n(g_{\mathbf{x}},\operatorname{Id}) + M_n(m_{\mathbf{x}},1), \qquad \mathbf{x} \in \mathcal{B}.
\end{align*}

Leveraging ideas in \cite{Cattaneo-Yu_2025_AOS}, Theorem~\ref{sa-thm: Strong Approximation} gives a new Gaussian strong approximation that can be applied to $\overline{\operatorname{T}}(\mathbf{x})$. This new theorem allows for polynomial moment bound on the conditional distribution of $Y_i|\mathbf{X}_i$.

\begin{thm}[Gaussian Strong Approximation: $\overline{\operatorname{T}}$]\label{sa-thm: Gaussian Strong Approximation: Tstat}
    Suppose Assumptions \ref{sa-assump: DGP} and \ref{sa-assump: Kernel, Distance and Boundary} hold, and that there exists a constant $C > 0$ such that for $t \in \{0,1\}$ and for any $\mathbf{x} \in \mathcal{B}$, the De Giorgi perimeter of the set $E_{t,\mathbf{x}} = \{\mathbf{y} \in \mathcal{A}_t: (\mathbf{y} - \mathbf{x})/h \in \operatorname{Supp}(K)\}$ satisfies  $\mathcal{L}(E_{t,\mathbf{x}}) \leq C h^{d-1}$. If $\liminf_{n \to \infty} \frac{\log h}{\log n} > - \infty$ and $n h^d \to \infty$ as $n \to \infty$, then (on a possibly enlarged probability space) there exists a mean-zero Gaussian process $Z$ indexed by $\mathcal{B}$ with almost surely continuous sample path such that
    \begin{align*}
        \mathbb{E}\Big[\sup_{\mathbf{x} \in \mathcal{B}} \big|\overline{\operatorname{T}}(\mathbf{x})- z(\mathbf{x}) \big| \Big]
        \lesssim (\log(n))^{\frac{3}{2}} \Big(\frac{1}{n h^d}\Big)^{\frac{1}{2d+2} \frac{v}{v + 2}}
                + \log(n) \Big(\frac{1}{n^{\frac{v}{2+v}}h^d}\Big)^{\frac{1}{2}},
    \end{align*}
    where $\lesssim$ is up to a universal constant, and $Z^{(\boldsymbol{\nu})}$ has the same covariance structure as $\overline{\operatorname{T}}$; i.e., $\mathbb{C}\mathrm{ov}[\overline{\operatorname{T}}(\mathbf{x}_1), \overline{\operatorname{T}}(\mathbf{x}_2)] = \mathbb{C}\mathrm{ov}[Z(\mathbf{x}_1), Z(\mathbf{x}_2)]$ for all $\mathbf{x}_1, \mathbf{x}_2 \in \mathcal{B}$.
\end{thm}

Theorem \ref{sa-thm: Gaussian Strong Approximation: Tstat} can be used to construct confidence bands for $(\tau(\mathbf{x}):\mathbf{x}\in\mathcal{B})$. Let $(\widehat{Z}(\mathbf{x}):\mathbf{x} \in \mathcal{B})$ be a (conditionally on $\mathbf{W}$) mean-zero Gaussian process with feasible (conditional) covariance function
\begin{align*}
    \mathbb{C}\mathrm{ov} \Big[\widehat{Z}(\mathbf{x}_1),\widehat{Z}(\mathbf{x}_2) \Big|\mathbf{W} \Big]
    = \frac{\sqrt{\widehat{\Xi}_{\mathbf{x}_1,\mathbf{x}_2}}}
           {\sqrt{\widehat{\Xi}_{\mathbf{x}_1,\mathbf{x}_1}} \sqrt{\widehat{\Xi}_{\mathbf{x}_2,\mathbf{x}_2}}},
    \qquad \mathbf{x}_1, \mathbf{x}_2 \in \mathcal{B}.
\end{align*}

\begin{thm}[Confidence Bands]\label{sa-thm: Confidence Bands}
    Suppose the assumptions and conditions in Theorem \ref{sa-thm: Gaussian Strong Approximation: Tstat} hold. If $\liminf_{n \to \infty} \frac{\log h}{\log n} > - \infty$, $\frac{n^{\frac{v}{2+v}}h^d}{(\log n)^3} \to \infty$ and $\sqrt{n h^d} \sup_{\mathbf{x} \in \mathcal{B}} |\mathfrak{B}(\mathbf{x})| \to 0$, then
    \begin{align*}
        \sup_{u \in \mathbb{R}} \Big|\mathbb{P} \Big(\sup_{\mathbf{x} \in \mathcal{B}} \big|\widehat{\operatorname{T}}(\mathbf{x})\big| \leq u \Big)
                            - \mathbb{P} \Big(\sup_{\mathbf{x} \in \mathcal{B}} \big|\widehat{Z}(\mathbf{x})\big| \leq u \Big| \mathbf{W} \Big) \Big| = o_{\mathbb{P}}(1)
    \end{align*}
    and
    \begin{align*}
        \mathbb{P}\Big[\tau^{(\boldsymbol{\nu})}(\mathbf{x}) \in \widehat{\operatorname{I}}_{\alpha}^{(\boldsymbol{\nu})}(\mathbf{x}), \text{ for all } \mathbf{x} \in \mathcal{B} \Big] = 1 - \alpha + o(1),
    \end{align*}
    provided that $\mathfrak{q}_{\alpha} = \inf \big\{c > 0: \mathbb{P} \big(\sup_{\mathbf{x} \in \mathcal{B}} \big|\widehat{Z}^{(\boldsymbol{\nu})}(\mathbf{x})\big|\geq c \big| \mathbf{W} \big) \leq \alpha \big\}$.
\end{thm}


\section{Gaussian Strong Approximation}\label{sa-sec: Gaussian Strong Approximation}

We present a Gaussian strong approximation theorem, which is the key technical tool behind Theorem~\ref{sa-thm: Gaussian Strong Approximation: Tstat}. The theorem builds on and generalizes the results in \cite{Cattaneo-Yu_2025_AOS}. Consider the \emph{residual-based empirical process} given by
\begin{align*}
    M_n[g,r] = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \Big[g(\mathbf{x}_i) r(y_i) - \mathbb{E}[g(\mathbf{x}_i) r(y_i)] \Big], \qquad g \in \mathcal{G}, r \in \mathcal{R}.
\end{align*}
where $\mathcal{G}$ and $\mathcal{R}$ are classes of functions satisfying certain regularity conditions.

\subsection{Definitions for Function Spaces}

Let $\mathcal{F}$ be a class of measurable functions from a probability space $(\mathbb{R}^q, \mathcal{B}(\mathbb{R}^q), \mathbb{P})$ to $\mathbb{R}$. We introduce several definitions that capture properties of $\mathcal{F}$.

\begin{enumerate}[label=(\roman*)]
    \item  $\mathcal{F}$ is pointwise measurable if it contains a countable subset $\mathcal{G}$ such that for any $f \in \mathcal{F}$, there exists a sequence $(g_m:m\geq1) \subseteq \mathcal{G}$ such that $\lim_{m \to \infty} g_m(\mathbf{u}) = f(\mathbf{u})$ for all $\mathbf{u} \in \mathbb{R}^q$.
    \item  Let $\operatorname{Supp}(\mathcal{F}) = \cup_{f \in \mathcal{F}}\operatorname{Supp}(f)$. A probability measure $\mathbb{Q}_\mathcal{F}$ on $(\mathbb{R}^q,\mathcal{B}(\mathbb{R}^q))$ is a surrogate measure for $\mathbb{P}$ with respect to $\mathcal{F}$ if
    \begin{enumerate}[label=(\roman*)]
        \item $\mathbb{Q}_\mathcal{F}$ agrees with $\mathbb{P}$ on $\operatorname{Supp}(\mathbb{P}) \cap \operatorname{Supp}(\mathcal{F})$.
        \item $\mathbb{Q}_\mathcal{F}(\operatorname{Supp}(\mathcal{F}) \setminus \operatorname{Supp}(\mathbb{P})) = 0$.
    \end{enumerate}
    Let $\mathcal{Q}_\mathcal{F}=\operatorname{Supp}(\mathbb{Q}_\mathcal{F})$.
    \item For $q=1$ and an interval $\mathcal{I}\subseteq\mathbb{R}$, the pointwise total variation of $\mathcal{F}$ over $\mathcal{I}$ is
    \begin{align*}
        \mathtt{pTV}_{\mathcal{F},\mathcal{I}} = \sup_{f \in \mathcal{F}} \sup_{P\geq1}\sup_{\mathcal{P}_P \in \mathcal{I}} \sum_{i = 1}^{P-1}|f(a_{i+1}) - f(a_i)|,
    \end{align*}
    where $\mathcal{P}_P=\{(a_1,\dots,a_P):a_1 \leq \cdots \leq a_P\}$ denotes the collection of all partitions of $\mathcal{I}$.
    \item For a non-empty $\mathcal{C} \subseteq \mathbb{R}^q$, the total variation of $\mathcal{F}$ over $\mathcal{C}$ is
    \begin{align*}
        \mathtt{TV}_{\mathcal{F}, \mathcal{C}} = \inf_{\mathcal{U} \in \mathcal{O}(\mathcal{C})}\sup_{f \in \mathcal{F}} \sup_{\phi \in \mathscr{D}_{q}(\mathcal{U})} \int_{\mathbb{R}^q} f(\mathbf{u})\operatorname{div}(\phi)(\mathbf{u}) d \mathbf{u} / \lVert \left\lVert\phi\right\rVert_2 \rVert_{\infty},
    \end{align*}
    where $\mathcal{O}(\mathcal{C})$ denotes the collection of all open sets that contains $\mathcal{C}$, and $\mathscr{D}_{q}(\mathcal{U})$ denotes the space of infinitely differentiable functions from $\mathbb{R}^q$ to $\mathbb{R}^q$ with compact support contained in $\mathcal{U}$.
    \item For a non-empty $\mathcal{C} \subseteq \mathbb{R}^q$, the local total variation constant of $\mathcal{F}$ over $\mathcal{C}$, is a positive number $\mathtt{K}_{\mathcal{F},\mathcal{C}}$ such that for any cube $\mathcal{D} \subseteq \mathbb{R}^q$ with edges of length $\ell$ parallel to the coordinate axises,
    \begin{align*}
        \mathtt{TV}_{\mathcal{F}, \mathcal{D} \cap \mathcal{C}}  \leq \mathtt{K}_{\mathcal{F}, \mathcal{C}} \ell^{d-1}.
    \end{align*}
    \item For a non-empty $\mathcal{C} \subseteq \mathbb{R}^q$, the envelopes of $\mathcal{F}$ over $\mathcal{C}$ are
    \begin{align*}
        \mathtt{M}_{\mathcal{F},\mathcal{C}} = \sup_{\mathbf{u} \in \mathcal{C} }M_{\mathcal{F},\mathcal{C}}(\mathbf{u}),
        \qquad M_{\mathcal{F},\mathcal{C}}(\mathbf{u}) = \sup_{f \in \mathcal{F}}|f(\mathbf{u})|,
        \qquad \mathbf{u} \in \mathcal{C}.
    \end{align*}
    \item  For a non-empty $\mathcal{C} \subseteq \mathbb{R}^q$, the Lipschitz constant of $\mathcal{F}$ over $\mathcal{C}$ is
    \begin{align*}
        \mathtt{L}_{\mathcal{F},\mathcal{C}} = \sup_{f \in \mathcal{F}}\sup_{\mathbf{u}_1, \mathbf{u}_2 \in \mathcal{C}} \frac{|f(\mathbf{u}_1) - f(\mathbf{u}_2)|}{\|\mathbf{u}_1 - \mathbf{u}_2\|_\infty}.
    \end{align*}
    \item For a non-empty $\mathcal{C} \subseteq \mathbb{R}^q$, the $L_1$ bound of $\mathcal{F}$ over $\mathcal{C}$ is
    \begin{align*}
        \mathtt{E}_{\mathcal{F},\mathcal{C}} = \sup_{f \in \mathcal{F}} \int_{\mathcal{C}} |f| d\mathbb{P}.
    \end{align*}
    \item   For a non-empty $\mathcal{C} \subseteq \mathbb{R}^q$, the uniform covering number of $\mathcal{F}$ with envelope $M_{\mathcal{F},\mathcal{C}}$ over $\mathcal{C}$ is
    \begin{align*}
        \mathtt{N}_{\mathcal{F},\mathcal{C}}(\delta,M_{\mathcal{F},\mathcal{C}}) = \sup_{\mu} N(\mathcal{F},\left\lVert\cdot\right\rVert_{\mu,2},\delta \left\lVertM_{\mathcal{F},\mathcal{C}}\right\rVert_{\mu,2}),
        \qquad \delta \in (0, \infty),
    \end{align*}
    where the supremum is taken over all finite discrete measures on $(\mathcal{C}, \mathcal{B}(\mathcal{C}))$. We assume that $M_{\mathcal{F},\mathcal{C}}(\mathbf{u})$ is finite for every $\mathbf{u} \in \mathcal{C}$.
    \item  For a non-empty $\mathcal{C} \subseteq \mathbb{R}^q$, the uniform entropy integral of $\mathcal{F}$ with envelope $M_{\mathcal{F},\mathcal{C}}$ over $\mathcal{C}$ is
    \begin{align*}
        J_\mathcal{C}(\delta, \mathcal{F}, M_{\mathcal{F},\mathcal{C}}) = \int_0^{\delta} \sqrt{1 + \log \mathtt{N}_{\mathcal{F},\mathcal{C}}(\varepsilon,M_{\mathcal{F},\mathcal{C}})} d \varepsilon,
    \end{align*}
    where it is assumed that $M_{\mathcal{F},\mathcal{C}}(\mathbf{u})$ is finite for every $\mathbf{u} \in \mathcal{C}$.
    \item For a non-empty $\mathcal{C} \subseteq \mathbb{R}^q$, $\mathcal{F}$ is a VC-type class with envelope $M_{\mathcal{F},\mathcal{C}}$ over $\mathcal{C}$ if (i) $M_{\mathcal{F},\mathcal{C}}$ is measurable and $M_{\mathcal{F},\mathcal{C}}(\mathbf{u})$ is finite for every $\mathbf{u} \in \mathcal{C}$, and (ii) there exist $\mathtt{c}_{\mathcal{F},\mathcal{C}}>0$ and $\mathtt{d}_{\mathcal{F},\mathcal{C}}>0$ such that
    \begin{align*}
        \mathtt{N}_{\mathcal{F},\mathcal{C}}(\varepsilon,M_{\mathcal{F},\mathcal{C}}) \leq \mathtt{c}_{\mathcal{F},\mathcal{C}} \varepsilon^{-\mathtt{d}_{\mathcal{F},\mathcal{C}}},
        \qquad \varepsilon\in(0,1).
    \end{align*}

\end{enumerate}

If a surrogate measure $\mathbb{Q}_\mathcal{F}$ for $\mathbb{P}$ with respect to $\mathcal{F}$ has been assumed, and it is clear from the context, we drop the dependence on $\mathcal{C} = \mathcal{Q}_{\mathcal{F}}$ for all quantities in the previous definitions. That is, to save notation, we set $\mathtt{TV}_{\mathcal{F}}=\mathtt{TV}_{\mathcal{F},\mathcal{Q}_{\mathcal{F}}}$, $\mathtt{K}_{\mathcal{F}}=\mathtt{K}_{\mathcal{F},\mathcal{Q}_{\mathcal{F}}}$, $\mathtt{M}_{\mathcal{F}}=\mathtt{M}_{\mathcal{F},\mathcal{Q}_{\mathcal{F}}}$, $M_{\mathcal{F}}(\mathbf{u})=M_{\mathcal{F},\mathcal{Q}_{\mathcal{F}}}(\mathbf{u})$, $\mathtt{L}_{\mathcal{F}}=\mathtt{L}_{\mathcal{F},\mathcal{Q}_{\mathcal{F}}}$, and so on, whenever there is no confusion.

\subsection{Multiplicative-Separable Empirical Process}

The following theorem generalizes \citet[Theorem SA.1]{Cattaneo-Yu_2025_AOS} by requiring only bounded polynomial moments for $y_i$ conditional on $\mathbf{x}_i$.

\begin{thm}[Strong Approximation for $(M_n(g,r) + M_n(h,s): g \in \mathcal{G}, r \in \mathcal{R}, h \in \mathscr{H}, s \in \mathcal{S})$]\label{sa-thm: Strong Approximation}
    Suppose $(\mathbf{z}_i=(\mathbf{x}_i, y_i): 1 \leq i \leq n)$ are i.i.d. random vectors taking values in $(\mathbb{R}^{d+1}, \mathcal{B}(\mathbb{R}^{d+1}))$ with common law $\mathbb{P}_Z$, where $\mathbf{x}_i$ has distribution $\mathbb{P}_X$ supported on $\mathcal{X}\subseteq\mathbb{R}^d$, $y_i$ has distribution $\mathbb{P}_Y$ supported on $\mathcal{Y}\subseteq\mathbb{R}$, $\sup_{\mathbf{x} \in \mathcal{X}}\mathbb{E}[|y_i|^{2 + v}|\mathbf{x}_i = \mathbf{x}] \leq 2$ for some $v > 0$, and the following conditions hold.

    \begin{enumerate}[label=\emph{(\roman*)}]
        \item $\mathcal{G}$ and $\mathscr{H}$ are real-valued pointwise measurable classes of functions on $(\mathbb{R}^d, \mathcal{B}(\mathbb{R}^d), \mathbb{P}_X)$.

        \item There exists a surrogate measure $\mathbb{Q}_{\mathcal{G} \cup \mathscr{H}}$ for $\mathbb{P}_X$ with respect to $\mathcal{G} \cup \mathscr{H}$ such that $\mathbb{Q}_{\mathcal{G} \cup \mathscr{H}} = \mathfrak{m} \circ \phi_{\mathcal{G} \cup \mathscr{H}}$, where the \textit{normalizing transformation} $\phi_{\mathcal{G} \cup \mathscr{H}}: \mathcal{Q}_{\mathcal{G} \cup \mathscr{H}} \mapsto [0,1]^d$ is a diffeomorphism.

        \item $\mathcal{G}$ is a VC-type class with envelope $\mathtt{M}_{\mathcal{G},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}}$ over $\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}$ with $\mathtt{c}_{\mathcal{G},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}} \geq e$ and $\mathtt{d}_{\mathcal{G},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}} \geq 1$. $\mathscr{H}$ is a VC-type class with envelope $\mathtt{M}_{\mathscr{H},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}}$ over $\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}$ with $\mathtt{c}_{\mathscr{H},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}} \geq e$ and $\mathtt{d}_{\mathscr{H},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}} \geq 1$.

        \item $\mathcal{R}$ and $\mathcal{S}$ are real-valued pointwise measurable classes of functions on $(\mathbb{R}, \mathcal{B}(\mathbb{R}),\mathbb{P}_Y)$.

        \item $\mathcal{R}$ is a VC-type class with envelope $M_{\mathcal{R},\mathcal{Y}}$ over $\mathcal{Y}$ with $\mathtt{c}_{\mathcal{R},\mathcal{Y}}\geq e$ and $\mathtt{d}_{\mathcal{R},\mathcal{Y}}\geq 1$, where $M_{\mathcal{R},\mathcal{Y}}(y) + \mathtt{pTV}_{\mathcal{R},(-|y|,|y|)} \leq \mathtt{v} (1 + |y|)$ for all $y \in \mathcal{Y}$, for some $\mathtt{v}>0$. $\mathcal{S}$ is a VC-type class with envelope $M_{\mathcal{S},\mathcal{Y}}$ over $\mathcal{Y}$ with $\mathtt{c}_{\mathcal{S},\mathcal{Y}}\geq e$ and $\mathtt{d}_{\mathcal{S},\mathcal{Y}}\geq 1$, where $M_{\mathcal{S},\mathcal{Y}}(y) + \mathtt{pTV}_{\mathcal{S},(-|y|,|y|)} \leq \mathtt{v} (1 + |y|)$ for all $y \in \mathcal{Y}$, for some $\mathtt{v}>0$.

        \item There exists a constant $\mathtt{k}$ such that $|\log_2 \mathtt{E}| + |\log_2 \mathtt{TV}| + |\log_2 \mathtt{M}| \leq \mathtt{k} \log_2(n)$, where $\mathtt{E} = \max \{\mathtt{E}_{\mathcal{G},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}}, \mathtt{E}_{\mathscr{H},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}}\}$, $\mathtt{TV} = \max \{\mathtt{TV}_{\mathcal{G},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}}, \mathtt{TV}_{\mathscr{H},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}}\}$ and $\mathtt{M} = \max \{\mathtt{M}_{\mathcal{G},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}}, \mathtt{M}_{\mathscr{H},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}}\}$.
    \end{enumerate}
    Consider the empirical process
    \begin{align*}
        A_n(g,h,r,s) = M_n(g,r) + M_n(h,s), \qquad g \in \mathcal{G}, r \in \mathcal{R}, h \in \mathscr{H}, s \in \mathcal{S}.
    \end{align*}
    Then, on a possibly enlarged probability space, there exists a sequence of mean-zero Gaussian processes $(Z_n^A(g,h,r,s): g\in\mathcal{G}, h \in \mathscr{H}, r \in \mathcal{R}, s \in \mathcal{S})$ with almost sure continuous trajectories such that:
    \begin{itemize}
        \item $\mathbb{E}[A_n(g_1,h_1,r_1,s_1) A_n(g_2,h_2,r_2,s_2)] = \mathbb{E}[Z_n^A(g_1, h_1, r_1,s_1) Z_n^A(g_2, h_2, r_2,s_2)]$ holds for all $(g_1, h_1, r_1, s_1)$, $(g_2, h_2, r_2, s_2) \in \mathcal{G} \times \mathscr{H} \times \mathcal{R} \times \mathcal{S}$, and
        \item $\mathbb{E}\big[\left\lVertA_n - Z_n^A\right\rVert_{\mathcal{G} \times \mathscr{H} \times \mathcal{R} \times \mathcal{S}}\big] \leq C \mathtt{v} ((\mathtt{d} \log (\mathtt{c} n))^{\frac{3}{2}} \mathtt{r}_n^{\frac{v}{v +2}}(\sqrt{\mathtt{M} \mathtt{E}})^{\frac{2}{v+2}} + \mathtt{d} \log(\mathtt{c} n) \mathtt{M} n^{-\frac{v/2}{2+v}} + \mathtt{d} \log(\mathtt{c} n) \mathtt{M} n^{-\frac{1}{2}} \Big(\frac{\sqrt{\mathtt{M} \mathtt{E}}}{\mathtt{r}_n}\Big)^{\frac{2}{v+2}})$,
    \end{itemize}
    where $C$ is a universal constant, $\mathtt{c} = \mathtt{c}_{\mathcal{G}, \mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}} + \mathtt{c}_{\mathscr{H},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}} + \mathtt{c}_{\mathcal{R},\mathcal{Y}} + \mathtt{c}_{\mathcal{S},\mathcal{Y}} + \mathtt{k}$, $\mathtt{d} = \mathtt{d}_{\mathcal{G},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}} \mathtt{d}_{\mathscr{H},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}} \mathtt{d}_{\mathcal{R},\mathcal{Y}} \mathtt{d}_{\mathcal{S},\mathcal{Y}} \mathtt{k}$,
    \begin{gather*}
        \mathtt{r}_n = \min\Big\{\frac{(\mathtt{c}_1^d \mathtt{M}^{d+1} \mathtt{TV}^d \mathtt{E})^{1/(2d+2)} }{n^{1/(2d+2)}}, \frac{(\mathtt{c}_1^{\frac{d}{2}} \mathtt{c}_2^{\frac{d}{2}}\mathtt{M} \mathtt{TV}^{\frac{d}{2}} \mathtt{E} \mathtt{L}^{\frac{d}{2}})^{1/(d+2)}}{n^{1/(d+2)}} \Big\}, \\
        \mathtt{c}_1 = d \sup_{\mathbf{x} \in \mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}} \prod_{j = 1}^{d-1} \sigma_j(\nabla \phi_{\mathcal{G} \cup \mathscr{H}}(\mathbf{x})), \qquad
        \mathtt{c}_2 = \sup_{\mathbf{x} \in \mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}} \frac{1}{\sigma_{d}(\nabla \phi_{\mathcal{G} \cup \mathscr{H}}(\mathbf{x}))}.
    \end{gather*}
\end{thm}


\section{Proofs}


\subsection{Proof of Lemma~\ref{sa-lem: Gram invert}}

Assumption~\ref{sa-assump: DGP} (ii) implies
\begin{align*}
    \boldsymbol{\Psi}_{t,\mathbf{x}} & = \mathbb{E} \Big[\mathbf{r}_p \Big(\frac{\lVert \mathbf{X}_i - \mathbf{x} \rVert}{h}\Big) \mathbf{r}_p \Big(\frac{\lVert \mathbf{X}_i - \mathbf{x} \rVert}{h}\Big)^{\top} K_h(\lVert \mathbf{X}_i - \mathbf{x} \rVert) \mathds{1}(\mathbf{X}_i \in \mathcal{A}_t) \Big] \\
    & = \int_{\mathcal{A}_t} \mathbf{r}_p \Big(\frac{\lVert \mathbf{u} - \mathbf{x} \rVert}{h}\Big) \mathbf{r}_p \Big(\frac{\lVert \mathbf{u} - \mathbf{x} \rVert}{h}\Big)^{\top} K_h(\lVert \mathbf{u} - \mathbf{x} \rVert) f(\mathbf{u}) d \mathbf{u} \\
    & = f(\mathbf{x})\int_{\mathcal{A}_t} \mathbf{r}_p \Big(\frac{\lVert \mathbf{u} - \mathbf{x} \rVert}{h}\Big) \mathbf{r}_p \Big(\frac{\lVert \mathbf{u} - \mathbf{x} \rVert}{h}\Big)^{\top} K_h(\mathbf{u} - \mathbf{x}) d \mathbf{u} + o(1),
\end{align*}
where in the last line we have used $\int_{\mathcal{A}_t} (\frac{\lVert \mathbf{u} - \mathbf{x} \rVert }{h})^{\mathbf{v}} K_h(\lVert \mathbf{u} - \mathbf{x} \rVert) d \mathbf{u} = O(1)$ for any multi-index $\mathbf{v}$ from standard change of variable argument.

\begin{center}
\textbf{I. Polynomial Representation of Minimum Eigenvalue}
\end{center}
For simplicity, call
\begin{align*}
    \mathbf{S}_{t,\mathbf{x}} = \lim_{h \to 0} \mathbf{S}_{t,\mathbf{x}}(h), \qquad \mathbf{S}_{t,\mathbf{x}}(h) = \int_{\mathcal{A}_t} \mathbf{r}_p \Big(\frac{\lVert \mathbf{u} - \mathbf{x} \rVert}{h}\Big) \mathbf{r}_p \Big(\frac{\lVert \mathbf{u} - \mathbf{x} \rVert}{h}\Big)^{\top} K_h(\lVert \mathbf{u} - \mathbf{x} \rVert) d \mathbf{u}.
\end{align*}
A change of variable gives
\begin{align*}
    \mathbf{S}_{t,\mathbf{x}}(h) = \int \mathbf{r}_p(\lVert \mathbf{z} \rVert) \mathbf{r}_p(\lVert \mathbf{z} \rVert)^{\top} K(\lVert \mathbf{z} \rVert) \mathds{1}(\mathbf{x} + h \mathbf{z} \in \mathcal{A}_t) d \mathbf{z}.
\end{align*}
Let $\mathbf{a} \in \mathbb{R}^{\mathfrak{p}_p}$, where $\mathfrak{p}_p = \frac{(d + p)!}{d! p !}$.
Then the equivalent representation of minimum eigenvalue gives
\begin{align}\label{sa-eq: min eigenvalue}
    \nonumber \lambda_{\min}(\mathbf{S}_{t,\mathbf{x}}(h))
    & = \min_{\left\lVert\mathbf{a}\right\rVert = 1} \int (\mathbf{a}^{\top} \mathbf{r}_p(\lVert \mathbf{z} \rVert))^2 K(\lVert \mathbf{z} \rVert) \mathds{1}(\mathbf{x} + h \mathbf{z} \in \mathcal{A}_t) d \mathbf{z} \\
    & \geq \kappa \min_{\left\lVert\mathbf{a}\right\rVert = 1} \int_{U} (\mathbf{a}^{\top} \mathbf{r}_p(\lVert \mathbf{z} \rVert))^2  \mathds{1}(\mathbf{x} + h \mathbf{z} \in \mathcal{A}_t) d \mathbf{z},
\end{align}
where in the last line we have used $K(\mathbf{u}) \geq \kappa$ for all $u \in U$.

\begin{center}
\textbf{II. Mass Retaining Ratio in Treatment/Control Region}
\end{center}

Denote $E_h(\mathbf{x},t) = \{\mathbf{z} \in U: \mathbf{x} + h \mathbf{z} \in \mathcal{A}_t \}$. Assumption~\ref{sa-assump: Kernel, Distance and Boundary} (iii) implies there is some upper bound $\Lambda > 0$ of $K(\cdot)$. Hence for $c_0 = 1/2 \; \liminf_{h \downarrow 0}\inf_{\mathbf{x} \in \mathcal{B}} \int_{U} K(\lVert \mathbf{u} \rVert) \mathds{1}(\mathbf{x} + h \mathbf{u} \in \mathcal{A}_t) d \mathbf{u}$, we have
\begin{align*}
    \Lambda \mathfrak{m}(E_h(\mathbf{x},t))
    & \geq \int_{U} K(\lVert \mathbf{u} \rVert) \mathds{1}(\mathbf{x} + h \mathbf{u} \in \mathcal{A}_t)
    \geq c_0
\end{align*}
for small enough $h$, which implies
\begin{align}\label{sa-eq: mass relation}
      \mathfrak{m}(E_h(\mathbf{x},t))  \geq \alpha \mathfrak{m}(U), \qquad \alpha = \frac{c_0}{\Lambda \mathfrak{m}(U)}.
\end{align}

\begin{center}
\textbf{III. $L_2$ Integral of Polynomials in Full v.s. Treatment/Control Regions}
\end{center}

Consider $S = \{f \in \mathcal{P}_{p+1}: \int_U f(\lVert \mathbf{u} \rVert)^2 d \mathbf{u} = 1\}$, where $\mathcal{P}_{p+1}$ is the collection of all $(p+1)$-order polynomials. Let $(\phi_j, 1 \leq j \leq p+1)$ be a set of orthonormal basis of $(\mathcal{P}_{p+1}, \lVert \cdot \rVert_{L_2})$. Then $T(\mathbf{a}) = \sum_{j = 1}^{p+1} a_j \phi_j$ is an isometry. Since $T(S) = \{\mathbf{a} \in \mathbb{R}^{p+1}: \left\lVert\mathbf{a}\right\rVert = 1\}$ is compact, $S$ is also compact in $(\mathcal{P}_{p+1}, \lVert \cdot \rVert_{L_2})$. Since $\mathcal{P}_{p+1}$ is $(p+1)$-dimensional, equivalent of norms implies that $S$ is also compact in $(\mathcal{P}_{p+1}, \lVert \cdot \rVert_{L_{\infty}})$. Now consider
\begin{align*}
    \Phi_q(\varepsilon) = \mathfrak{m}(\{\mathbf{u} \in U: |q(u)| < \varepsilon \}), \qquad q \in S, \varepsilon > 0,
\end{align*}
and
\begin{align*}
    \psi(q) = \sup \Big\{\varepsilon > 0: \Phi_q(\varepsilon) \leq \frac{\alpha}{2} \mathfrak{m}(U) \Big\}.
\end{align*}
Since $\int_U q^2 = 1$ and $q$ is polynomial on norm, $\lim_{\varepsilon \downarrow 0} \Phi_q(\varepsilon) = 0$ and $\Phi_q(\lVert q\rVert_{\infty}) = \mathfrak{m}(U)$. Continuity and Lipchitzness of $q \in S$ imply $\psi(q) > 0$ for all $q \in S$.

Next, we want to show $\psi$ is lower-semicontinous function on $(\mathcal{P}_{p+1}, \lVert \cdot \rVert_{L_{\infty}})$. Suppose $q_n \to q$ uniformly on $U$. For every $\varepsilon_0 \in (0, \psi(q))$, there exists $\eta > 0$ such that $\Phi_q(\varepsilon_0) \leq \frac{\alpha}{2} \mathfrak{m}(U) - \eta$. Continuity of polynomials and the fact that level sets of polynomials have zero Lebesgue measure imply $\mathds{1}_{\{|q_n| < \varepsilon_0\}}(\cdot) \to \mathds{1}_{\{|q| < \varepsilon_0\}}(\cdot)$ almost surely. By Dominated Convergence Theorem, $\Phi_{q_n}(\varepsilon_0) \to \Phi_q(\varepsilon_0)$. Hence for large enough $n$, $\Phi_{q_n}(\varepsilon_0) \leq \frac{\alpha}{2}\mathfrak{m}(U)$, which implies $\varepsilon_0 \leq \psi(q_n)$. This implies $\liminf_{n \to \infty} \psi(q_n) \geq \varepsilon_0$. Since $\varepsilon_0$ is arbitrary in $(0, \psi(q))$, we have $\liminf_{n \to \infty} \psi(q_n) \geq \psi(q)$.

Compactness of $S$ and lower-semicontinuity of $\psi$ implies $\psi$ attains its minimum on $S$. Since $\psi(q) > 0$ for all $q \in S$, we know $\varepsilon_* = \inf_{q \in S} \psi(q) > 0$. Then for every $q \in S$,
\begin{align*}
    \int_{E_h(\mathbf{x},t)} q^2
    & \geq \varepsilon_*^2 \; \mathfrak{m} \Big(E_h(\mathbf{x},t) \setminus \{|q| \leq \varepsilon_*\} \Big) \\
    & \geq \varepsilon_*^2 \; \Big(\mathfrak{m}(E_h(\mathbf{x},t)) - \mathfrak{m}(\{|q| \leq \varepsilon_*\}) \Big) \\
    & \geq \varepsilon_*^2 \; \frac{\alpha}{2} \mathfrak{m}(U).
\end{align*}
Scaling $q$ from $S$ gives
\begin{align}\label{sa-eq: integral relation}
    \int_{E_h(\mathbf{x},t)} q^2 \geq \varepsilon_*^2 \; \frac{\alpha}{2} \int_{U} q^2, \qquad q \in \mathcal{P}_{p+1}.
\end{align}

\begin{center}
\textbf{IV. Lower Bound of Minimum Eigenvalue}
\end{center}

Equations~\eqref{sa-eq: min eigenvalue}, \eqref{sa-eq: mass relation} and \eqref{sa-eq: integral relation} together give for small enough $h$,
\begin{align*}
    \inf_{\mathbf{x} \in \mathcal{B}}\lambda_{\min}(\mathbf{S}_{t,\mathbf{x}}(h))
    & \geq \kappa \inf_{\mathbf{x} \in \mathcal{B}}\min_{\left\lVert\mathbf{a}\right\rVert = 1} \int_{E_h(\mathbf{x},t)} (\mathbf{a}^{\top} \mathbf{r}_p(\lVert \mathbf{z} \rVert))^2   d \mathbf{z}, \\
    & \geq \kappa \varepsilon_*^2 \; \frac{\alpha}{2}  \min_{\left\lVert\mathbf{a}\right\rVert = 1} \int_{U} (\mathbf{a}^{\top} \mathbf{r}_p(\lVert \mathbf{z} \rVert))^2 d \mathbf{z} \\
    & \geq \kappa \varepsilon_*^2 \; \frac{\alpha}{2}  \lambda_{\min} \Big(\int_U \mathbf{r}_p(\lVert \mathbf{z} \rVert) \mathbf{r}_p(\lVert \mathbf{z} \rVert)^{\top} d \mathbf{z} \Big),
\end{align*}
which implies $\liminf_{h \to 0} \inf_{\mathbf{x} \in \mathcal{B}}\lambda_{\min}(\mathbf{S}_{t,\mathbf{x}}(h)) > 0$.

\subsection{Proof of Lemma~\ref{sa-lem: Gram}}

Since $\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}$ is a finite dimensional matrix, it suffices to show the stated rate of convergence for each entry. For $0 \leq v \leq p$, define $\mathcal{G} = \{g_n(\cdot, \mathbf{x}) \mathds{1}(\cdot \in \mathcal{A}_t): \mathbf{x} \in \mathcal{X} \}$ with
\begin{align*}
    g_n(\xi, \mathbf{x})
    = \Big(\frac{\mathcal{d}(\xi,\mathbf{x})}{h}\Big)^{v} \frac{1}{h^d} K\Big(\frac{\mathcal{d}(\xi, \mathbf{x})}{h}\Big),
    \qquad
    \xi,\mathbf{x} \in \mathcal{X}.
\end{align*}
We will show $\mathcal{G}$ is a VC-type of class.

\medskip\textit{Constant Envelope Function}. We assume $K$ is continuous and has compact support, and hence there exists a constant $C_1$ such that  $\sup_{\mathbf{x} \in \mathcal{X}} \lVert g_n(\cdot,\mathbf{x}) \rVert_{\infty} \leq C_1 h^{-d} = G$.

\medskip\textit{Diameter of $\mathcal{G}$ in $L_2$}. For each $\mathbf{x} \in \mathcal{X}$, $g_n(\cdot,\mathbf{x})$ is supported on $\{\xi: \mathcal{d}(\xi,\mathbf{x}) \leq h\}$. By Assumption~\ref{sa-assump: DGP}(ii) and Assumption~\ref{sa-assump: Kernel, Distance and Boundary}(i), $\sup_{\mathbf{x} \in \mathcal{X}} \mathbb{P} \left(\mathcal{d}(\mathbf{X}_i, \mathbf{x}) \leq h \right) \lesssim h^d$. It follows that
$ \sup_{\mathbf{x} \in \mathcal{X}} \left\lVertg_n(\cdot,\mathbf{x})\right\rVert_{\mathbb{P},2} \leq C_2 h^{-d/2}$ for some constant $C_2$. We can take $C_1$ large enough so that $\sigma = C_2 h^{-d/2} \leq G = C_1 h^{-d}$.

\medskip\textit{Ratio}. For some constant $C_3$, $\delta = \frac{\sigma}{F}  = C_3 \sqrt{h^d}$.

\medskip\textit{Covering Numbers}. Case 1: $K$ is Lipschitz. Let $\mathbf{x},\mathbf{x}' \in \mathcal{X}$. By Assumption~\ref{sa-assump: Kernel, Distance and Boundary},
\begin{align*}
    &\sup_{\xi \in \mathcal{X}} \big|g_n(\xi, \mathbf{x}) - g_n(\xi,\mathbf{x}')\big|\\
    &\leq \sup_{\xi \in \mathcal{X}}
          \Big[\Big(\frac{\mathcal{d}(\xi,\mathbf{x})}{h}\Big)^v - \Big(\frac{\mathcal{d}(\xi,\mathbf{x}')}{h}\Big)^v\Big] K_h(\mathcal{d}(\xi,\mathbf{x}))
     + \Big(\frac{\mathcal{d}(\xi,\mathbf{x}')}{h}\Big)^v \Big[K_h(\mathcal{d}(\xi,\mathbf{x})) - K_h(\mathcal{d}(\xi,\mathbf{x}'))\big]\\
    & \lesssim h^{-d-1} \lVert \mathbf{x} - \mathbf{x}' \rVert_{\infty}.
\end{align*}
By Lipschitz continuity property of $\mathcal{G}$, for any $\varepsilon \in (0,1]$ and for any finitely supported measure $Q$ and metric $\left\lVert\cdot\right\rVert_{Q,2}$ based on $L_2(Q)$,
\begin{align*}
    N(\{g_n(\cdot,\mathbf{x}): \mathbf{x} \in \mathcal{X}\}, \left\lVert\cdot\right\rVert_{Q,2}, \varepsilon \|G\|_{Q,2})
    \leq N(\mathcal{X}, \lVert \cdot \rVert_{\infty}, \varepsilon \|G\|_{Q,2}h^{d+1})
    \stackrel{(i)}{\lesssim} \Big(\frac{\operatorname{diam}(\mathcal{X})}{ \varepsilon \|G\|_{Q,2} h^{d+1}}\Big)^d
    \lesssim \Big(\frac{\operatorname{diam}(\mathcal{X})}{\varepsilon h}\Big)^d,
\end{align*}
where inequality (i) uses the fact that $\varepsilon \|G\|_{Q,2} h^{d+1} \lesssim \varepsilon h \lesssim 1$. Thus, $\mathcal{G}$ forms a VC-type class in that $\sup_{Q} N(\mathcal{G}, \left\lVert\cdot\right\rVert_{Q,2}, \varepsilon \|G\|_{Q,2}) \lesssim (C_1/\epsilon)^{C_2}$ for all $\epsilon \in (0,1]$ with $C_1 = \frac{\operatorname{diam}(\mathcal{X})}{h}$ and $C_2 = d$. Moreover, for any discrete measure $Q$, and for any $\mathbf{x}, \mathbf{x}' \in \mathcal{X}$, $\left\lVertg_n(\cdot,\mathbf{x}) \mathds{1}(\cdot \in \mathcal{A}_t) - g_n(\cdot,\mathbf{x}') \mathds{1}(\cdot \in \mathcal{A}_t)\right\rVert_{Q,2} \leq \left\lVertg_n(\cdot,\mathbf{x})- g_n(\cdot,\mathbf{x}')\right\rVert_{Q,2}$. Therefore,
\begin{align*}
    \sup_{Q} N(\mathcal{G}, \left\lVert\cdot\right\rVert_{Q,2}, \varepsilon \left\lVertG\right\rVert_{Q,2})
    \leq N(\mathcal{G}, \left\lVert\cdot\right\rVert_{Q,2}, \varepsilon \left\lVertG\right\rVert_{Q,2})
    \leq (C_1/\varepsilon)^{C_2}, \qquad \varepsilon \in (0,1],
\end{align*}
where the supremum is taken over all finite discrete measures on $\mathcal{X}$.

Case 2: $k = \mathds{1}(\cdot \in [-1,1])$. Consider
\begin{align*}
    m_n(\xi, \mathbf{x})
    = \Big(\frac{\mathcal{d}(\xi,\mathbf{x})}{h}\Big)^{v} \frac{1}{h} \mathds{1}(\xi \in \mathcal{A}_t),
    \qquad \xi, \mathbf{x} \in \mathcal{X},
\end{align*}
$\mathcal{M} = \{m_n(\mathcal{d}(\cdot,\mathbf{x}) : \mathbf{x} \in \mathcal{B}\}$ and the constant envelope function $M = C_4 h^{-v - 1}$, for some constant $C_4$ only depending on diameter of $\mathcal{X}$. The same argument as before shows that for any discrete measure $Q$, we have
\begin{align*}
    N(\mathcal{M}, \left\lVert\cdot\right\rVert_{Q,2}, \varepsilon \left\lVertM\right\rVert_{Q,2})
    & \leq N(\mathcal{X}, \lVert \cdot \rVert_{\infty}, \varepsilon \left\lVertM\right\rVert_{Q,2} h^{1 + v + 1})
    \lesssim \Big(\frac{\operatorname{diam}(\mathcal{X})}{\varepsilon \left\lVertM\right\rVert_{Q,2} h^{1 + v + 1}}\Big)^d
    \lesssim \Big(\frac{\operatorname{diam}(\mathcal{X})}{\varepsilon h}\Big)^d.
\end{align*}
The class $\mathscr{L} = \{\mathds{1}((\cdot - \mathbf{x})/h \in [-1,1]^d): \mathbf{x} \in \mathcal{B}\}$ has VC dimension no greater than $2d$ \cite[Example 2.6.1]{van-der-Vaart-Wellner_1996_Book}, and by \citet[Theorem 2.6.4]{van-der-Vaart-Wellner_1996_Book},
\begin{align*}
    \sup_{Q} N(\mathcal{G}, \left\lVert\cdot\right\rVert_{Q,2}, \varepsilon \left\lVertG\right\rVert_{Q,2})
    \leq N(\mathcal{G}, \left\lVert\cdot\right\rVert_{Q,2}, \varepsilon \left\lVertG\right\rVert_{Q,2})
    \leq (C_1/\varepsilon)^{C_2}, \qquad \varepsilon \in (0,1],
\end{align*}
where the supremum is taken over all finite discrete measures on $\mathcal{X}$.

\medskip\textit{Maximal Inequality}. By \citet[Corollary 5.1]{Chernozhukov-Chetverikov-Kato_2014b_AoS} for the empirical process on class $\mathcal{G}$,
\begin{align*}
    \mathbb{E} \Big[\sup_{l \in \mathcal{G}} \big|\mathbb{E}_n \left[l(\mathbf{X}_i)\right]  - \mathbb{E}[l(\mathbf{X}_i)]\big| \Big]
    &\lesssim \frac{\sigma}{\sqrt{n}}\sqrt{C_2\log(C_1/\delta)} + \frac{\|G\|_{\mathbb{P},2} C_2\log(C_1/\delta)}{n}\\
    &\lesssim \frac{1}{\sqrt{n h^d}}\sqrt{d\log \Big( \frac{\operatorname{diam}(\mathcal{X})}{h^{1 + d/2}} \Big)}
     + \frac{1}{n h^d}d \log \Big( \frac{\operatorname{diam}(\mathcal{X})}{h^{1 + d/2}} \Big)\\
    &\lesssim \sqrt{\frac{\log n}{n h^d}}.
\end{align*}
Thus, $\sup_{\mathbf{x} \in \mathcal{X}} \big\|\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}} - \boldsymbol{\Psi}_{t,\mathbf{x}}\big\| \lesssim_{\mathbb{P}} \sqrt{\frac{\log n}{n h^d}}$.

By Weyl's Theorem, $\sup_{\mathbf{x} \in \mathcal{X}}|\lambda_{\min}(\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}) - \lambda_{\min}(\boldsymbol{\Psi}_{t,\mathbf{x}})| \leq \sup_{\mathbf{x} \in \mathcal{X}} \|\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}} - \boldsymbol{\Psi}_{t,\mathbf{x}}\| \lesssim_{\mathbb{P}} \sqrt{\frac{\log n}{n h^d }}$. Therefore, we can lower bound the minimum eigenvalue by $\inf_{\mathbf{x} \in \mathcal{X}} \lambda_{\min}(\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}) \geq \inf_{\mathbf{x} \in \mathcal{X}}\lambda_{\min}(\boldsymbol{\Psi}_{t,\mathbf{x}}) - \sup_{\mathbf{x} \in \mathcal{X}}|\lambda_{\min}(\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}) - \lambda_{\min}(\boldsymbol{\Psi}_{t,\mathbf{x}})| \gtrsim_{\mathbb{P}} 1$.

Finally, it follows that $\sup_{\mathbf{x} \in \mathcal{X}} \|\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}^{-1}\| \lesssim_{\mathbb{P}} 1$ and hence
\begin{align*}
    \sup_{\mathbf{x} \in \mathcal{X}} \big\|\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}^{-1} - \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1}\big\|
    \leq \sup_{\mathbf{x} \in \mathcal{X}} \big\|\boldsymbol{\Psi}_{t,\mathbf{x}}^{-1}\big\| \big\|\boldsymbol{\Psi}_{t,\mathbf{x}} -\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}\big\| \big\|\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}^{-1}\big\|
    \lesssim_{\mathbb{P}} \sqrt{\frac{\log n}{n h^{d}}},
\end{align*}
which completes the proof.
\qed


\subsection{Proof of Lemma~\ref{sa-lem: Stochastic Linear Approximation}}

Consider the class $\mathcal{F} = \{(\mathbf{z},u) \mapsto \mathbf{e}_{\nu}^{\top} g_{\mathbf{x}}(\mathbf{z})(u - h_{\mathbf{x}}(\mathbf{z})): \mathbf{x} \in \mathcal{B} \}$, $0 \leq \boldsymbol{\nu} \leq p$, where for $\mathbf{z} \in \mathcal{X}$,
\begin{gather*}
    g_{\mathbf{x}}(\mathbf{z}) = \mathbf{r}_p \Big(\frac{\mathcal{d}(\mathbf{z},\mathbf{x})}{h}\Big) K_h(\mathcal{d}(\mathbf{z},\mathbf{x})), \qquad
    h_{\mathbf{x}}(\mathbf{z}) = \boldsymbol{\gamma}_t^{\ast}(\mathbf{x})^{\top} \mathbf{r}_p\left(\mathcal{d}(\mathbf{z},\mathbf{x})\right).
\end{gather*}
By definition of $\boldsymbol{\gamma}_t^{\ast}(\mathbf{x})$,
\begin{align}\label{sa-eq: best fit}
    \boldsymbol{\gamma}_t^{\ast}(\mathbf{x})
    = \mathbf{H}^{-1}\boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{S}_{t,\mathbf{x}},
    \qquad
    \mathbf{S}_{t,\mathbf{x}}
    = \mathbb{E} \Big[\mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big) K_h(D_i(\mathbf{x})) Y_i \mathds{1}(\mathbf{X}_i \in \mathcal{A}_t) \Big].
\end{align}
Assumption~\ref{sa-assump: DGP} implies $\mathbf{S}_{t,\mathbf{x}}$ is continuous in $\mathbf{x}$, hence $\sup_{\mathbf{x} \in \mathcal{X}} \left\lVert\mathbf{S}_{t,\mathbf{x}}\right\rVert \lesssim 1$. And by Assumption~\ref{sa-assump: Kernel, Distance and Boundary}(ii), $\inf_{\mathbf{x} \in \mathcal{X}} \lambda_{\min}(\boldsymbol{\Psi}_{t,\mathbf{x}}) \gtrsim 1$. Hence
\begin{align}\label{sa-eq: surrogate}
    \sup_{\mathbf{x} \in \mathcal{B}} \left\lVert\boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{S}_{t,\mathbf{x}}\right\rVert \lesssim 1.
\end{align}

Now, consider properties of $\mathcal{F}$. Definition of $\boldsymbol{\gamma}_t^{\ast}(\mathbf{x})$ implies $\mathbb{E}[f(\mathbf{X}_i,Y_i)] = 0$ for all $f \in \mathcal{F}$. Since $K$ is compactly supported, there exists $C_1, C_2 > 0$ such that $F(\mathbf{z},u) = C_1 h^{-d} (|u| + C_2)$ is an envelope function for $\mathcal{F}$. Denote $M = \max_{1 \leq i \leq n} F(\mathbf{X}_i, Y_i)$, then
\begin{align*}
    \mathbb{E}[M^2]^{1/2} & \lesssim h^{-d} \mathbb{E} \left[\max_{1 \leq i \leq n} |Y_i|^2 + 1\right]^{1/2} \lesssim h^{-d} \mathbb{E} \left[\max_{1 \leq i \leq n} |Y_i|^{2+v}\right]^{1/(2+v)} \\
    & \lesssim h^{-d} \bigg[\sum_{i = 1}^n \mathbb{E}[|\varepsilon_i + \sum_{t \in \{0,1\}} \mathds{1}(\mathbf{x} \in \mathcal{A}_t)\mu_t(\mathbf{x})|^{2+v}] \bigg]^{1/(2+v)}
    \lesssim h^{-d} n^{1/(2+v)},
\end{align*}
where we have used $\mathbf{X}$ is compact and $\mu_t$ is continuous, hence $\sup_{\mathbf{x} \in \mathcal{X}}|\sum_{t \in \{0,1\}} \mathds{1}(\mathbf{x} \in \mathcal{A}_t)\mu_t(\mathbf{x})| \lesssim 1$. Denote $\sigma = \sup_{f \in \mathcal{F}} \mathbb{E}[f(\mathbf{X}_i,Y_i)^{2}]^{1/2}$. Then,
\begin{align*}
    \sigma^2 \lesssim \sup_{\mathbf{x} \in \mathcal{B}}\mathbb{E} [\lVert \mathbf{e}_{\nu}^{\top}g_{\mathbf{x}} \rVert_{\infty}^2(|Y_i| + \lVert \mathbf{e}_{\nu}^{\top}h_{\mathbf{x}} \rVert_{\infty})^2\mathds{1}(K_h(D_i(\mathbf{x})) \neq 0)] \lesssim h^{-d}.
\end{align*}
To check for the covering number of $\mathcal{F}$, notice that compare to the proof of Lemma~\ref{sa-lem: Gram}, we have one more term $\mathbf{e}_{\boldsymbol{\nu}}^\top g_{\mathbf{x}} h_{\mathbf{x}} = \mathbf{r}_p \left(\frac{\mathcal{d}(\mathbf{z},\mathbf{x})}{h}\right)K_h(\mathcal{d}(\mathbf{z},\mathbf{x})) \boldsymbol{\gamma}_t^{\ast}(\mathbf{x})^{\top} \mathbf{r}_p\left(\mathcal{d}(\mathbf{z},\mathbf{x})\right)$. All terms except for $\boldsymbol{\gamma}_t^{\ast}(\mathbf{x})$ can be handled as in the proof of Lemma~\ref{sa-lem: Gram}. Recall Equation~\eqref{sa-eq: best fit}, and consider $l_{t,\mathbf{x}} = \mathbf{e}_{\mathbf{v}}^\top[ \mathbf{R}(\mathcal{d}(\cdot,\mathbf{x})/h) K_h(\mathcal{d}(\cdot, \mathbf{x})) \mu_t \mathds{1}(\cdot \in \mathcal{A}_t)$ and $\mathscr{L}_t = \{l_{t,\mathbf{x}}: \mathbf{x} \in \mathcal{B}\}$, $\mathbf{v}$ is a any multi-index. Then, for any $\mathbf{x}_1, \mathbf{x}_2 \in \mathcal{B}$,
\begin{align*}
    |\mathbf{S}_{t,\mathbf{x}_1} - \mathbf{S}_{t,\mathbf{x}_2}| \leq \left\lVertl_{t,\mathbf{x}_1} - l_{t,\mathbf{x}_2}\right\rVert_{\mathbb{P}_X,2},
\end{align*}
and hence
\begin{align*}
    N(\{\mathbf{e}_{\mathbf{v}}^{\top}\mathbf{S}_{t,\mathbf{x}}: \mathbf{x} \in \mathcal{B}\}, |\cdot|, \varepsilon h^{-d}) & \leq N(\mathscr{L}_t, \left\lVert\cdot\right\rVert_{\mathbb{P}_X,2}, \varepsilon h^{-d})
    \leq \sup_{Q} N(\mathscr{L}_t, \left\lVert\cdot\right\rVert_{Q,2}, \varepsilon h^{-d}),
\end{align*}
Same argument as paragraph \textbf{Covering Numbers} in the proof of Lemma~\ref{sa-lem: Gram} then shows
\begin{align*}
   &  \sup_{Q} N(\{g_{\mathbf{x}}: \mathbf{x} \in \mathcal{B}\}, \left\lVert\cdot\right\rVert_{Q,2}, \varepsilon C_1 h^{-d}) \leq \left(\frac{\operatorname{diam}(\mathcal{X})}{h \varepsilon}\right)^d, \qquad 0 < \varepsilon \leq 1, \\
   &  \sup_{Q} N(\{g_{\mathbf{x}} h_{\mathbf{x}}: \mathbf{x} \in \mathcal{B}\}, \left\lVert\cdot\right\rVert_{Q,2}, \varepsilon C_1 h^{-d}) \leq \left(\frac{\operatorname{diam}(\mathcal{X})}{h \varepsilon}\right)^d, \qquad 0 < \varepsilon \leq 1,
\end{align*}
where $\sup$ is taken over all discrete measures on $\mathcal{X}$. Product $\{g_{\mathbf{x}}: \mathbf{x} \in \mathcal{B}\}$ with the singleton of identity function $\{u \mapsto u, u \in \mathbb{R}\}$, and adding $\{g_{\mathbf{x}} h_{\mathbf{x}}: \mathbf{x} \in \mathcal{B}\}$,
\begin{align*}
    \sup_{Q} N(\mathcal{F}, \left\lVert\cdot\right\rVert_{Q,2}, \varepsilon \left\lVertF\right\rVert_{Q,2}) \leq 2\left(\frac{2\operatorname{diam}(\mathcal{X})}{h \varepsilon}\right)^d, \qquad 0 < \varepsilon \leq 1,
\end{align*}
where $\sup$ is taken over all discrete measures on $\mathcal{X} \times \mathbb{R}$. Denote $\mathtt{C}_1 = d$, $\mathtt{C}_2 = \frac{2 (2 \operatorname{diam}(\mathcal{X}))^d}{h^d}$. Hence, by \citet[Corollary 5.1]{Chernozhukov-Chetverikov-Kato_2014b_AoS}
\begin{align*}
 \mathbb{E}\bigg[\sup_{\mathbf{x} \in \mathcal{B}}|\mathbf{e}_{\boldsymbol{\nu}}^{\top} \mathbf{O}_{t,\mathbf{x}}|\bigg]& = \mathbb{E} \bigg[\sup_{f \in \mathcal{F}} \left|\mathbb{E}_n \left[f(\mathbf{X}_i,Y_i)\right]  - \mathbb{E}[f(\mathbf{X}_i,Y_i)]\right| \bigg] \\
 & \lesssim \frac{\sigma}{\sqrt{n}}\sqrt{\mathtt{C}_2\log(\mathtt{C}_1 \left\lVertM\right\rVert_{\mathbb{P},2}/\sigma)} + \frac{\|M\|_{\mathbb{P},2} \mathtt{C}_2\log(\mathtt{C}_1 \left\lVertM\right\rVert_{\mathbb{P},2}/\sigma)}{n}\\
 & \lesssim \frac{1}{\sqrt{n h^d}}\sqrt{d\log \left( \frac{\operatorname{diam}(\mathcal{X})}{h^{1 + d/2}} \right)}
+ \frac{1}{n^{\frac{1+v}{2+v}} h^d}d \log \left( \frac{\operatorname{diam}(\mathcal{X})}{h^{1 + d/2}} \right) \\
 & \lesssim \sqrt{\frac{\log (1/h)}{n h^d}} + \frac{\log (1/h)}{n^{\frac{1+v}{2+v}}h^d}.
\end{align*}
The rest follows from finite dimensionality of $\mathbf{O}_{t,\mathbf{x}}$, and Lemma~\ref{sa-lem: Gram}.
\qed


\subsection{Proof of Lemma~\ref{sa-lem: Covariance}}

Denote $\eta_{i,t,\mathbf{x}} = Y_i - \theta_{t,\mathbf{x}}^{\ast}(D_i(\mathbf{x}))$ and $\xi_{i,t,\mathbf{x}} = \theta_{t,\mathbf{x}}^{\ast}(D_i(\mathbf{x}))
- \widehat{\theta}_{t,\mathbf{x}}(D_i(\mathbf{x}))$. Then
\begin{align*}
    \widehat{\boldsymbol{\Upsilon}}_{t,\mathbf{x},\mathbf{y}} = \mathbb{E}_n \bigg[\mathbf{r}_p \left(\frac{D_i(\mathbf{x})}{h}\right) \mathbf{r}_p \left( \frac{D_i(\mathbf{y})}{h}\right)^{\top} h^d K_h \left(D_i(\mathbf{x})\right) K_h \left(D_i(\mathbf{y})\right) (\eta_{i,t,\mathbf{x}} + \xi_{i,t,\mathbf{x}})^2 \mathds{1}(D_i(\mathbf{x}) \in \mathcal{I}_t)\bigg],
\end{align*}
and we decompose the error into
\begin{align*}
    & \widehat{\boldsymbol{\Upsilon}}_{t,\mathbf{x},\mathbf{y}} - \boldsymbol{\Upsilon}_{t,\mathbf{x},\mathbf{y}} = \Delta_{1,t,\mathbf{x},\mathbf{y}} + \Delta_{2,t,\mathbf{x},\mathbf{y}} + \Delta_{3,t,\mathbf{x},\mathbf{y}}, \\
    & \Delta_{1,t,\mathbf{x},\mathbf{y}} = \mathbb{E}_n \bigg[\mathbf{r}_p \left(\frac{D_i(\mathbf{x})}{h}\right) \mathbf{r}_p \left( \frac{D_i(\mathbf{y})}{h}\right)^{\top} h^d K_h \left(D_i(\mathbf{x})\right) K_h \left(D_i(\mathbf{y})\right) \xi_{i,t,\mathbf{x}}^2 \mathds{1}(D_i(\mathbf{x}) \in \mathcal{I}_t)\bigg], \\
    & \Delta_{2,t,\mathbf{x},\mathbf{y}} = 2 \mathbb{E}_n \bigg[\mathbf{r}_p \left(\frac{D_i(\mathbf{x})}{h}\right) \mathbf{r}_p \left( \frac{D_i(\mathbf{y})}{h}\right)^{\top} h^d K_h \left(D_i(\mathbf{x})\right) K_h \left(D_i(\mathbf{y})\right) \eta_{i,t,\mathbf{x}} \xi_{i,t,\mathbf{x}} \mathds{1}(D_i(\mathbf{x}) \in \mathcal{I}_t)\bigg], \\
    & \Delta_{3,t,\mathbf{x},\mathbf{y}} =  \mathbb{E}_n \bigg[\mathbf{r}_p \left(\frac{D_i(\mathbf{x})}{h}\right) \mathbf{r}_p \left( \frac{D_i(\mathbf{y})}{h}\right)^{\top} h^d K_h \left(D_i(\mathbf{x})\right) K_h \left(D_i(\mathbf{y})\right) \eta_{i,t,\mathbf{x}}^2 \mathds{1}(D_i(\mathbf{x}) \in \mathcal{I}_t)\bigg] \\
    & \qquad \qquad \qquad - \mathbb{E} \bigg[\mathbf{r}_p \left(\frac{D_i(\mathbf{x})}{h}\right) \mathbf{r}_p \left( \frac{D_i(\mathbf{y})}{h}\right)^{\top} h^d K_h \left(D_i(\mathbf{x})\right) K_h \left(D_i(\mathbf{y})\right) \eta_{i,t,\mathbf{x}}^2 \mathds{1}(D_i(\mathbf{x}) \in \mathcal{I}_t)\bigg].
\end{align*}
By Assumption~\ref{sa-assump: Kernel, Distance and Boundary}, $K_h(D_i(\mathbf{x})) \neq 0$ implies $\left\lVert\mathbf{r}_p(D_i(\mathbf{x})/h)\right\rVert_2 \lesssim 1$. Hence by Lemma~\ref{sa-lem: Gram} and \ref{sa-lem: Stochastic Linear Approximation},
\begin{align*}
    & \max_{t \in \{0,1\}} \max_{1 \leq i \leq n} \sup_{\mathbf{x} \in \mathcal{B}}|\xi_{i,t,\mathbf{x}}| \\
    & = \max_{t \in \{0,1\}} \max_{1 \leq i \leq n} \sup_{\mathbf{x} \in \mathcal{B}} |\mathbf{r}_p(D_i(\mathbf{x}))^{\top}(\widehat{\boldsymbol{\gamma}}_{t,\mathbf{x}} - \boldsymbol{\gamma}_{t,\mathbf{x}}^{\ast})| \mathds{1}(K_h(D_i(\mathbf{x})) \geq 0) \\
    & = \max_{t \in \{0,1\}} \max_{1 \leq i \leq n} \sup_{\mathbf{x} \in \mathcal{B}} |\mathbf{r}_p(D_i(\mathbf{x}))^{\top} \mathbf{H}^{-1} (\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}^{-1} \mathbf{O}_{t,\mathbf{x}} +  (\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}^{-1} - \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1})  \mathbf{U}_{t,\mathbf{x}})| \mathds{1}(K_h(D_i(\mathbf{x})) \geq 0)  \\
    & \leq \max_{t \in \{0,1\}} \sup_{\mathbf{x} \in \mathcal{B}} \left\lVert\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}^{-1} \mathbf{O}_{t,\mathbf{x}}\right\rVert_2 + \max_{t \in \{0,1\}} \sup_{\mathbf{x} \in \mathcal{B}} \left\lVert(\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}^{-1} - \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1})\mathbf{U}_{t,\mathbf{x}}\right\rVert_2\\
    & \lesssim \sqrt{\frac{\log(1/h)}{n h^d}} + \frac{\log(1/h)}{n^{\frac{1+v}{2+v}}h^d},
\end{align*}
where
\begin{align*}
     \mathbf{U}_{t,\mathbf{x}} = \mathbb{E}_n \left[\mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big) K_h(D_i(\mathbf{x})) \theta_{t,\mathbf{x}}^{\ast}(\mathbf{X}_i)\mathds{1}(D_i(\mathbf{x}) \in \mathcal{I}_t)\right].
\end{align*}
Assuming $\frac{\log(1/h)}{n^{\frac{1+v}{2+v}}h^d} \to \infty$, similar maximal inequality as in the proof of Lemma~\ref{sa-lem: Gram} shows
\begin{align}\label{sa-eq: covariance for distance -- 1}
    \nonumber & \sup_{\mathbf{x},\mathbf{y} \in \mathcal{X}}\left\lVert\Delta_{1,t,\mathbf{x},\mathbf{y}}\right\rVert \lesssim_{\mathbb{P}} \max_{t \in \{0,1\}} \max_{1 \leq i \leq n} \sup_{\mathbf{x} \in \mathcal{B}}|\xi_{i,t,\mathbf{x}}|^2 \lesssim \bigg(\sqrt{\frac{\log(1/h)}{n h^d}} + \frac{\log(1/h)}{n^{\frac{1+v}{2+v}}h^d}\bigg)^2, \\
    & \sup_{\mathbf{x},\mathbf{y} \in \mathcal{X}}\left\lVert\Delta_{2,t,\mathbf{x},\mathbf{y}}\right\rVert \lesssim_{\mathbb{P}} \max_{t \in \{0,1\}} \max_{1 \leq i \leq n} \sup_{\mathbf{x} \in \mathcal{B}}|\xi_{i,t,\mathbf{x}}| \lesssim \sqrt{\frac{\log(1/h)}{n h^d}} + \frac{\log(1/h)}{n^{\frac{1+v}{2+v}}h^d}.
\end{align}
Consider the $(\mu,\nu)$ entry of $\Delta_{3,t,\mathbf{x},\mathbf{y}}$. Consider the class $$\mathcal{F} = \bigg\{(\mathbf{z},u) \mapsto \left(\frac{\mathcal{d}(\mathbf{z},\mathbf{x})}{h}\right)^{\mu + \nu}h^d K_h(\mathcal{d}(\mathbf{z},\mathbf{x})) K_h(\mathcal{d}(\mathbf{z},\mathbf{y}))(u - \mathbf{r}_p(\mathcal{d}(\mathbf{z},\mathbf{x}))^{\top}\boldsymbol{\gamma}_{t,\mathbf{x}}^{\ast})^2: \mathbf{x}, \mathbf{y} \in \mathcal{X}\bigg\}.$$
 By Assumption~\ref{sa-assump: Kernel, Distance and Boundary} and \ref{sa-assump: DGP}(v), we have $\sup_{f \in \mathcal{F}} \mathbb{E}[f(\mathbf{X}_i, Y_i)^2]^{1/2} \lesssim h^{-d/2}$. Moreover, Assumption~\ref{sa-assump: Kernel, Distance and Boundary} and Equation~\eqref{sa-eq: surrogate} imply there exists $C_1, C_2 > 0$ such that $F(\mathbf{z},u) = C_1 h^{-d}(u^2 + C_2)$ is an envelope function for $\mathcal{F}$, with
\begin{align*}
    \mathbb{E} \big[ \max_{1 \leq i \leq n} F(\mathbf{X}_i,Y_i)^2 \big]^{\frac{1}{2}}
    \lesssim C_1 h^{-d} (\mathbb{E} \big[\max_{1 \leq i \leq n} Y_i^4 \big]^{\frac{1}{2}} + C_2)
    \lesssim C_1 h^{-d} (\mathbb{E} \big[\max_{1 \leq i \leq n} Y_i^{2+v} \big]^{\frac{2}{2+v}} + C_2) \lesssim h^{-d} n^{\frac{2}{2+v}}.
\end{align*}
Apply \citet[Corollary 5.1]{Chernozhukov-Chetverikov-Kato_2014b_AoS} similarly as in Lemma~\ref{sa-lem: Stochastic Linear Approximation} gives
\begin{align*}
    \mathbb{E} \left[\sup_{f \in \mathcal{F}}\left|\mathbb{E}_n[f(\mathbf{X}_i,Y_i)] - \mathbb{E}[f(\mathbf{X}_i,Y_i)] \right|\right] \lesssim \sqrt{\frac{\log(1/h)}{n h^d}} + \frac{\log(1/h)}{n^{\frac{v}{2+v}}h^d}.
\end{align*}
Finite dimensionality of $\Delta_{3,t,\mathbf{x},\mathbf{y}}$ then implies
\begin{align}\label{sa-eq: covariance for distance -- 2}
     \mathbb{E} \left[\sup_{\mathbf{x},\mathbf{y} \in \mathcal{X}}\left\lVert\Delta_{3,t,\mathbf{x},\mathbf{y}}\right\rVert\right] \lesssim \sqrt{\frac{\log(1/h)}{n h^d}} + \frac{\log(1/h)}{n^{\frac{v}{2+v}}h^d}.
\end{align}
Putting together Equations~\eqref{sa-eq: covariance for distance -- 1}, \eqref{sa-eq: covariance for distance -- 2} and Lemma~\ref{sa-lem: Gram} gives the result.
\qed


\subsection{Proof of Lemma~\ref{sa-lem: Uniform Bias: Minimal Guarantee}}

By Theorem~\ref{sa-thm: Identification} and Equation~\eqref{sa-eq: best fit}, we have
\begin{align*}
    \sup_{\mathbf{x} \in \mathcal{B}}|\mathfrak{B}_{n,t}(\mathbf{x})|
    & = \sup_{\mathbf{x} \in \mathcal{B}} \Big|\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{S}_{t,\mathbf{x}} - \mu_t(\mathbf{x}) \Big| \\
    & = \sup_{\mathbf{x} \in \mathcal{B}} \Big|\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbb{E} \Big[\mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big) K_h(D_i(\mathbf{x})) \mathbf{R}_p(D_i(\mathbf{x}))^{\top} (\mu_t(\mathbf{X}_i) - \mu_t(\mathbf{x}),0,\cdots,0)) \mathds{1}(\mathbf{X}_i \in \mathcal{A}_t) \Big] \Big| \\
    & \lesssim \sup_{\mathbf{x} \in \mathcal{B}} \sup_{\mathbf{z} \in \mathcal{X}} |\mu_t(\mathbf{x}) - \mu_t(\mathbf{z})| \mathds{1}(K_h(\mathcal{d}(\mathbf{z},\mathbf{x})) > 0) \\
    & \lesssim h.
\end{align*}
\qed


\subsection{Proof of Theorem~\ref{sa-thm: Identification}}

Since $\theta_{\mathbf{x}}(0) = \theta_{1,\mathbf{x}}(0) - \theta_{0,\mathbf{x}}(0)$ and $\tau(\mathbf{x}) = \mu_1(\mathbf{x}) - \mu_0(\mathbf{x})$, it is enough to prove the result for one treatment assignment group $t\in\{0,1\}$. By Assumption~\ref{sa-assump: DGP}(iii) and Assumption~\ref{sa-assump: Kernel, Distance and Boundary}(ii), for any $r \neq 0$, for any $\mathbf{x} \in \mathcal{B}$ and $\mathbf{y} \in S_{t,\mathbf{x}}(r)$, $|\mu_t(\mathbf{y}) - \mu_t(\mathbf{x})| \lesssim |r|$. Hence, for any $r \neq 0$, for any $\mathbf{x} \in \mathcal{B}$, $t \in \{0,1\}$,
\begin{align*}
    |\theta_{t,\mathbf{x}}(r) - \mu_t(\mathbf{x})|
    \leq \frac{\int_{S_{t,\mathbf{x}}(|r|)} |\mu_t(\mathbf{y}) - \mu_t(\mathbf{x})| f_X(\mathbf{y}) \mathfrak{H}^{d-1}(d \mathbf{y})}
              {\int_{S_{t,\mathbf{x}}(|r|)} f_X(\mathbf{y}) \mathfrak{H}^{d-1}(d \mathbf{y})} \lesssim r.
\end{align*}
implying
\begin{align*}
    |\theta_{t,\mathbf{x}}(0) - \mu_t(\mathbf{x})| \leq \lim_{r \to 0} |\theta_{t,\mathbf{x}}(r) - \mu_t(\mathbf{x})| = 0,
\end{align*}
which establishes the result.
\qed


\subsection{Proof of Theorem~\ref{sa-thm: Pointwise Convergence Rate}}

The proofs of Lemma~\ref{sa-lem: Gram} and Lemma~\ref{sa-lem: Stochastic Linear Approximation} can be done when the index set is the singleton $\{\mathbf{x}\}$ instead of $\mathcal{B}$, replacing \citet[Corollary 5.1]{Chernozhukov-Chetverikov-Kato_2014b_AoS} by Bernstein inequality, and thus obtaining
\begin{align*}
    \Big|\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{O}_{t,\mathbf{x}}\Big|
    & \lesssim_{\mathbb{P}} \sqrt{\frac{1}{n h^d}} + \frac{1}{n^{\frac{1+v}{2+v}}h^d}, \\
    \Big|\mathbf{e}_1^{\top} (\widehat{\boldsymbol{\Psi}}_{t,\mathbf{x}}^{-1} - \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1}) \mathbf{O}_{t,\mathbf{x}}\Big|
    & \lesssim_{\mathbb{P}} \sqrt{\frac{1}{n h^d}} \bigg(\sqrt{\frac{1}{n h^d}} + \frac{1}{n^{\frac{1+v}{2+v}}h^d} \bigg).
\end{align*}
for all $\mathbf{x} \in \mathcal{B}$. In words, uniformity only adds a $\log(1/h)$ penalty. Therefore, using decomposition \eqref{sa-eq: distance error decomposition}, the pointwise convergence rate follows.
\qed


\subsection{Proof of Theorem~\ref{sa-thm: Uniform Convergence Rate}}

Follows from Lemma~\ref{sa-lem: Gram}, Lemma~\ref{sa-lem: Stochastic Linear Approximation} and decomposition \eqref{sa-eq: distance error decomposition}.
\qed


\subsection{Proof of Theorem~\ref{sa-thm: Confidence Intervals}}

Define $\overline{\operatorname{T}}(\mathbf{x}) = \sum_{i = 1}^n Z_i$, with $Z_i = Z_{1,i} - Z_{0,i}$ independent random variables ($i=1,2,\ldots,n)$,
\begin{align*}
    Z_{t,i} = \frac{1}{n} \Xi_{\mathbf{x},\mathbf{x}}^{-1/2} \mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1}
              \mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big) K_h(D_i(\mathbf{x})) (Y_i - \theta_{t,\mathbf{x}}^{\ast}(D_i(\mathbf{x}))) \mathds{1}(\mathbf{X}_i \in \mathcal{A}_t),
\end{align*}
$\mathbb{E}[Z_{i}] = 0$ and $\mathbb{V}[Z_{i}] = n^{-1}$. By the Berry-Essen Theorem,
\begin{align*}
    \sup_{u \in \mathbb{R}} \Big|\mathbb{P} \big(\overline{\operatorname{T}}(\mathbf{x}) \leq u \big) - \Phi(u) \Big|
    \lesssim \sum_{i = 1}^n \mathbb{E} \big[|Z_{i}|^3 \big]
    \lesssim \sum_{i = 1}^n \mathbb{E} \big[|Z_{1,i}|^3 \big] + \sum_{i = 1}^n \mathbb{E} \big[|Z_{0,i}|^3 \big]
\end{align*}
where
\begin{align*}
    \mathbb{E} \big[|Z_{t,i}|^3 \big]
    &= \sum_{i = 1}^n n^{-3} \Xi_{\mathbf{x},\mathbf{x}}^{-3/2}
       \mathbb{E} \Big[ \Big|\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{r}_p \Big(\frac{D_i(\mathbf{x})}{h}\Big) K_h(D_i(\mathbf{x})) \mathds{1}(\mathbf{X}_i \in \mathcal{A}_t) (Y_i - \theta_{t,\mathbf{x}}^{\ast}(D_i(\mathbf{x})) \Big|^3 \Big] \\
    & \lesssim n^{-2} \Xi_{\mathbf{x},\mathbf{x}}^{-3/2} \mathbb{E}[|K_h(D_i(\mathbf{x}))(Y_i - \theta_{t,\mathbf{x}}^{\ast}(D_i(\mathbf{x})))|^3]\\
    & \lesssim n^{-2} \Xi_{\mathbf{x},\mathbf{x}}^{-3/2} \mathbb{E}[|K_h(D_i(\mathbf{x}))(\mathbb{E}[|Y_i|^3|\mathbf{X}_i] + |\theta_{t,\mathbf{x}}^{\ast}(D_i(\mathbf{x}))|^3)|]\\
    & \lesssim (n h^d)^{-1/2},
\end{align*}
noting that $\sup_{\mathbf{x} \in \mathcal{B}} \big\|\mathbf{r}_p \big(\frac{D_i(\mathbf{x})}{h}\big) K_h(D_i(\mathbf{x}))\big\| \lesssim 1$ holds almost surely in $\mathbf{X}_i$, $\Xi_{\mathbf{x},\mathbf{x}} \gtrsim (n h^d)^{-1/2}$ by Lemma~\ref{sa-lem: Covariance}, $\mathbb{E}[|Y_i|^3|\mathbf{X}_i] \lesssim 1$ by Assumption~\ref{sa-assump: DGP}(v), and $\max_{1 \leq i \leq n}\sup_{\mathbf{x} \in \mathcal{B}}|\theta_{t,\mathbf{x}}^{\ast}(D_i(\mathbf{x}))| \lesssim 1$ because
\begin{align*}
    \theta_{t,\mathbf{x}}^{\ast}(D_i(\mathbf{x}))
    = \gamma_t^{\ast}(\mathbf{x})^{\top} \mathbf{r}_p (D_i(\mathbf{x}))
    = (\boldsymbol{\Psi}_{t,\mathbf{x}} \mathbf{S}_{t,\mathbf{x}})^{-1} \mathbf{r}_p \Big(\frac{D_i(\mathbf{x})}{h}\Big).
\end{align*}

Since
\begin{align*}
    \big|\widehat{\operatorname{T}}(\mathbf{x}) - \overline{\operatorname{T}}(\mathbf{x})\big|
    \lesssim_{\mathbb{P}} \frac{1}{\sqrt{n h^d}}
           + \frac{1}{n^{\frac{v}{2+v}}h^d}
           + \sqrt{n h^d} |\mathfrak{B}(\mathbf{x})|,
\end{align*}
the pointwise asymptotic normality follows, under the conditions imposed. Finally, validity of the confidence interval estimator is immediate.
\qed


\subsection{Proof of Theorem~\ref{sa-thm: Stochastic Linearization}}

We make the decomposition based on Equation~\eqref{sa-eq: distance error decomposition} and convergence of $\widehat{\Xi}_{\mathbf{x},\mathbf{x}}$,
\begin{align*}
    \widehat{\operatorname{T}}_{\operatorname{dis}}(\mathbf{x}) - \overline{\operatorname{T}}_{\operatorname{dis}}(\mathbf{x}) & = \widehat{\Xi}_{\mathbf{x},\mathbf{x}}^{-1/2}\bigg(\sum_{t \in \{0,1\}}(-1)^{\frac{t + 1}{2}}(\widehat{\theta}_{t,\mathbf{x}}(0) - \theta_{t,\mathbf{x}}(0))\bigg) - \Xi_{\mathbf{x},\mathbf{x}}^{-1/2} \bigg(\sum_{t \in \{0,1\}}(-1)^{\frac{t + 1}{2}}\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{O}_{t,\mathbf{x}}\bigg) \\
    & = \widehat{\Xi}_{\mathbf{x},\mathbf{x}}^{-1/2}\bigg(\sum_{t \in \{0,1\}}(-1)^{\frac{t + 1}{2}}(\widehat{\theta}_{t,\mathbf{x}}(0) - \theta_{t,\mathbf{x}}(0)) -\sum_{t \in \{0,1\}}(-1)^{\frac{t + 1}{2}}\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{O}_{t,\mathbf{x}} \bigg)   \tag{$=\Delta_{1,\mathbf{x}}$} \\
    & \phantom{=} + (\widehat{\Xi}_{\mathbf{x},\mathbf{x}}^{-1/2} - \Xi_{\mathbf{x},\mathbf{x}}^{-1/2}) \sum_{t \in \{0,1\}}(-1)^{\frac{t + 1}{2}}\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{O}_{t,\mathbf{x}} \tag{$=\Delta_{2,\mathbf{x}}$}
\end{align*}
By Lemma~\ref{sa-lem: Gram} and \ref{sa-lem: Stochastic Linear Approximation}, and the decomposition Equation~\eqref{sa-eq: distance error decomposition},
\begin{align*}
    & \sup_{\mathbf{x} \in \mathcal{X}} \bigg|\sum_{t \in \{0,1\}}(-1)^{\frac{t + 1}{2}}(\widehat{\theta}_{t,\mathbf{x}}(0) - \theta_{t,\mathbf{x}}(0)) -\sum_{t \in \{0,1\}}(-1)^{\frac{t + 1}{2}}\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{O}_{t,\mathbf{x}}\bigg| \\
    & \lesssim_{\mathbb{P}} \sqrt{\frac{\log(1/h)}{n h^d}} \left(\sqrt{\frac{\log(1/h)}{n h^d}} + \frac{\log(1/h)}{n^{\frac{1+v}{2+v}}h^d} \right) + \sup_{\mathbf{x} \in \mathcal{B}}\sum_{t \in \{0,1\}}|\theta_{t,\mathbf{x}}^{\ast}(0) - \theta_{t,\mathbf{x}}(0)|.
\end{align*}
Together with Lemma~\ref{sa-lem: Covariance},
\begin{align}\label{sa-eq: stochastic linearization for distance -- 1}
    \sup_{\mathbf{x} \in \mathcal{B}}|\Delta_{1,\mathbf{x}}| \lesssim_{\mathbb{P}} \frac{\log(1/h)}{\sqrt{n h^d}} + \frac{(\log(1/h))^{\frac{3}{2}}}{n^{\frac{1+v}{2+v}}h^d} + \sqrt{n h^d} \sup_{\mathbf{x} \in \mathcal{B}}\sum_{t \in \{0,1\}}|\theta_{t,\mathbf{x}}^{\ast}(0) - \theta_{t,\mathbf{x}}(0)|.
\end{align}
By Lemma~\ref{sa-lem: Gram}, Lemma~\ref{sa-lem: Stochastic Linear Approximation} and Lemma~\ref{sa-lem: Covariance}, and assume $\frac{n^{\frac{v}{2+v}}h^d}{\log(1/h)} \to \infty$, then
\begin{align*}
    \sup_{\mathbf{x} \in \mathcal{X}} \left|\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{O}_{t,\mathbf{x}} \left(\Xi_{\mathbf{x},\mathbf{x}}^{-1/2} - \widehat{\Xi}_{\mathbf{x},\mathbf{x}}^{-1/2}\right) \right| & \lesssim_{\mathbb{P}} \sqrt{n h^d} \bigg(\sqrt{\frac{\log(1/h)}{n h^d}} + \frac{\log(1/h)}{n^{\frac{1+v}{2+v}}h^d}\bigg) \left(\sqrt{\frac{\log(1/h)}{n h^d}} + \frac{\log(1/h)}{n^{\frac{v}{2+v}}h^d} \right) \\
    & = \sqrt{\log(1/h)}\bigg(1 + \sqrt{\frac{\log(1/h)}{n^{\frac{v}{2+v}}h^d}}\bigg) \left(\sqrt{\frac{\log(1/h)}{n h^d}} + \frac{\log(1/h)}{n^{\frac{v}{2+v}}h^d} \right)  \\
    & \lesssim \sqrt{\log(1/h)} \left(\sqrt{\frac{\log(1/h)}{n h^d}} + \frac{\log(1/h)}{n^{\frac{v}{2+v}}h^d}\right).
\end{align*}
Hence
\begin{align}\label{sa-eq: stochastic linearization for distance -- 2}
    \sup_{\mathbf{x} \in \mathcal{B}}|\Delta_{2,\mathbf{x}}| \lesssim_{\mathbb{P}} \frac{\log(1/h)}{\sqrt{n h^d}} + \frac{(\log(1/h))^{\frac{3}{2}}}{n^{\frac{1+v}{2+v}}h^d}.
\end{align}
Putting together Equations~\eqref{sa-eq: stochastic linearization for distance -- 1}, \eqref{sa-eq: stochastic linearization for distance -- 2} give the result.
\qed

\subsection{Proof of Theorem~\ref{sa-thm: Gaussian Strong Approximation: Tstat}}

We will verify the high level conditions stated in Theorem~\ref{sa-thm: Strong Approximation}.

Without loss of generality, we can assume $\mathcal{X} = [0,1]^d$, and $\mathcal{Q}_{\mathcal{F}_t} = \mathbb{P}_X$ is a valid surrogate measure for $\mathbb{P}_X$ with respect to $\mathcal{G}$, and $\phi_{\mathcal{G}} = \operatorname{Id}$ is a valid normalizing transformation (as in Theorem~\ref{sa-thm: Strong Approximation}). This implies the constants $\mathtt{c}_1$ and $\mathtt{c}_2$ from Theorem~\ref{sa-thm: Strong Approximation} are all $1$.

Recall $\mathcal{G} = \{g_{\mathbf{x}}: \mathbf{x} \in \mathcal{B}\}$ where
\begin{align*}
    g_{\mathbf{x}}(\mathbf{u})
    &= \mathds{1}(\mathbf{u} \in \mathcal{A}_1) \mathfrak{K}_1(\mathbf{u};\mathbf{x})
     - \mathds{1}(\mathbf{u} \in \mathcal{A}_0) \mathfrak{K}_0(\mathbf{u};\mathbf{x}).
\end{align*}
By standard arguments and \cite[Lemma 7]{Cattaneo-Chandak-Jansson-Ma_2024_Bernoulli}, we get properties of $\mathcal{G}$ as follows:
\begin{align*}
    \mathtt{M}_{\mathcal{G}} \lesssim h^{-d/2}, \qquad \mathtt{E}_{\mathcal{G}} \lesssim h^{d/2}, \qquad
    \mathtt{TV}_{\mathcal{G}} \lesssim h^{d/2-1}, \qquad \sup_{Q} N(\mathcal{G}, \left\lVert\cdot\right\rVert_{Q,2}, \varepsilon (2 c +1)^{d+1} \mathtt{M}_{\mathcal{G}}) \leq 2 \mathbf{c}' \varepsilon^{-d-1} + 2.
\end{align*}
By definition of $\theta_{t,\mathbf{x}}^{\ast}(\cdot)$, for each $\mathbf{x} \in \mathcal{B}$, $t \in \{0,1\}$,
\begin{align*}
    \theta_{t,\mathbf{x}}^{\ast}(\mathcal{d}(\mathbf{u},\mathbf{x})) = \boldsymbol{\gamma}_t^{\ast}(\mathbf{x})^{\top}\mathbf{r}_p(\mathcal{d}(\mathbf{u},\mathbf{x})) = (\mathbf{H}^{-1} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{S}_{t,\mathbf{x}})^{\top} \mathbf{r}_p(\mathcal{d}(\mathbf{u},\mathbf{x}))
    = (\boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{S}_{t,\mathbf{x}})^{\top} \mathbf{r}_p\Big(\frac{\mathcal{d}(\mathbf{u},\mathbf{x})}{h}\Big),
\end{align*}
recalling
\begin{align*}
    \boldsymbol{\Psi}_{t,\mathbf{x}} = \mathbb{E} \left[\mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big) \mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big)^{\top} K_h(D_i(\mathbf{x})) \mathds{1}(D_i(\mathbf{x}) \in \mathcal{I}_t)\right], \quad \mathbf{S}_{t,\mathbf{x}} = \mathbb{E} \left[\mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big) K_h(D_i(\mathbf{x})) Y_i \mathds{1}(\mathbf{X}_i \in \mathcal{A}_t) \right].
\end{align*}
We can check that $\left\lVert\boldsymbol{\Psi}_{t,\mathbf{x}}^{-1}\right\rVert \lesssim 1$, $\left\lVert\mathbf{S}_{t,\mathbf{x}}\right\rVert \lesssim 1$ and
\begin{align*}
    \mathtt{M}_{\mathcal{M}_t} \lesssim h^{-d/2}, \qquad \mathtt{E}_{\mathcal{M}_t} \lesssim h^{-d/2}, \qquad t \in \{0,1\}.
\end{align*}
In what follows, we verify the entropy and total variation properties of $\mathcal{M}$. Using product rule we can verify
\begin{gather*}
    \sup_{\mathbf{u} \in \mathcal{X}}\sup_{\mathbf{x},\mathbf{x}' \in \mathcal{B}}\frac{|\theta_{t,\mathbf{x}}^{\ast}(\mathcal{d}(\mathbf{u},\mathbf{x})) - \theta_{t,\mathbf{x}}^{\ast}(\mathcal{d}(\mathbf{u},\mathbf{x}'))|}{\left\lVert\mathbf{x} - \mathbf{x}'\right\rVert} \lesssim h^{-1}.
\end{gather*}
Define $f_{t,\mathbf{x}}(\cdot) = \frac{h^{-d/2}}{\sqrt{n \Xi_{\mathbf{x},\mathbf{x}}}}\mathbf{e}_{1}^{\top}  \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{r}_p \left(\cdot\right) K(\cdot) (\boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{S}_{t,\mathbf{x}})^{\top} \mathbf{r}_p(\cdot)$. Then,
\begin{align*}
    \mathfrak{K}_t(\mathbf{u};\mathbf{x}) \theta_{t,\mathbf{x}}^{\ast}(\mathcal{d}(\mathbf{u},\mathbf{x})) = h^{-d/2} f_{t,\mathbf{x}} \Big(\frac{d(\mathbf{u},\mathbf{x})}{h}\Big), \qquad \mathbf{u} \in \mathcal{X}, \mathbf{x} \in \mathcal{B}, t \in \{0,1\}.
\end{align*}
Take $\mathcal{M}_t = \{\mathfrak{K}_t(\cdot;\mathbf{x}) \theta_{t,\mathbf{x}}^{\ast}(\mathcal{d}(\cdot,\mathbf{x})): \mathbf{x} \in \mathcal{B}\}$, $t \in \{0,1\}$. For $t \in \{0,1\}$, $f_{t,\mathbf{x}}$ satisfies:
\begin{align*}
 & \textit{(i) boundedness}  && \sup_{\mathbf{x} \in \mathcal{B}} \sup_{\mathbf{u} \in \mathcal{X}} \left|f_{t,\mathbf{x}}(\mathbf{u}) \right| \leq \mathbf{c}, \\
 & \textit{(ii) compact support} && \operatorname{supp}(f_{t,\mathbf{x}}(\cdot)) \subseteq [-\mathbf{c}, \mathbf{c}]^d, \forall \mathbf{x} \in \mathcal{B}, \\
 & \textit{(iii) Lipschitz continuity} && \sup_{\mathbf{x} \in \mathcal{B}} \sup_{\mathbf{u},\mathbf{u}' \in \mathcal{X}}\frac{\left|f_{\mathbf{x}}(\mathbf{u}) - f_{\mathbf{x}}(\mathbf{u}') \right|}{\left\lVert\mathbf{u} - \mathbf{u}'\right\rVert} \leq \mathbf{c} \\
 & \phantom{xx} && \sup_{\mathbf{u} \in \mathbf{X}} \sup_{\mathbf{x},\mathbf{x}' \in \mathcal{B}}\frac{|f_{\mathbf{x}}(\mathbf{u}) - f_{\mathbf{x}'}(\mathbf{u})|}{\left\lVert\mathbf{x} - \mathbf{x}'\right\rVert} \leq \mathbf{c} h^{-1},
\end{align*}
for some constant $\mathbf{c}$ not depending on $n$.
Then, by an argument similar to \citet[Lemma 7]{Cattaneo-Chandak-Jansson-Ma_2024_Bernoulli}, there exists a constant $\mathbf{c}'$ only depending on $\mathbf{c}$ and $d$ that for any $0 \leq \varepsilon \leq 1$,
\begin{align*}
    \sup_{Q} N\left(h^{d/2}\mathcal{H}_t, \left\lVert\cdot\right\rVert_{Q,1}, (2c+1)^{d+1}\varepsilon \right) \leq \mathbf{c}' \varepsilon^{-d-1} + 1,
\end{align*}
where supremum is taken over all finite discrete measures.
Taking a constant envelope function $\mathtt{M}_{\mathcal{M}_t} = (2c +1)^{d+1} h^{-d/2}$, we have for any $0 < \varepsilon \leq 1$,
\begin{align*}
    \sup_{Q} N\left(\mathcal{H}_t, \left\lVert\cdot\right\rVert_{Q,1}, \varepsilon \mathtt{M}_{\mathcal{F}_t}\right) \leq \mathbf{c}' \varepsilon^{-d-1} + 1.
\end{align*}
By Lemma~\ref{sa-lem: VC Class to VC2 Class}, above implies the uniform covering number for $\mathcal{H}_t$ satisfies
\begin{align*}
    \mathtt{N}_{\mathcal{M}_t}(\varepsilon) \leq 4 \mathbf{c}'(\varepsilon/2)^{-d-1}, \qquad 0 < \varepsilon \leq 1.
\end{align*}
Since $\mathcal{M} \subseteq \mathcal{M}_0 + \mathcal{M}_1$, here $+$ denotes the Minkowski sum, with $\mathtt{M}_{\mathcal{M}}$ taken to be $\mathtt{M}_{\mathcal{M}_0} + \mathtt{M}_{\mathcal{M}_1}$, a bound on the uniform covering number of $\mathcal{M}$ can be given by
\begin{align*}
    \mathtt{N}_{\mathcal{M}}(\varepsilon) \leq 16 (\mathbf{c}')^2(\varepsilon/2)^{-2d-2}, \qquad 0 < \varepsilon \leq 1.
\end{align*}
With the assumption that $\mathcal{L}(E_{t,\mathbf{x}}) \leq C h^{d-1}$ for $E_{t,\mathbf{x}} = \{\mathbf{y} \in \mathcal{A}_t: (\mathbf{y} - \mathbf{x})/h \in \operatorname{Supp}(K)\}$ for all $t \in \{0,1\}$, $\mathbf{x} \in \mathcal{B}$, and the fact that $\mathtt{TV}_{\mathcal{M}_t} \lesssim h^{d/2-1}$ for $t \in \{0,1\}$, the same argument as in the paragraph \textbf{Total Variation} in the proof of Theorem~\ref{sa-thm: Strong Approximation} shows
\begin{align*}
    \mathtt{TV}_{\mathcal{M}} \lesssim h^{d/2-1}.
\end{align*}
Now apply Theorem~\ref{sa-thm: Strong Approximation} with $\mathcal{G}$, $\mathcal{M}$ defined in Equation~\eqref{sa-eq: def G and M}, $\mathcal{R} = \{\operatorname{Id}\}$, $\mathcal{S} = \{1\}$, noticing that
\begin{align*}
    (\overline{\operatorname{T}}_{\operatorname{dis}}: \mathbf{x} \in \mathcal{B})
    = (A_n(g,m,r,s): (g,m,r,s) \in \mathcal{F} \times \mathcal{R} \times \mathcal{S}), \qquad \mathcal{F} = \{(g_{\mathbf{x}},m_{\mathbf{x}}): \mathbf{x} \in \mathcal{B}\} \subseteq \mathcal{G} \times \mathcal{M},
\end{align*}
the result then follows.
\qed

\begin{lem}[VC Class to VC2 Class]\label{sa-lem: VC Class to VC2 Class}
    Assume $\mathcal{F}$ is a VC class on a measure space $(\mathcal{X}, \mathcal{B})$: there exists an envelope function $F$ and positive constants $c(\mathcal{F}), d(\mathcal{F})$ such that for all $\varepsilon\in(0,1)$,
    \begin{align*}
        \sup_{Q}N(\mathcal{F}, \left\lVert\cdot\right\rVert_{Q,1}, \varepsilon \left\lVertF\right\rVert_{Q,1}) \leq c(\mathcal{F})\varepsilon^{-d(\mathcal{F})},
    \end{align*}
    where the supremum is taken over all finite discrete measures. Then, $\mathcal{F}$ is also VC2 class: for all $\varepsilon\in(0,1)$,
    \begin{align*}
        \sup_{Q} N(\mathcal{F}, \left\lVert\cdot\right\rVert_{Q,2}, \varepsilon \left\lVertF\right\rVert_{Q,2}) \leq c(\mathcal{F}) (\varepsilon^2/2)^{-d(\mathcal{F})},
    \end{align*}
    where the supremum is taken over all finite discrete measures.
\end{lem}

\noindent\textbf{Proof of Lemma~\ref{sa-lem: VC Class to VC2 Class}}. Let $Q$ be a finite discrete probability measure. Let $f,g \in \mathcal{F}$. Then, $\int |f - g|^2 d Q \leq 2 \int |f - g| |F| d Q$. Define another probability measure $\tilde{Q}(c_k) = F(c_k) Q(c_k) / \left\lVertF\right\rVert_{Q,1}$ on the support of $Q$, denoted by $\{c_1, \ldots, c_k, \ldots\}$. Then,
\begin{align*}
    \int |f - g|^2 d Q
    \leq 2 \left\lVertF\right\rVert_{Q,1} \int |f - g| d \tilde{Q}
    \leq 2 \left\lVertF\right\rVert_{Q,1} \left\lVertf - g\right\rVert_{\tilde{Q},1}.
\end{align*}
Hence, if we take an $\varepsilon^2/2$-net in $(\mathcal{F}, \left\lVert\cdot\right\rVert_{\tilde{Q},1})$ with cardinality no greater than $c(\mathcal{F}) \varepsilon^{- d (\mathcal{F})}$, then for any $f \in \mathcal{F}$, there exists a $g \in \mathcal{F}$ such that $\left\lVertf - g\right\rVert_{\tilde{Q},1} \leq \varepsilon^2/2 \left\lVertF\right\rVert_{\tilde{Q},1}$, and hence
\begin{align*}
    \left\lVertf - g\right\rVert_{Q,2}^2 \leq 2 \varepsilon^2/2 \left\lVertF\right\rVert_{Q,1} \left\lVertF\right\rVert_{\tilde{Q},1} \leq \varepsilon^2 \left\lVertF\right\rVert_{Q,2}^2,
\end{align*}
which gives the result.
\qed

\subsection{Proof of Theorem~\ref{sa-thm: Confidence Bands}}

The result follows from Theorems \ref{sa-thm: Stochastic Linearization} and \ref{sa-thm: Gaussian Strong Approximation: Tstat}, \cite{Chernozhukov-Chetverikov-Kato_2014a_AoS}, and \cite{chernozhuokov2022improved}.
\qed

\subsection{Proof of Theorem~\ref{sa-thm: Strong Approximation}}

Since $A_n$ is the addition of two $M_n$ processes, indexed by $\mathcal{G} \times \mathcal{R}$ and $\mathscr{H} \times \mathcal{S}$ respectively, the Gaussian strong approximation error essentially depends on the \emph{worst case scenario} between $\mathcal{G}$ and $\mathscr{H}$, and between $\mathcal{R}$ and $\mathcal{S}$. Hence (1) taking maximums $\mathtt{E} = \max\{\mathtt{E}_{\mathcal{G}}, \mathtt{E}_{\mathscr{H}}\}$, $\mathtt{M} = \max \{\mathtt{M}_{\mathcal{G}}, \mathtt{M}_{\mathscr{H}}\}$ and $\mathtt{TV} = \max \{\mathtt{TV}_{\mathcal{G}}, \mathtt{TV}_{\mathscr{H}}\}$; (2) noticing that $A_n$ is still indexed by a VC-type class of functions, we can get the claimed result.

For a more rigor proof, we can not apply \citet[Theorem SA.1]{Cattaneo-Yu_2025_AOS} on $(M_n(g,r): g \in \mathcal{G}, r \in \mathcal{R})$ and $(M_n(h,s): h \in \mathscr{H}, s \in \mathcal{S})$ directly, since this ignores the dependence structure between the two empirical processes. However, we can still project the functions onto a Haar basis, and control the \emph{strong approximation error for projected process} and the \emph{projection error} as in the proof of \citet[Theorem SA.1]{Cattaneo-Yu_2025_AOS} and show both errors can be controlled via \emph{worst case scenario} between $\mathcal{G}$ and $\mathscr{H}$, and between $\mathcal{R}$ and $\mathcal{S}$.

\paragraph*{Reductions:} Here we present some reductions to our problem. By the same argument as in Section SA-II.3 (Proofs of Theorem 1) in the supplemental appendix of \cite{Cattaneo-Yu_2025_AOS}, we can show there exists $\mathbf{u}_i, 1 \leq i \leq n$ i.i.d $\mathsf{Uniform}([0,1]^d)$ on a possibly enlarged probability space, such that
\begin{align*}
    f(\mathbf{x}_i) = f(\phi_{\mathcal{G} \cup \mathscr{H}}^{-1}(\mathbf{u}_i)), \qquad \forall f \in \mathcal{G} \cup \mathscr{H}, \forall 1 \leq i \leq n.
\end{align*}
With the help of \citet[Lemma SA.10]{Cattaneo-Yu_2025_AOS}, we can assume w.l.o.g. that $\mathbf{x}_i$'s are i.i.d $\mathsf{Uniform}(\mathcal{X})$ with $\mathcal{X} = [0,1]^d$, and $\phi_{\mathcal{G} \cup \mathscr{H}}: [0,1]^d \to [0,1]^d$ is the identity function. Although we assume $\sup_{\mathbf{x} \in \mathcal{X}}\mathbb{E}[|Y_i|^{2+v}|\mathbf{X}_i = \mathbf{x}] < \infty$, we first present the result under the assumption $\sup_{\mathbf{x} \in \mathcal{X}}\mathbb{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] \leq 2$, which is the same as in \citet[Theorem 2]{Cattaneo-Yu_2025_AOS}. Also in correspondence to the notations in \citet[Theorem 2]{Cattaneo-Yu_2025_AOS}, we set $\alpha = 1$ throughout this proof.

\paragraph*{Cell Constructions and Projections:} The constructions here are the same as those in \cite{Cattaneo-Yu_2025_AOS}, and we present them here for completeness. Let $\mathcal{A}_{M,N}(\mathbb{P},1)= \{\mathcal{C}_{j,k}: 0 \leq k < 2^{M + N - j}, 0 \leq j \leq M + N\}$ be an axis-aligned cylindered
quasi-dyadic expansion of $\mathbb{R}^{d+1}$, with depth $M$ for the main subspace $\mathbb{R}^d$ and depth $N$ for the multiplier subspace $\mathbb{R}$, with respect to $\mathbb{P}$, the joint distribution of $(\mathbf{x}_i, y_i)$ taking values in $\mathbb{R}^d \times \mathbb{R}$, as in \citet[Definition SA.4]{Cattaneo-Yu_2025_AOS}. To see what $\mathcal{A}_{M,N}(\mathbb{P},1)$ is, it can be given by the following iterative partition procedure:
\begin{enumerate}
    \item \emph{Initialization ($q=0$)}: Take $\mathcal{C}_{M + N -q, 0} = \mathcal{X} \times \mathbb{R}$ where $\mathcal{X} = [0,1]^d$.
    \item \emph{Iteration ($q=1,\dots,M$):} Given $\mathcal{C}_{K - l,k}$ for $0 \leq l \leq q-1, 0 \leq k < 2^l$, take $s = (q\mod d) + 1$, and construct $\mathcal{C}_{K - q, 2k} = \mathcal{C}_{K - q + 1,k} \cap \{(\mathbf{x},y) \in [0,1]^d \times \mathbb{R}: \mathbf{e}_s^\top\mathbf{x} \leq c_{K - q + 1,k}\}$ and $\mathcal{C}_{K - q, 2k+1} = \mathcal{C}_{K - q + 1,k} \cap \{(\mathbf{x},y) \in [0,1]^d \times \mathbb{R}: \mathbf{e}_s^\top\mathbf{x} > c_{K - j + 1,k}\}$ such that $\mathbb{P}(\mathcal{C}_{K - q,2k})/\mathbb{P}(\mathcal{C}_{K-q + 1,k}) \in [\frac{1}{1 + \rho}, \frac{\rho}{1 + \rho}]$ for all $0 \leq k < 2^{q-1}$. Continue until $(\mathcal{C}_{N,k}:0 \leq k < 2^{M})$ has been constructed. By construction, for each $0 \leq l < M$, $\mathcal{C}_{N,l} = \mathcal{X}_{0,l} \times \mathcal{Y}_{0,N,0}$, with $\mathcal{Y}_{0,N,0} = \mathbb{R}$.
    \item \emph{Iteration ($q = M + 1, \cdots, M + N$):} Given $\mathcal{C}_{K - l,k}$ for $0 \leq l \leq q-1, 0 \leq k < 2^l$, each $\mathcal{C}_{M + N - q,k}$ can be written as $\mathcal{X}_{0,l} \times \mathcal{Y}_{l,M + N - q,m}$ with $k = 2^{q - M}l + m$. Construct $\mathcal{C}_{M + N - q - 1, 2k} = \mathcal{X}_{0,l} \times \mathcal{Y}_{l,M + N - q-1,2m}$ and $\mathcal{C}_{M + N - q - 1, 2k+1} = \mathcal{X}_{0,l} \times \mathcal{Y}_{l,M + N - q-1,2m+1}$, such that there exists some $\mathfrak{q}_{M + N - q,k} \in \mathbb{R}$ with $\mathcal{Y}_{l,M + N - q-1,2m} = \mathcal{Y}_{l,M + N - q,m} \cap (-\infty, \mathfrak{q}_{M + N - q,k} )$ and $\mathcal{Y}_{l,M + N - q-1,2m+1} = \mathcal{Y}_{l,M + N - q,m} \cap (\mathfrak{q}_{M + N - q,k}, \infty)$,
    $\mathbb{P}(y_i \in \mathcal{Y}_{l,M + N - q-1,2m}|\mathbf{x}_i \in \mathcal{X}_{0,l}) = \mathbb{P}(y_i \in \mathcal{Y}_{l,M + N - q-1,2m+1}|\mathbf{x}_i \in \mathcal{X}_{0,l}) = \frac{1}{2} \mathbb{P}(y_i \in \mathcal{Y}_{l,M + N - q-1,m}|\mathbf{x}_i \in \mathcal{X}_{0,l})$.
\end{enumerate}
Consider the projection $\mathtt{\Pi}_1(\mathcal{A}_{M,n}(\mathbb{P},1))$ given in Equation (\textcolor{blue}{SA-7}) in \cite{Cattaneo-Yu_2025_AOS}, noticing that $\mathcal{A}_{M,N}(\mathbb{P},1)$ is one special instance of $\mathcal{C}_{M,N}(\mathbb{P},\rho)$. That is, define $e_{j,k} = \mathds{1}_{\mathcal{C}_{j,k}}$ and $\widetilde{e}_{j,k} = e_{j-1,2k} - e_{j-1,2k+1}$,
\begin{align}\label{sa-eq: projmult}
    \mathtt{\Pi}_1(\mathcal{C}_{M,N}(\mathbb{P},\rho))[g,r] = \gamma_{M+N,0}(g,r)e_{M+N,0} + \sum_{1 \leq j \leq M + N} \sum_{0 \leq k < 2^{M + N - j}} \widetilde{\gamma}_{j,k}(g,r) \widetilde{e}_{j,k},
\end{align}
where $e_{j,k} = \mathds{1}(\mathcal{C}_{j,k})$ and $\widetilde{e}_{j,k} = \mathds{1}(\mathcal{C}_{j-1,2k}) - \mathds{1}(\mathcal{C}_{j-1,2k+1})$, and
\begin{align*}
    \gamma_{j,k}(g,r) & =
    \begin{cases}
        \mathbb{E}[g(X)r(Y)|X \in \mathcal{X}_{j-N,k}], & \text{ if } N \leq j \leq M+N, \\
        \mathbb{E}[g(X)|X \in \mathcal{X}_{0,l}]\cdot \mathbb{E}[r(Y)|X \in \mathcal{X}_{0,l}, Y \in \mathcal{Y}_{l,0,m}], & \text{ if } j < N, k = 2^{N - j}l + m,
    \end{cases}
\end{align*}
and $\widetilde{\gamma}_{j,k}(g,r) = \gamma_{j-1,2k}(g,r) - \gamma_{j-1,2k+1}(g,r)$. We will use $\mathtt{\Pi}_{1}$ as a shorthand for $\mathtt{\Pi}_1(\mathcal{C}_{M,N}(\mathbb{P},\rho))$.

For simplicity, we denote
$\mathtt{\Pi}_1(\mathcal{A}_{M,n}(\mathbb{P},1))$ by $\mathtt{\Pi}_1$ instead. Now define the projected empirical process
\begin{align*}
    \mathtt{\Pi_1}A_n(g,h,r,s) = \mathtt{\Pi_1} M_n(g,r) + \mathtt{\Pi_1} M_n(h,s), \qquad g \in \mathcal{G}, h \in \mathscr{H}, r \in \mathcal{R}, s \in \mathcal{S},
\end{align*}
where $\mathtt{\Pi_1} M_n(g,r)$ and $\mathtt{\Pi_1} M_n(h,s)$ are given in Equation (\textcolor{blue}{SA-10}) in \cite{Cattaneo-Yu_2025_AOS}, that is,
\begin{align*}
    \mathtt{\Pi_1} M_n(g,r) & = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \big(\mathtt{\Pi_1}[g,r](\mathbf{x}_i, y_i)
    - \mathbb{E}[\mathtt{\Pi_1}[g,r](\mathbf{x}_i, y_i)]\big), \\
    \mathtt{\Pi_1} M_n(h,s) & = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \big(\mathtt{\Pi_1}[h,s](\mathbf{x}_i, y_i)
    - \mathbb{E}[\mathtt{\Pi_1}[h,s](\mathbf{x}_i, y_i)]\big).
\end{align*}

\paragraph*{Construction of Gaussian Process}

Suppose $(\widetilde{\xi}_{j,k}:0 \leq k < 2^{M + N - j},1 \leq j \leq M + N)$ are i.i.d. standard Gaussian random variables. Take $F_{(j,k),m}$ to be the cumulative distribution function of $(S_{j,k} - m p_{j,k})/\sqrt{m p_{j,k}(1 - p_{j,k})}$, where $p_{j,k} = \mathbb{P}(\mathcal{C}_{j-1,2k})/\mathbb{P}(\mathcal{C}_{j,k})$ and $S_{j,k}$ is a $\operatorname{Bin}(m, p_{j,k})$ random variable, and $G_{(j,k), m}(t) = \sup \{x: F_{(j,k),m}(x) \leq t\}$. We define $U_{j,k}, \widetilde{U}_{j,k}$'s via the following iterative scheme:
\begin{enumerate}
    \item \emph{Initialization:} Take $U_{M + N, 0} = n$.
    \item \emph{Iteration:} Suppose we've defined $U_{l,k}$ for $j < l \leq M + N, 0 \leq k < 2^{M + N - l}$, then solve for $U_{j,k}$'s s.t.
\begin{align*}
    & \widetilde{U}_{j,k} = \sqrt{U_{j,k}p_{j,k}(1 - p_{j,k})} G_{(j,k), U_{j,k}} \circ \Phi(\widetilde{\xi}_{j,k}), \\
    & \widetilde{U}_{j,k} = (1 - p_{j,k}) U_{j-1, 2k} - p_{j,k} U_{j-1, 2k+1} = U_{j-1, 2k} - p_{j,k} U_{j,k}, \\
    & U_{j-1, 2k} + U_{j-1, 2k+1} = U_{j,k}, \quad 0 \leq  k < 2^{M + N - j}.
\end{align*}
Continue till we have defined $U_{0,k}$ for $0 \leq k < 2^{M + N}$.
\end{enumerate}
Then, $\{U_{j,k}: 0 \leq j \leq K, 0 \leq k < 2^{M + N - j}\}$ have the same joint distribution as $\{\sum_{i = 1}^n e_{j,k}(\mathbf{x}_i, y_i): 0 \leq j \leq K, 0 \leq k < 2^{M + N - j}\}$. By Vorob'ev–Berkes–Philipp theorem \citep[Theorem 1.31]{dudley2014uniform}, $\{\widetilde{\xi}_{j,k}: 0 \leq k < 2^{M + N -j}, 1 \leq j \leq M + N\}$ can be constructed on a possibly enlarged probability space such that the previously constructed $U_{j,k}$ satisfies $U_{j,k} = \sum_{i = 1}^n e_{j,k}(\mathbf{x}_i)$ almost surely for all $0 \leq j \leq M + N, 0 \leq k < 2^{M + N - j}$. We will show $\widetilde{\xi}_{j,k}$'s can be given as a Brownian bridge indexed by $\widetilde{e}_{j,k}$'s.

Since all of $\mathcal{G}$, $\mathscr{H}$, $\mathcal{R}$ and $\mathcal{S}$ are VC-type, we can show $\mathcal{G} \times \mathscr{H} + \mathcal{R} \times \mathcal{S}$ is also VC-type, here $+$ is the Minkowski sum. Hence $\mathcal{F} = \mathcal{G} \times \mathscr{H} + \mathcal{R} \times \mathcal{S} \cup \mathtt{\Pi}_1[G \times \mathscr{H} + \mathcal{R} \times \mathcal{S}]$ is pre-Gaussian.

Then, by Skorohod Embedding lemma \citep[Lemma 3.35]{dudley2014uniform}, on a possibly enlarged probability space, we can construct a Brownian bridge $(Z_n(f): f \in \mathcal{F})$ that satisfies
\begin{align*}
    \widetilde{\xi}_{j,k} = \frac{\mathbb{P}(\mathcal{C}_{j,k})}{\sqrt{\mathbb{P}(\mathcal{C}_{j-1,2k})\mathbb{P}(\mathcal{C}_{j-1,2k+1})}} Z_n(\widetilde{e}_{j,k}),
\end{align*}
for $0 \leq k < 2^{M + N - j},1 \leq j \leq M + N$. Moreover, call
\begin{align*}
    V_{j,k} = \sqrt{n}Z_n(e_{j,k}), \qquad
    \widetilde{V}_{j,k} = \sqrt{n}Z_n(\widetilde{e}_{j,k}),\qquad
    \widetilde{\xi}_{j,k} = \frac{\mathbb{P}(\mathcal{C}_{j,k})}{\sqrt{n\mathbb{P}(\mathcal{C}_{j-1,2k})\mathbb{P}(\mathcal{C}_{j-1,2k+1})}} \widetilde{V}_{j,k}.
\end{align*}
for $0 \leq k < 2^{K - j},1 \leq j \leq K$. We have for $g \in \mathcal{G}, h \in \mathscr{H}, r \in \mathcal{R}, s \in \mathcal{S}$,
\begin{gather*}
    \sqrt{n}\mathtt{\Pi}_1 A_n(g,h,r,s) = \sum_{j = 1}^{M + N} \sum_{0 \leq k < 2^{M + N - j}} (\widetilde{\gamma}_{j,k}[g,r] + \widetilde{\gamma}_{j,k}[h,s]) \widetilde{U}_{j,k}, \\
    \sqrt{n}\mathtt{\Pi}_1 Z_n(g,h,r,s) = \sum_{j = 1}^{M + N} \sum_{0 \leq k < 2^{M + N - j}} (\widetilde{\gamma}_{j,k}[g,r] + \widetilde{\gamma}_{j,k}[h,s]) \widetilde{V}_{j,k}.
\end{gather*}

\paragraph*{Decomposition}
Fix one $(g,h,r,s) \in \mathcal{G} \times \mathscr{H} \times \mathcal{R} \times \mathcal{S}$, we decompose by
\begin{align*}
& A_n(g,h,r,s) - Z_n(g,h,r,s) \\
& = \underbrace{\mathtt{\Pi}_1 A_n(g,h,r,s) - \mathtt{\Pi}_1 Z_n(g,h,r,s)}_\text{strong approximation (SA) error for projected} + \underbrace{A_n(g,h,r,s) - \mathtt{\Pi}_1 A_n(g,h,r,s) + \mathtt{\Pi}_1 Z_n(g,h,r,s) - Z_n(g,h,r,s)}_\text{projection error}.
\end{align*}
\paragraph*{SA error for Projected Process} The strong approximation error essentially depends on the Hilbertian pseudo norm
\begin{align*}
    \sum_{j = 1}^{M + N} \sum_{0 \leq k < 2^{M + N - j}} (\widetilde{\gamma}_{j,k}[g,r] + \widetilde{\gamma}_{j,k}[h,s])^2
    \leq 2 \sum_{j = 1}^{M + N} \sum_{0 \leq k < 2^{M + N - j}} (\widetilde{\gamma}_{j,k}[g,r])^2
    + 2 \sum_{j = 1}^{M + N} \sum_{0 \leq k < 2^{M + N - j}} (\widetilde{\gamma}_{j,k}[h,s])^2.
\end{align*}
Hence, \citet[Lemma SA.19]{Cattaneo-Yu_2025_AOS} gives with probability at least $1 - 2 e^{-t}$,
\begin{align*}
    & |\mathtt{\Pi}_{1} A_n(g,h,r,s)- \mathtt{\Pi}_{1} Z_n(g,h,r,s)|
       \leq C_1 C_{\alpha} \sqrt{\frac{N^{2\alpha + 1} 2^{M} \mathtt{E} \mathtt{M}}{n} t}
        + C_1 C_{\alpha}\sqrt{\frac{(\lVert \mathtt{\Pi}_1[g,r] \rVert_{\infty} + \lVert \mathtt{\Pi}_1[h,s] \rVert_{\infty})^2(M + N)}{n}} t,
\end{align*}
where $C_1 > 0$ is a universal constant and $C_{\alpha} = 1 + (2 \alpha)^{\alpha/2}$.
\paragraph*{Projection Error} For the projection error, we use the simple observation that
\begin{align*}
    |A_n(g,h,r,s) - \mathtt{\Pi}_1 A_n(g,h,r,s)|
    & \leq |M_n(g,r) - \mathtt{\Pi}_1 M_n(g,r)| + |M_n(h,s) - \mathtt{\Pi}_1 M_n(h,s)|,
\end{align*}
and \citet[Lemma SA.23]{Cattaneo-Yu_2025_AOS} to get for all $t > N$,
\begin{gather*}
        \mathbb{P}\Big[|A_n(g,h,r,s)- \mathtt{\Pi}_{1} A_n(g,h,r,s)|
           > C_2 \sqrt{C_{2\alpha}}\sqrt{N^2 \mathtt{V} + 2^{-N} \mathtt{M}^2} t^{\alpha + \frac{1}{2}} + C_2 C_{\alpha} \frac{\mathtt{M}}{\sqrt{n}}t^{\alpha + 1}\Big]
        \leq 4 n e^{-t},\\
        \mathbb{P}\Big[|Z_n(g,h,r,s) - \mathtt{\Pi}_{1} Z_n(g,h,r,s) |
           > C_2 \sqrt{C_{2\alpha}} \sqrt{N^2 \mathtt{V} + C_2 C_{\alpha} 2^{-N} \mathtt{M}^2} t^{\frac{1}{2}} + C_2 C_{\alpha} \frac{\mathtt{M}}{\sqrt{n}}t \Big]
        \leq 4  n e^{-t},
\end{gather*}
where $C_{\alpha} = 1 + (2 \alpha)^{\frac{\alpha}{2}}$ and $C_{2 \alpha} = 1 + (4 \alpha)^{\alpha}$ and $C_2$ is a constant that only depends on the distribution of $(\mathbf{x}_1,y_1)$, with
\begin{align*}
    \mathtt{V} = \min\{2 \mathtt{M}, \sqrt{d} \mathtt{L} 2^{-M/d} \} 2^{-M/d} \mathtt{TV}_{\mathscr{H}}.
\end{align*}
\paragraph*{Uniform SA Error:} Since all of $\mathcal{G}$, $\mathscr{H}$, $\mathcal{R}$ and $\mathcal{S}$ are VC-type class, from a union bound argument and the same control over fluctuation error as in \citet[Lemma SA.18]{Cattaneo-Yu_2025_AOS}, denoting $\mathcal{F} = \mathcal{G} \times \mathscr{H} \times \mathcal{R} \times \mathcal{S}$, we get for all $t > 0$ and $0<\delta<1$,
    \begin{align*}
        \mathbb{P}\big[\left\lVertA_n - A_n \circ\pi_{\mathcal{F}_{\delta}}\right\rVert_{\mathcal{F}} + \left\lVertZ_n - Z_n\circ\pi_{\mathcal{F}_\delta}\right\rVert_{\mathcal{F}}> C_1 C_{\alpha} \mathsf{F}_n(t,\delta)\big] &\leq \exp(-t),
    \end{align*}
    where $C_{\alpha} = 1 + (2 \alpha)^{\frac{\alpha}{2}}$  and
    \begin{align*}
        \mathtt{F}_n(t,\delta) =  J(\delta) \mathtt{M} + \frac{(\log n)^{\alpha/2} \mathtt{M}  J^2(\delta)}{\delta^2 \sqrt{n}} + \frac{\mathtt{M}}{\sqrt{n}} t + (\log n)^{\alpha} \frac{\mathtt{M}}{\sqrt{n}} t^{\alpha}.
    \end{align*}
where
\begin{align*}
    J(\delta) & = 3 \delta \big(\sqrt{\mathtt{d}_{\mathcal{G}} \log(\frac{2 \mathtt{c}_{\mathcal{G}}}{\delta})} + \sqrt{\mathtt{d}_{\mathscr{H}} \log(\frac{2 \mathtt{c}_{\mathscr{H}}}{\delta})} + \sqrt{\mathtt{d}_{\mathcal{R}} \log(\frac{2 \mathtt{c}_{\mathcal{R}}}{\delta})} + \sqrt{\mathtt{d}_{\mathcal{S}} \log(\frac{2 \mathtt{c}_{\mathcal{S}}}{\delta})}\big) \\
    & \lesssim \sqrt{\mathtt{d} \log (\mathtt{c} /\delta)},
\end{align*}
recalling $\mathtt{c} = \mathtt{c}_{\mathcal{G}, \mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}} + \mathtt{c}_{\mathscr{H},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}} + \mathtt{c}_{\mathcal{R},\mathcal{Y}} + \mathtt{c}_{\mathcal{S},\mathcal{Y}} + \mathtt{k}$, $\mathtt{d} = \mathtt{d}_{\mathcal{G},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}} \mathtt{d}_{\mathscr{H},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}} \mathtt{d}_{\mathcal{R},\mathcal{Y}} \mathtt{d}_{\mathcal{S},\mathcal{Y}} \mathtt{k}$. Choosing the optimal $M^{\ast}$, $N^{\ast}$ gives $\mathbb{P}\big[\left\lVertA_n - Z_n^A\right\rVert_{\mathcal{F}} > C_1 \mathtt{v} \mathsf{T}_n(t)\big] \leq C_2 e^{-t}$ for all $t > 0$, where
    \begin{align*}
        \mathsf{T}_n(t) = \min_{\delta \in (0,1)}\{\mathsf{A}_n(t,\delta) + \mathsf{F}_n(t,\delta)\},
    \end{align*}
with
    \begin{align*}
        \mathsf{A}_n(t,\delta)
        &= \sqrt{d} \min \Big\{ \Big( \frac{\mathtt{c}_1^{d} \mathtt{E} \mathtt{TV}^{d} \mathtt{M}^{d+1}}{n}\Big)^{\frac{1}{2(d+1)}}, \Big(\frac{\mathtt{c}_1^{d} \mathtt{c}_2^{d}\mathtt{E}^2 \mathtt{M}^2 \mathtt{TV}^{d}   \mathtt{L}^{d}}{n^2} \Big)^{\frac{1}{2(d+2)}} \Big\} (t + \log(n \mathtt{N}(\delta) N^{\ast}))^{\alpha + 1} \\
        &\qquad + \sqrt{\frac{\mathtt{M}^2(M^{\ast} + N^{\ast})}{n}} (\log n)^{\alpha} (t + \log(n \mathtt{N}(\delta) N^{\ast}))^{\alpha + 1}, \\
        \mathsf{F}_n(t,\delta)
        &= J(\delta) \mathtt{M} + \frac{(\log n)^{\alpha/2} \mathtt{M} J^2(\delta)}{\delta^2 \sqrt{n}} + \frac{\mathtt{M}}{\sqrt{n}} \sqrt{t} + (\log n)^{\alpha}\frac{\mathtt{M}}{\sqrt{n}} t^{\alpha},
    \end{align*}
    where
    \begin{align*}
        \mathscr{V}_{\mathcal{R}} &= \{\theta(\cdot,r): r \in \mathcal{R}\}, \\
        \mathtt{N}(\delta) & = \mathtt{N}_{\mathcal{G},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}}(\delta/2, \mathtt{M}_{\mathcal{G}, \mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}}) \mathtt{N}_{\mathscr{H},\mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}}(\delta/2, \mathtt{M}_{\mathscr{H}, \mathcal{Q}_{\mathcal{G} \cup \mathscr{H}}}) \mathtt{N}_{\mathcal{R},\mathcal{Y}}(\delta/2, M_{\mathcal{R}}) \mathtt{N}_{\mathcal{S},\mathcal{Y}}(\delta/2, M_{\mathcal{S},\mathcal{Y}}), \\
        J(\delta) & = 2J_{\mathcal{Q}_{\mathcal{G} \cup  \mathscr{H}}}(\mathcal{G},\mathtt{M}_{\mathcal{G},\mathcal{Q}_{\mathcal{G} \cup  \mathscr{H}}},\delta/2) + 2J_{\mathcal{Q}_{\mathcal{G} \cup  \mathscr{H}}}(\mathscr{H},\mathtt{M}_{\mathscr{H},\mathcal{Q}_{\mathcal{G} \cup  \mathscr{H}}},\delta/2) + 2 J_{\mathcal{Y}} (\mathcal{R},M_{\mathcal{R},\mathcal{Y}}, \delta/2) + 2 J_{\mathcal{Y}} (\mathcal{S},M_{\mathcal{S},\mathcal{Y}}, \delta/2),\\
        M^{\ast} &=  \Big\lfloor\log_2\min \Big\{\Big(\frac{\mathtt{c}_1 n \mathtt{TV}}{\mathtt{E}}\Big)^{\frac{d}{d+1}}, \Big( \frac{\mathtt{c}_1 \mathtt{c}_2 n \mathtt{L} \mathtt{TV} }{\mathtt{E} \mathtt{M}}\Big)^{\frac{d}{d+2}} \Big\} \Big\rfloor, \\
        N^{\ast} & = \Big\lceil \log_2 \max \Big\{\Big(\frac{n \mathtt{M}^{d+1}}{\mathtt{c}_1^d \mathtt{E} \mathtt{TV}^d}\Big)^{\frac{1}{d+1}}, \Big( \frac{n^2 \mathtt{M}^{2d+2}}{\mathtt{c}_1^d \mathtt{c}_2^d \mathtt{TV}^d \mathtt{L}^d \mathtt{E}^2}\Big)^{\frac{1}{d+2}}\Big\}\Big\rceil.
    \end{align*}


\paragraph*{Truncation Argument for $y_i$'s with Finite Moments} The above result is derived under the assumption that $\sup_{\mathbf{x} \in \mathcal{X}} \mathbb{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] < \infty$. For the result under the condition $\sup_{\mathbf{x} \in \mathcal{X}} \mathbb{E}[|y_i|^{2+v}|\mathbf{x}_i = \mathbf{x}] < \infty$, we can use the same truncation argument as in \cite[Theorem SA-11 in the supplemental material]{Cattaneo-Titiunik-Yu_2026_BDD-Location} and the VC-type conditions for $\mathcal{G}, \mathscr{H}, \mathcal{R}, \mathcal{S}$ to get the stated conclusions.
\qed


\subsection{Proof of Theorem 2}\label{sa-sec: Proof of Theorem 2}

\begin{center}
    \textbf{Part I: Upper Bound}.
\end{center}
The proof is essentially the proof for Lemma~\ref{sa-lem: Uniform Bias: Minimal Guarantee} with the data generating process ranging over $\mathcal{P}$. By Theorem~\ref{sa-thm: Identification} and Equation~\eqref{sa-eq: best fit}, we have
\begin{align*}
    & \sup_{\mathbb{P} \in \mathcal{P}} \sup_{\mathbf{x} \in \mathcal{B}}|\mathfrak{B}_{n,t}(\mathbf{x})| \\
    & = \sup_{\mathbb{P} \in \mathcal{P}} \sup_{\mathbf{x} \in \mathcal{B}} \Big|\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbf{S}_{t,\mathbf{x}} - \mu_t(\mathbf{x}) \Big| \\
    & = \sup_{\mathbb{P} \in \mathcal{P}} \sup_{\mathbf{x} \in \mathcal{B}} \Big|\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbb{E} \Big[\mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big) K_h(D_i(\mathbf{x})) \mathbf{r}_p(D_i(\mathbf{x}))^{\top} (\mu_t(\mathbf{X}_i) - \mu_t(\mathbf{x}),0,\cdots,0)) \mathds{1}(\mathbf{X}_i \in \mathcal{A}_t) \Big] \Big| \\
    & \lesssim \sup_{\mathbb{P} \in \mathcal{P}} \sup_{\mathbf{x} \in \mathcal{B}} \sup_{\mathbf{z} \in \mathcal{X}} \Big|\mathbf{e}_1^{\top} \boldsymbol{\Psi}_{t,\mathbf{x}}^{-1} \mathbb{E} \Big[\mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big) K_h(D_i(\mathbf{x})) \mathbf{r}_p\Big(\frac{D_i(\mathbf{x})}{h}\Big)^{\top} \Big| \\
    & \qquad \cdot \sup_{\mathbb{P} \in \mathcal{P}} \sup_{\mathbf{x} \in \mathcal{B}} \sup_{\mathbf{z} \in \mathcal{X}} |\mu_t(\mathbf{x}) - \mu_t(\mathbf{z})| \mathds{1}(K_h(\mathcal{d}(\mathbf{z},\mathbf{x})) > 0) \\
    & \lesssim h.
\end{align*}

\begin{center}
    \textbf{Part II: Lower Bound}.
\end{center}
The lower bound is proved by considering the following data generating process. Suppose $\mathbf{X}_i \thicksim \mathsf{Uniform}([-2,2]^2)$, and $\mu_0(x_1, x_2) = 0$ and $\mu_1(x_1,x_2) = x_2$ for all $(x_1,x_2) \in \mathcal{X} = [-2,2]^2$. Suppose $Y_i(0)\thicksim \mathsf{Normal}(\mu_0(\mathbf{X}_i),1)$ and $Y_i(1) \thicksim \mathsf{Normal}(\mu_1(\mathbf{X}_i),1)$. Define the treatment and control region by $\mathcal{A}_1 = \{(x,y) \in \mathcal{X}: x \geq 0, y \geq 0\}$, $\mathcal{A}_0 = \mathcal{X} / \mathcal{A}_1$, $\mathcal{B} = \{(x,y) \in \mathbb{R}: 0 \leq x \leq 2, y = 0 \text{ or } x = 0, 0 \leq y \leq 2\}$. Suppose $Y_i = \mathds{1}(\mathbf{X}_i \in \mathcal{A}_0)Y_i(0) + \mathds{1}(\mathbf{X}_i \in \mathcal{A}_1)Y_i(1)$. Suppose we choose $\mathcal{d}$ to be the Euclidean distance and $D_i(\mathbf{x}) = \left\lVert\mathbf{X}_i - \mathbf{x}\right\rVert$. In this case, although the underlying conditional mean functions $\mu_t$, $t \in \{0,1\}$ are smooth, the conditional mean given distance $\theta_{t,\mathbf{x}}$ may not even be differentiable. In this example,
\begin{align*}
    \theta_{1,(s,0)}(r) =
    \begin{cases}
        \frac{2}{\pi r}, & \text{if } 0 \leq r \leq s,\\
        \frac{r + s}{\pi - \arccos(s/r)}, & \text{if } r > s.
    \end{cases}
\end{align*}

Figure~\ref{sa-fig: cond mean 1d} plots $r\mapsto\theta_{1,(3/4,0)}(r)$ with the notation $\mathbf{x}_s = (s,0)$.

\begin{figure}
    \centering
    \includegraphics[width=0.5\linewidth]{figures/cond_mean.png}
    \caption{Conditional Mean Given Distance with One Kink}
    \label{sa-fig: cond mean 1d}
\end{figure}
Under this data generating process, we can show
\begin{align*}
   \inf_{0 < h < 1} \sup_{\mathbf{x} \in \mathcal{B}} \frac{|\mathfrak{B}_{n,1}(\mathbf{x}) - \mathfrak{B}_{n,0}(\mathbf{x})|}{h} > 0.
\end{align*}

The proof proceeds in two steps. First, we show a scaling property of the asymptotic bias under our example, which gives a reduction to fixed-$h$ bias calculation. Second, we prove the lower bound via the reduction from previous step.

\subsubsection*{Step 1: A Scaling Property} Let $0 < h < 1, 0 < s < 1, 0 < C < 1$. Define $h' = Ch$ and $s' = Cs$. Here $C$ is the scaling factor and denote $\mathbf{x}_s = (s,0)$ and $\mathbf{x}_{s'} = (s',0)$. Denote bias for $\mathbf{x}_{s'}$ under bandwidth $h'$ to be
\begin{align}\label{sa-eq: bias defn 1d}
    \nonumber \operatorname{bias}_{n,1}(h',s') & = \mathbf{e}_1^{\top} \mathbb{E} \bigg[\mathbf{r}_p \left(\frac{D_i((s',0))}{h'}\right) \mathbf{r}_p \left(\frac{D_i((s',0))}{h'}\right)^{\top} K_{h'}(D_i((s',0))) \mathds{1}(\mathbf{X}_i \in \mathcal{A}_1)\bigg]^{-1} \\
    & \mathbb{E} \left[\mathbf{r}_p \left(\frac{D_i((s',0))}{h'}\right)K_{h'}\left(D_i((s',0))\right)\left(\mu_1(\mathbf{X}_i - (s',0))\right)\mathds{1}(\mathbf{X}_i \in \mathcal{A}_1)\right],
\end{align}
where we have used the fact that $\mu_1$ is linear in our example, hence $\mu_1(\mathbf{X}_i) - \mu_1((s',0)) = \mu_1(\mathbf{X}_i - (s',0))$. We reserve the notation $\mathfrak{B}_{n,t}$, $t = 0,1$, to the bias when bandwidth is $h$, that is,
\begin{align*}
    \mathfrak{B}_{n,t}(\mathbf{x}_s) \equiv \operatorname{bias}_{n,t}(h,s), \qquad h \in (0,1), s \in (0,1), t = 0,1.
\end{align*}

Inspecting each element of the last vector, for all $l \in \mathbb{N}$,
\begin{align*}
    & \mathbb{E} \bigg[\left(\frac{\left\lVert\mathbf{X}_i - (s',0)\right\rVert}{h'}\right)^l K_{h'}\left(\left\lVert\mathbf{X}_i - (s',0)\right\rVert\right)\left( \mu_1(\mathbf{X}_i - (s',0))\right)\mathds{1}(\mathbf{X}_i \in \mathcal{A}_1)\bigg] \\
    & = \int_0^2 \int_0^2 \left(\frac{1}{h'}\right)^2 \left(\frac{\left\lVert(u' - s', v')\right\rVert}{h'}\right)^{l} k \left(\frac{\left\lVert(u' - s', v')\right\rVert}{h'}\right) \mu_1 \left( (u', v') - (s',0)\right) \frac{1}{4} d u' d v' \\
    \stackrel{(1)}{=} & \int_{0}^{2/C} \int_{0}^{2/C} \left(\frac{1}{C h}\right)^2 \left(\frac{\left\lVert(C u - C s, C v)\right\rVert}{C h}\right)^{l} k \left(\frac{\left\lVert(C u - Cs, C v) \right\rVert}{C h}\right) \mu_1 \left(C(u-s,v)\right)  \frac{C^2}{4} d u d v \\
    & = \int_{0}^{2/C} \int_{0}^{2/C} \left(\frac{1}{h}\right)^2  \left(\frac{\left\lVert(u - s, v)\right\rVert}{h}\right)^{l} k \left(\frac{\left\lVert(u-s, v) \right\rVert}{h}\right) C \mu_1 \left( (u -s, v) \right) \frac{1}{4} d u d v \\
    \stackrel{(2)}{=} &  \int_{0}^{2} \int_{0}^{2} \left(\frac{1}{h}\right)^2 \left(\frac{\left\lVert(u - s, v)\right\rVert}{h}\right)^{l} k \left(\frac{\left\lVert(u, v) - (s,0)\right\rVert}{h}\right) C \mu_1 \left( (u, v) - (s,0)\right) \frac{1}{4} d u d v \\
    & = C \mathbb{E}\bigg[\left(\frac{\left\lVert\mathbf{X}_i - (s,0)\right\rVert}{h}\right)^{l} K_{h}\left(\left\lVert\mathbf{X}_i - (s,0)\right\rVert\right) \mu_1(\mathbf{X}_i - (s,0))\mathds{1}(\mathbf{X}_i \in \mathcal{A}_1)\bigg],
\end{align*}
where in (1) we have used a change of variable $(u,v) = \frac{1}{C} (u', v')$, and (2) holds since $k \left( \frac{\left\lVert\cdot - (s,0)\right\rVert}{h}\right)$ is supported in $(s,0) + h B(0,1)$, which is contained in $[0,2] \times [0,2] \subseteq [0,2/C] \times [0,2/C]$ for all $0 < h < 1$, $0 < s < 1$, $0 < C < 1$. This means
\begin{gather*}
    \mathbb{E} \left[\mathbf{r}_p \left(\frac{D_i((s',0))}{h'}\right)K_{h'}\left(D_i((s',0))\right)\left(\mu_1(\mathbf{X}_i - (s',0))\right)\mathds{1}(\mathbf{X}_i \in \mathcal{A}_1)\right] \\
    = C \mathbb{E} \left[\mathbf{r}_p \left(\frac{D_i((s,0))}{h}\right)K_{h}\left(D_i((s,0))\right)\left(\mu_1(\mathbf{X}_i - (s,0))\right)\mathds{1}(\mathbf{X}_i \in \mathcal{A}_1)\right].
\end{gather*}
Similarly, for all $l \in \mathbb{N}$ and $0 < h < 1$, $0 < s < 1$, $0 < C < 1$,
\begin{align*}
    & \mathbb{E}\bigg[\left(\frac{D_i((s',0)))}{h'}\right)^l K_{h'}\left(D_i((s',0))\right)\mathds{1}(\mathbf{X}_i \in \mathcal{A}_1)\bigg] = \mathbb{E}\bigg[\left(\frac{D_i((s,0))}{h}\right)^l K_{h}\left(D_i((s,0))\right)\mathds{1}(\mathbf{X}_i \in \mathcal{A}_1)\bigg],
\end{align*}
implying
\begin{gather*}
    \mathbb{E} \bigg[\mathbf{r}_p \left(\frac{D_i((s',0))}{h'}\right) \mathbf{r}_p \left(\frac{D_i((s',0))}{h'}\right)^{\top} K_{h'}(D_i((s',0))) \mathds{1}(\mathbf{X}_i \in \mathcal{A}_1)\bigg] \\
    =  \mathbb{E} \bigg[\mathbf{r}_p \left(\frac{D_i((s,0))}{h}\right) \mathbf{r}_p \left(\frac{D_i((s,0))}{h}\right)^{\top} K_{h}(D_i((s,0))) \mathds{1}(\mathbf{X}_i \in \mathcal{A}_1)\bigg].
\end{gather*}
It then follows that for all $0 < h < 1$, $0 < s < 1$, $0 < C < 1$,
\begin{align*}
   \operatorname{bias}_{n,1}(h',s') = C\operatorname{bias}_{n,1}(h,s).
\end{align*}
Moreover, for all $0< h< 1$, $0 < s < h$,
\begin{align}\label{sa-eq: scaling}
    \mathfrak{B}_{n,1}(\mathbf{x}_s) =\operatorname{bias}_{n,1}(h,s) = h\operatorname{bias}_{n,1}\left(1, \frac{s}{h}\right).
\end{align}
Since $\mu_0 \equiv 0$, it is easy to check that
\begin{align*}
    \mathfrak{B}_{n,0}(\mathbf{x}_s) = \operatorname{bias}_{n,0}(h,s) \equiv 0, \qquad 0 < h < 1, 0 < s < h.
\end{align*}

\subsubsection*{Step 2: Lower Bound on Bias} Now we want to show $\sup_{0 \leq s \leq 1}|\operatorname{bias}_{n,1}(1,s) - \operatorname{bias}_{n,0}(1,s)| > 0$.
By Equation~\eqref{sa-eq: bias defn 1d},
\begin{gather*}
   \operatorname{bias}_{n,1}(1,s) - \operatorname{bias}_{n,0}(1,s)  =  \mathbf{e}_1^{\top} \boldsymbol{\Psi}_s^{-1}\mathbf{S}_s - \mu_1(\mathbf{x}_s) - 0 = \mathbf{e}_1^{\top} \boldsymbol{\Psi}_s^{-1}\mathbf{S}_s,\\
    \boldsymbol{\Psi}_s = \mathbb{E}\left[\mathbf{r}_p \left(D_i(\mathbf{x}_s)\right) \mathbf{r}_p \left(D_i(\mathbf{x}_s)\right)^{\top} K(D_i(\mathbf{x}_s)) \mathds{1}(\mathbf{X}_i \in \mathcal{A}_1)\right] , \\
    \mathbf{S}_{s} =  \mathbb{E} \left[\mathbf{r}_p \left(D_i(\mathbf{x}_s)\right) K(D_i(\mathbf{x}_s)) \mu_1(\mathbf{X}_i)\mathds{1}(\mathbf{X}_i \in \mathcal{A}_1)\right].
\end{gather*}
Changing to polar coordinates, we have
\begin{gather*}
    \boldsymbol{\Psi}_s = \int_{0}^{\infty} \int_{\Theta_s(r)}^{\pi} \mathbf{r}_p(r) \mathbf{r}_p(r)^{\top} K(r) r d \theta d r, \\
    \mathbf{S}_s = \int_{0}^{\infty} \int_{\Theta_s(r)}^{\pi} \mathbf{r}_p(r)  K(r) r \sin(\theta) r d \theta d r,
\end{gather*}
with
\begin{align*}
    \Theta_s(r) =
    \begin{cases}
        0, & \text{if } 0 \leq r \leq s,\\
        \arccos(s/r), & \text{if } r > s.
    \end{cases}
\end{align*}
For notation simplicity, denote
\begin{gather*}
    \mathbf{A}(s) =  \int_{0}^{\infty} \int_{\Theta_s(u)}^{\pi} \mathbf{r}_p(u) \mathbf{r}_p(u)^{\top} K(u) u d \theta d u = \mathbf{A}_1(s) + \mathbf{A}_2(s), \\
    \mathbf{B}(s) = \int_{0}^{\infty} \int_{\Theta_s(u)}^{\pi} \mathbf{r}_p(u)  K(u) u \sin(\theta) u d \theta d u = \mathbf{B}_1(s) + \mathbf{B}_2(s),
\end{gather*}
where
\begin{align*}
    \mathbf{A}_1(s)  & = \int_0^s \int_0^{\pi} \mathbf{r}_p(u) \mathbf{r}_p(u)^{\top} K(u) u d \theta d u = \pi \int_0^s \mathbf{r}_p(u) \mathbf{r}_p(u)^{\top} K(u) u d u, \\
    \mathbf{A}_2(s) & = \int_s^{\infty} \int_{\arccos(s/u)}^{\pi} \mathbf{r}_p(u) \mathbf{r}_p(u)^{\top} K(u) u d \theta d u = \int_s^{\infty} (\pi - \arccos(s/u)) \mathbf{r}_p(u) \mathbf{r}_p(u)^{\top} K(u) u d u, \\
    \mathbf{B}_1(s) & = \int_{0}^{s} \int_{0}^{\pi} \mathbf{r}_p(u)  K(u) u \sin(\theta) u d \theta d u = 2 \int_{0}^{s} \mathbf{r}_p(u)  K(u) u^2 d u, \\
    \mathbf{B}_2(s) & = \int_{s}^{\infty} \int_{\arccos(s/u)}^{\pi} \mathbf{r}_p(u)  K(u) u \sin(\theta) u d \theta d u = \int_{s}^{\infty} (1 + \frac{s}{u}) \mathbf{r}_p(u)  K(u) u^2 du.
\end{align*}
Evaluating the above at zero gives
\begin{align*}
    \mathbf{A}(0) = \frac{\pi}{2}  \int_0^{\infty} u \mathbf{r}_p(u) \mathbf{r}_p(u)^{\top} K(u)  d u, \quad \mathbf{B}(0) = \int_{0}^{\infty} u^2 \mathbf{r}_p(u)  K(u)  du.
\end{align*}
Hence
\begin{align}\label{sa-eq: value at zero}
   \operatorname{bias}_{n,1}(1,0) - \operatorname{bias}_{n,0}(1,0) = \mathbf{e}_1^{\top} \mathbf{A}(0)^{-1} \mathbf{B}(0) = \mathbf{e}_1^{\top} \mathbf{A}(0)^{-1} \bigg[\frac{2}{\pi} \mathbf{A}(0) \mathbf{e}_2 \bigg] = 0.
\end{align}
Taking derivatives with respect to $s$, we have
\begin{align*}
    & \dot{\mathbf{A}}_1(s) = \pi \mathbf{r}_p(s) \mathbf{r}_p(s)^{\top} K(s) s, \\
    & \dot{\mathbf{A}}_2(s) = - \pi \mathbf{r}_p(s) \mathbf{r}_p(s)^{\top} K(s) s + \int_s^{\infty} \frac{1}{\sqrt{u^2 - s^2}} u \mathbf{r}_p(u) \mathbf{r}_p(u)^{\top} K(u) d u, \\
    & \dot{\mathbf{B}}_1(s) = 2 \mathbf{r}_p(s)  K(s) s^2, \\
    & \dot{\mathbf{B}}_2(s) = - 2 \mathbf{r}_p(s)  K(s) s^2 + \int_{s}^{\infty} u \mathbf{r}_p(u)  K(u) du.
\end{align*}
Evaluating the above at zero gives
\begin{align*}
    \dot{\mathbf{A}}(0) = \int_0^{\infty} \mathbf{r}_p(u) \mathbf{r}_p(u)^{\top} K(u) d u, \quad \dot{\mathbf{B}}(0) = \int_{0}^{\infty} u \mathbf{r}_p(u) K(u) du.
\end{align*}
Using matrix calculus, we know
\begin{align}\label{sa-eq: derivative at zero}
    & \nonumber \frac{d}{d s}\operatorname{bias}_{n,1}(1,s) - \operatorname{bias}_{n,0}(1,s)\bigg|_{s = 0} \\
    & = \frac{d}{d s} \mathbf{e}_1^{\top} \mathbf{A}(s)^{-1} \mathbf{B}(s) \bigg|_{s = 0} \\
    & = - \mathbf{e}_1^{\top} \mathbf{A}(0)^{-1} \dot{\mathbf{A}}(0) \big[\mathbf{A}(0)^{-1} \mathbf{B}(0)\big] + \mathbf{e}_1^{\top} \mathbf{A}(0)^{-1} \dot{\mathbf{B}}(0) \\
    \nonumber & = - \mathbf{e}_1^{\top} \mathbf{A}(0)^{-1} \dot{\mathbf{A}}(0) \bigg[\frac{2}{\pi} \mathbf{e}_2\bigg] + \mathbf{e}_1^{\top} \bigg[\frac{2}{\pi} \mathbf{e}_1 \bigg] \\
    \nonumber & = - \frac{2}{\pi} \mathbf{e}_1^{\top} \mathbf{A}(0)^{-1} \int_0^{\infty} \begin{bmatrix}
        u \\
        u^2 \\
        \cdots \\
        u^{p+1}
    \end{bmatrix}
    K(u) d u + \mathbf{e}_1^{\top} \bigg[\frac{2}{\pi} \mathbf{e}_1 \bigg] \\
    & = - \frac{4}{\pi^2} + \frac{2}{\pi}.
\end{align}
Combining Equations~\eqref{sa-eq: value at zero} and \eqref{sa-eq: derivative at zero}, and the fact that $\frac{d}{ds}\operatorname{bias}_{n,1}(1,s) - \operatorname{bias}_{n,0}(1,s)$ is continuous in s, we can show $\sup_{0 \leq s \leq 1}|\operatorname{bias}_{n,1}(1,s) - \operatorname{bias}_{n,0}(1,s)| > 0$. Combining with Equation~\eqref{sa-eq: scaling}, we have
\begin{align*}
    \inf_{0 < h < 1}\sup_{\mathbf{x} \in \mathcal{B}}\frac{|\mathfrak{B}_{n,1}(\mathbf{x}) - \mathfrak{B}_{n,0}(\mathbf{x})|}{h} & \geq \inf_{0 < h < 1}\sup_{0 < s < h}\frac{|\operatorname{bias}_{n,1}(s,h) - \operatorname{bias}_{n,0}(s,h)|}{h} \\
    & = \inf_{0 < h < 1}\sup_{0 < s < h}\Big|\operatorname{bias}_{n,1}\left( 1,\frac{s}{h}\right)\Big| \\
    & > 0.
\end{align*}
\qed


\subsection{Proof of Theorem 3}\label{sa-sec: Proof of Theorem 3}

The proof of part (i) follows from part (ii) with $\mathcal{B} \cap B(\mathbf{x},\varepsilon)$ as the boundary. To prove part (ii), without loss of generality, we assume that $\iota = p + 1$, and want to show $\sup_{\mathbf{x} \in \mathcal{B}^o}|\mathfrak{B}_{n,t}(\mathbf{x})| \lesssim h^{p+1}$. This means we have assumed that $\mathcal{B}$ has a one-to-one curve length parametrization $\gamma$ that is $C^{p+3}$ with curve length $L$, there exists $\varepsilon, \delta > 0$ such that for all $\mathbf{x} \in \gamma([\delta, L - \delta])$ and $0 < r < \varepsilon$, $S(\mathbf{x},r)$ intersects $\mathcal{B}$ with two points, $s(\mathbf{x},r)$ and $t(\mathbf{x},r)$. Define $a(\mathbf{x},r)$ and $b(\mathbf{x},r)$ to be the number in $[0,2\pi]$ such that
\begin{align*}
    [a(\mathbf{x},r), b(\mathbf{x},r)] = \{\theta: \mathbf{x} + r(\cos \theta, \sin \theta) \in \mathcal{A}_1\}.
\end{align*}
Then, for $\mathbf{x} \in \mathcal{B}$ and $0 < r < \varepsilon$, $\theta_{1,\mathbf{x}}(r)$ has the following explicit representation:
\begin{align*}
    \theta_{1,\mathbf{x}}(r) = \frac{\int_{a(\mathbf{x},r)}^{b(\mathbf{x},r)}\mu_1(\mathbf{x} + r(\cos \theta, \sin \theta))f_{X}(\mathbf{x} + r(\cos \theta, \sin \theta))d \theta}{\int_{a(\mathbf{x},r)}^{b(\mathbf{x},r)}f_X(\mathbf{x} + r(\cos \theta, \sin \theta))d \theta}.
\end{align*}

\begin{center}
    \textbf{Step 1: Curve length v.s. Distance to $\gamma(0)$}
\end{center}

W.l.o.g., assume $\gamma(0) = \mathbf{x}$ and $\gamma'(0) = (1,0)$. Let $T: [0,\infty) \to [0,\infty)$ to be a continuous increasing function that satisfies
\begin{align*}
    \left\lVert\gamma \circ T(r)\right\rVert^2 = r^2, \qquad \forall r \in [0,h].
\end{align*}

\subsubsection*{Initial Case: $l = 1,2,3$.}
We will show that $T$ is $C^l$ on $(0,h)$. For notational simplicity, define another function $\phi: [0,\infty) \to [0,\infty)$ by $\phi(t) = \left\lVert\gamma(t)\right\rVert^2$. Using implicit derivations iteratively,
\begin{align*}
    & \phi \circ T(r) = r^2, \\
    & \phi'(T(r)) T'(r) = 2 r, \\
    & \phi^{\prime \prime}(T(r)) (T'(r))^2 + \phi'(T(r))T^{\prime \prime}(r) = 2, \\
    & \phi^{\prime \prime \prime}(T(r))(T'(r))^3 + 3\phi^{\prime \prime}(T(r)) T'(r) T^{\prime \prime}(r) + \phi'(T(r))T^{\prime \prime \prime}(r) = 0. \tag{1}
\end{align*}
From the above equalities, we get
\begin{align*}
    & T'(r) = \frac{2 r}{\phi'(T(r))}, \\
    & T^{\prime \prime}(r) = \frac{2 - \phi^{\prime \prime}(T(r)) \left(T'(r)\right)^2}{\phi'(T(r))}, \\
    & T^{\prime \prime \prime}(r) = -\frac{\phi^{\prime \prime \prime}(T(r))(T'(r))^3 + 3 \phi^{\prime \prime}(T(r))T'(r) T^{\prime \prime}(r)}{\phi'(T(r))}.
\end{align*}
Since we have assumed $\gamma$ is $C^{p+3}$ on $(0,h)$, $\phi$ is also $C^{p+1}$ on $(0,h)$. It follows from the above calculation that $T$ is $C^{p+3}$ on $(0,h)$. In order to find the limit of derivatives of $T$ at $0$, we need
\begin{align*}
    & \phi(t) = \gamma_1(t)^2 + \gamma_2(t)^2, && \phi(0) = 0, \\
    & \phi'(t) = 2 \gamma_1(t) \gamma_1'(t) + 2 \gamma_2(t) \gamma_2'(t), && \phi'(0) = 0, \\
    & \phi^{\prime \prime}(t) = 2 \gamma_1'(t) \gamma_1'(t) + 2 \gamma_1(t) \gamma_1^{\prime \prime}(t) + 2 \gamma_2'(t) \gamma_2'(t) + 2 \gamma_2(t) \gamma_2^{\prime \prime}(t), && \phi^{\prime \prime}(0) = 2, \\
    & \phi^{\prime \prime \prime}(t) = 6 \gamma_1'(t) \gamma_1^{\prime \prime}(t) + 2 \gamma_1(t) \gamma_1^{\prime \prime \prime}(t) + 6 \gamma_2'(t)\gamma_2^{\prime \prime}(t) + 2 \gamma_2(t) \gamma_2^{\prime \prime \prime}(t).
\end{align*}
Using L'Hôpital's rule
\begin{align*}
    & \lim_{r \downarrow 0} T'(r) = \lim_{r \downarrow 0}\frac{2}{\phi^{\prime \prime}(T(r)) T'(r)} = \frac{2}{2 \lim_{r \downarrow 0}T'(r)} \implies \lim_{r \downarrow 0} T'(r) = 1,\\
    & \lim_{r \downarrow 0} T^{\prime \prime}(r) = \lim_{r \downarrow 0} \frac{- \phi^{\prime \prime \prime}(T(r))(T'(r))^3 - \phi^{\prime \prime}(T(r))2 T'(r)T^{\prime \prime}(r)}{\phi^{\prime \prime}(T(r))T'(r)} \\
    & \phantom{\lim_{r \downarrow 0} T^{\prime \prime}(r) } = \frac{-\phi^{(3)}(0) - 4 \lim_{r \downarrow 0}T^{\prime \prime}(r)}{2}\\
    & \phantom{\lim_{r \downarrow 0} T^{\prime \prime}(r) }= \frac{- \phi^{(3)}(0)}{6}
\end{align*}
\begin{align*}
    & \lim_{r \downarrow 0} T^{(3)}(r) = - \lim_{r \downarrow 0} \frac{\phi^{(4)}(T(r))(T'(r))^4 + \phi^{(3)}(T(r))3 (T'(r))^2 T^{\prime \prime}(r) + 3 \phi^{(3)}(T(r))(T'(r))^2 T^{\prime \prime}(r) }{\phi^{\prime \prime}(T(r))T'(r)} \\
    &  \phantom{\lim_{r \downarrow 0} T^{\prime \prime}(r) }+ \lim_{r \downarrow 0} \frac{3 \phi^{\prime \prime}(T(r)) T'(r) T^{(3)}(r)}{\phi^{\prime \prime}(T(r))T'(r)} \\
    &  \phantom{\lim_{r \downarrow 0} T^{\prime \prime}(r) } = -\frac{\phi^{(4)}(0) - (\phi^{(3)}(0))^2 + 6 \lim_{r \downarrow 0} T^{(3)}(r)}{2} \\
    & \phantom{\lim_{r \downarrow 0} T^{\prime \prime}(r) } = - \frac{\phi^{(4)}(0) - (\phi^{(3)}(0))^2}{8}.
\end{align*}

\subsubsection*{Induction Step: $l \geq 4$.} Assume $\lim_{r \downarrow 0}T^{(i)}(r)$ exists and is finite for $0 \leq i \leq l-2$ and there exists a function $q(r)$ such that (i) $q(r)$ is a polynomial of $\phi^{(j)}(T(r)), 1 \leq j \leq l-1$ and $T^{(k)}(r), 1 \leq k \leq l-2$, (ii) $\lim_{r \downarrow 0}q(r) = 0$ and (iii)
\begin{align*}
    q(r) + \phi'(T(r)) T^{(l-1)}(r) = 0. \tag{2}
\end{align*}
For $l = 4$, this assumption can be verified from Equation (1). Using L'hopital's rule,
\begin{align*}
    \lim_{r \downarrow 0}T^{(l-1)}(r) & = \lim_{r \downarrow 0} - \frac{q(r)}{\phi'(T(r)}\\
    & \stackrel{L'h}{=} \lim_{r \downarrow 0} - \frac{q'(r)}{\phi^{\prime \prime}(T(r))T'(r)}.
\end{align*}
From the previous paragraph, $\lim_{r \downarrow 0} \phi^{\prime \prime}(T(r))T'(r)$ exists and is finite. And $q'(r)$ is a polynomial of $\phi^{(j)}(T(r)), 1 \leq j \leq l$ and $T^{(k)}(r), 1 \leq k \leq l-1$. Hence $\lim_{r \downarrow 0}T^{(l-1)}(r)$ can be solved from the following equation and is finite:
\begin{align*}
    \lim_{r \downarrow 0}q'(r) + \lim_{r \downarrow 0}\phi^{\prime \prime}(T(r))T'(r) \cdot \lim_{r \downarrow 0}T^{(l-1)}(r) = 0. \tag{3}
\end{align*}
Taking derivatives on both sides of Equation (2),
\begin{align*}
    q'(r) + \phi^{\prime \prime}(T(r))T'(r) T^{(l-1)}(r) + \phi'(T(r)) T^{(l)}(r) = 0.
\end{align*}
Take $q_2(r) = q'(r) + \phi^{\prime \prime}(T(r))T'(r)T^{(l-1)}(r)$. Then, (i) $q_2(r)$ is a polynomial of $\phi^{(j)}(T(r)), 1 \leq j \leq l$ and $T^{(k)}(r), 1 \leq k \leq l-1$, (ii) $\lim_{r \downarrow 0}q_2(r) = 0$, and (iii)
\begin{align*}
    q_2(r) + \phi'(T(r))T^{(l)}(r) = 0.
\end{align*}
Continue this argument till $l = p+3$, $\lim_{r \downarrow 0}T^{(j)}(r)$ exists and is a polynomial of $\phi^{(0)}(0), \ldots, \phi^{(j+1)}(0)$, which implies that it is bounded by a constant only depending on $\gamma$.

\subsubsection*{Step 2: $(p+1)$-times continuously differentiable $S_r$}
We use the notation $\gamma(t) = (\gamma_1(t), \gamma_2(t))$. Define
\begin{align*}
    A(t) = \angle \gamma(t) - \gamma(0), \gamma'(0) = \arcsin\left( \frac{\gamma_2(t)}{\left\lVert\gamma(t)\right\rVert}\right).
\end{align*}
Since $\gamma$ is $C^{p+3}$, we can Taylor expand $\gamma$ at $0$ to get
\begin{align*}
    \gamma(t) = \begin{pmatrix} 0 \\ 0\end{pmatrix} + \begin{pmatrix} 1 \\ 0 \end{pmatrix} t + \begin{pmatrix} u_2 \\ v_2 \end{pmatrix} t^2 + \cdots \begin{pmatrix} u_{p+2} \\ v_{p+2} \end{pmatrix} t^{p+2} + \begin{pmatrix} R_1(t)\\ R_2(t)\end{pmatrix},
\end{align*}
where we have used the fact that $\gamma_2'(0) = 0$ and $\left\lVert\gamma'(0)\right\rVert = 1$ and
\begin{align*}
    R_1(t) & = \int_0^t \frac{\gamma_1^{(p+3)}(s)(t-s)^{p+2}}{(p+2)!} d s, \qquad R_2(t) = \int_0^t \frac{\gamma_2^{(p+3)}(s)(t-s)^{p+2}}{(p+2)!}ds.
\end{align*}
Since $\gamma$ is $C^{p+3}$, $R_1(t)/t$ and $R_2(t)/t$ are $C^{p+3}$ on $(0,\infty)$. We \emph{claim} that $\lim_{t \downarrow 0} \frac{d^v}{d t^v} (R_1(t)/t)$ exists and is uniformly bounded for all $\mathbf{x} \in \mathcal{B}$, for all $0 \leq v \leq p+1$. Define $\varphi(t) = R_1(t)/t$. Then
\begin{align*}
    \varphi'(t) & = - \frac{R_1(t)}{t^2} + \frac{R_1'(t)}{t}, \\
    \varphi^{\prime \prime}(t) & = \frac{2 R_1(t)}{t^3} - \frac{2 R_1'(t)}{t^2} + \frac{R_1^{\prime \prime}(t)}{t},\\
    \varphi^{(3)}(t) &= - \frac{6 R_1(t)}{t^4} + \frac{6 R_1'(t)}{t^3} - \frac{3 R_1^{(2)}(t)}{t^2} + \frac{R_1^{(3)}(t)}{t}
    & \cdots
\end{align*}
where
\begin{align*}
    R_1'(t) = \int_0^t \frac{\gamma_1^{(p+1)}(s)(t-s)^{p-1}}{(p-1)!}ds, \qquad R_1^{\prime \prime}(t) = \int_0^t \frac{\gamma_1^{(p+1)}(s)(t-s)^{p-2}}{(p-2)!}ds, \qquad \cdots
\end{align*}
Since $\gamma_1$ is $C^{p+3}$, there exists $C_1 > 0$ only depending on $\gamma$ such that for all $0 \leq v \leq p+3$,$\left|\frac{d^v}{d t^v} R_1(t) \right| \leq C_1 t^{p+1-v}$. Hence
\begin{align*}
    \lim_{r \downarrow 0} \varphi^{(j)}(r) = 0, \qquad \forall 0 \leq j \leq p+1.
\end{align*}
Similarly, $\lim_{r \downarrow 0} \frac{d^v}{d t^v}(R_2(t)/t)$ exists and is uniformly bounded for all $0 \leq v \leq p+1$. Then
\begin{align*}
    \frac{\gamma_2(t)}{\left\lVert\gamma(t)\right\rVert} & = \frac{v_2 t + \cdots + v_{p+2} t^{p+2} + R_2(t)/t}{\sqrt{(1 + u_2t + \cdots u_{p+2} t^{p+2} + R_1(t)/t)^2 + (v_2 t + \cdots + v_{p+2}t^{p+2} + R_2(t)/t)^2}}, t > 0.
\end{align*}
Notice that $\gamma_2(t)/ \left\lVert\gamma(t)\right\rVert$ is of the form $$p(t)(1 + q(t))^{\alpha},$$ where $\alpha < 0$ and $p(t), q(t)$ are $C^{p+1}$ on $(0,\infty)$ with $\lim_{r \downarrow 0}d^v / d t^v p(t)$ and $\lim_{r \downarrow 0}d^v / d t^v q(t)$ finite. Since the derivative of $p(t)(1 + q(t))^{\alpha}$ is $$p'(t)(1 + q(t))^{\alpha} + p(t) \alpha (1 + q(t))^{\alpha - 1} q'(t),$$ which is the sum of two terms of the form $p_2(t)(1 + q_2(t))^{\alpha}$ with $p_2$ and $q_2$ functions that are $C^{p}$ with finite limits at $0$. Continue this argument, we see that $\frac{\gamma_2(\cdot)}{\left\lVert\gamma(\cdot)\right\rVert}$ is $C^{p+1}$ on $(0,\infty)$ and $\lim_{r \downarrow 0} \frac{d^v}{d t^v} \left(\gamma_2(t) / \left\lVert\gamma(t)\right\rVert \right)$ exist and are uniformly bounded for all $\mathbf{x} \in \mathcal{B}$ and for all $0 \leq v \leq p+1$.

Since $\arcsin$ is $C^{p+1}$ with bounded (higher order derivatives) on $[-1/2,1/2]$, $A$ is $C^{p+1}$ on $(0, \delta)$ and for all $0 \leq v \leq p+1$, $\lim_{r \downarrow 0}A^{(v)}(t)$ exist and are uniformly bounded for all $\mathbf{x} \in \mathcal{B}$.

\begin{center}
    \textbf{Step 3: $(p+1)$-times continuously differentiable conditional density}
\end{center}
By the previous two steps, $a(\mathbf{x},r) = A \circ T(r)$ is $C^{p+1}$ on $(0,\infty)$ with $|\lim_{r \downarrow 0}\frac{d^v}{d r^v} a(\mathbf{x},r)| < \infty$. Similarly, we can show that $b(\mathbf{x},r)$ is $C^{p+1}$ in $r$ with finite limits at $r = 0$. By the assumption that $f_{X}$ is $C^{p+1}$ and bounded below by $\underline{f}$, $\theta_{1,\mathbf{x}}$ is $C^{p+1}$ with $\lim_{r \downarrow 0} \frac{d^v}{d r^v} \theta_{1,\mathbf{x}}(r)$ uniformly bounded for all $\mathbf{x} \in \mathcal{B}$ and for all $0 \leq v \leq p+1$.
\bigskip

This completes the proof.
\qed


\subsection{Proof of Theorem 6}\label{sa-sec: Proof of Theorem 6}

Let $s > 0$ be a parameter that is chosen later. Consider the following two data generating processes.

\subsubsection*{Data Generating Process $\mathbb{P}_0$.} Let  $\mathcal{X} = \{r(\cos \theta, \sin \theta): 0 \leq r \leq 1, 0 \leq \theta \leq \Theta(r)\}$, where
\begin{align*}
    \Theta(r) =
    \begin{cases}
    \pi, & 0 \leq r < s, \\
    \theta_k, & s + k s^2 \leq r < s + (k + 1) s^2, 0 \leq k < K, \\
    \theta_K, & s + K s^2 \leq r < 1,
    \end{cases}
\end{align*}
with $K = \lfloor \frac{1 - s}{s^2} \rfloor$ and $\theta_k$ is the unique zero of  $$\frac{\sin(\theta)}{\theta} = \frac{(k + \frac{1}{2})s^2}{s + (k + \frac{1}{2})s^2}$$ over $\theta \in [0,\pi]$, and $\theta_K$ is the unique zero of $$\frac{\sin(\theta)}{\theta} = \frac{K s^2 + 1 - s}{s + K s^2 + 1}$$ over $\theta \in [0,\pi]$. Suppose $\mathbf{X}_i$ has density $f_X$ given by
\begin{align*}
    f_X(r(\cos \theta,\sin \theta)) = \frac{1}{2 \Theta(r)}, \qquad
    0 \leq r \leq 1, 0 \leq \theta \leq \Theta(r).
\end{align*}
Suppose
\begin{align*}
    \mu_0(x_1,x_2) = \frac{1}{2} + \frac{1}{100} x_1, \qquad (x_1,x_2) \in \mathbb{R}^2.
\end{align*}
Suppose $Y_i = \mathds{1}(\eta_i \leq \mu(\mathbf{X}_i))$ where $(\eta_i:i:1,\cdots,n)$ are i.i.d. random variables independent of $(\mathbf{X}_i:1,\cdots,n)$. Let $\eta_0(r) = \mathbb{E}_{\mathbb{P}_0}[Y_i|\left\lVert\mathbf{X}_i - (0,0)\right\rVert = r]$, for $r \geq 0$. In particular, $\mathtt{bd}(\mathcal{X})$ has length $\pi + 2$. Hence, $\mathtt{bd}(\mathcal{X})$ is a rectifiable curve.
\begin{figure}[!tbp]
  \centering
  \begin{minipage}[b]{0.8\textwidth}
    \includegraphics[width=\textwidth]{figures/region0.png}
    \caption{$\mathcal{X}$ from DGP $\mathbb{P}_0$}
  \end{minipage}
  \hfill
  \begin{minipage}[b]{0.8\textwidth}
    \includegraphics[width=\textwidth]{figures/region1.png}
    \caption{$\mathcal{X}$ from DGP $\mathbb{P}_1$}
  \end{minipage}
\end{figure}

\subsubsection*{Data Generating Process $\mathbb{P}_1$.} Let $\mathcal{X}=\{r(\cos \theta, \sin \theta): 0 \leq r \leq 1, 0 \leq \theta \leq \pi/2\}$, $\mathbf{X}_i$ is uniformly distributed on $\mathcal{X}$, and
\begin{align*}
    \mu_1(x_1, x_2) = \frac{1}{2} + \frac{1}{100}(x_1 - s), \qquad (x_1, x_2) \in \mathbb{R}^2.
\end{align*}
Suppose $Y_i = \mathds{1}(\eta_i \leq \mu(\mathbf{X}_i))$ where $(\eta_i:1,\cdots,n)$ are i.i.d random variables independent to $(\mathbf{X}_i:1,\cdots,n)$. Let $\eta_1(r) = \mathbb{E}_{\mathbb{P}_1}[Y_i|\left\lVert\mathbf{X}_i - (0,0)\right\rVert = r]$, for $r \geq 0$. In particular, $\mathtt{bd}(\mathcal{X})$ has length $\pi/2 + 2$. Hence, $\mathtt{bd}(\mathcal{X})$ is a rectifiable curve.

\subsubsection*{Minimax Lower Bound.} First, we show under the previous two models, $\mathbb{P}_0(\left\lVert\mathbf{X}_i\right\rVert \leq r) = \mathbb{P}_1(\left\lVert\mathbf{X}_i\right\rVert \leq r)$ for all $r \geq 0$. Since in $\mathbb{P}_1$, $\mathbf{X}_i$ is uniform distributed on $\mathbb{R}$, we know $\mathbb{P}_1(\left\lVert\mathbf{X}_i\right\rVert \leq r) = r^2$, $0 \leq r \leq 1$.
\begin{align*}
    \mathbb{P}_0(\left\lVert\mathbf{X}_i\right\rVert \leq r) = \int_0^r \int_0^{\Theta(s)} \frac{1}{2 \Theta(s)} s d \theta d s = r^2, \quad 0 \leq r \leq 1.
\end{align*}
Hence, choosing $(0,0)$ as the point of evaluation in both $\mathbb{P}_0$ and $\mathbb{P}_1$, we have
\begin{align*}
    & d_{\operatorname{KL}}(\mathbb{P}_0(\left\lVert\mathbf{X}_i - (0,0)\right\rVert,Y_i),\mathbb{P}_1(\left\lVert\mathbf{X}_i - (0,0)\right\rVert,Y_i))\\
    &= \int_0^{\infty} \int_{-\infty}^{\infty}d\mathbb{P}_0(r,y) \log \frac{d \mathbb{P}_0(r,y)}{d \mathbb{P}_1(r,y)} \\
    &= \int_0^{\infty} \int_{-\infty}^{\infty} d\mathbb{P}_0(r) d \mathbb{P}_0(y|r)\log \frac{d\mathbb{P}_0(r) d\mathbb{P}_0(y|r)}{d\mathbb{P}_1(r) d \mathbb{P}_1(y|r)} \\
    &= \int_0^{\infty} d \mathbb{P}_0(r) \int_{-\infty}^{\infty}  d \mathbb{P}_0(y|r)\log \frac{ d \mathbb{P}_0(y|r)}{d \mathbb{P}_1(y|r)} \\
    &= 2 \int_0^1 d_{\operatorname{KL}}(\mathsf{Bernoulli}(\eta_0(r)), \mathsf{Bernoulli}(\eta_1(r))) r d r.
\end{align*}
Under $\mathbb{P}_0$, $\mathbf{X}_i$ is uniformly distributed on $\{r(\cos \theta, \sin \theta): 0 \leq \theta \leq \Theta(r)\}$ for each $0 < r \leq 1$. Hence
\begin{align*}
    \eta_0(r) & = \frac{1}{2} + \frac{1}{100}\frac{1}{\Theta(r)}\int_0^{\Theta(r)} r \cos(u) d u - \frac{s}{100}
    = \frac{1}{2} + \frac{1}{100}r\frac{\sin(\Theta(r))}{\Theta(r)}.
\end{align*}
Thus, for $0 \leq k < K$,
\begin{align*}
    \eta_0 \Big(s + (k + \frac{1}{2})s^2\Big)
    & = \frac{1}{2} + \frac{1}{100} \Big((s + (k + \frac{1}{2})s^2) \frac{\sin(\Theta_k)}{\Theta_k} \Big)  \\
    & = \frac{1}{2} + \frac{1}{100} \Big((s + (k + \frac{1}{2})s^2) \frac{(k + \frac{1}{2})s^2}{s + (k + \frac{1}{2})s^2}\Big) \\
    & = \eta_1 \Big(s + (k + \frac{1}{2})s^2\Big).
\end{align*}
Since both $\eta_0$ and $\eta_1$ are $1$-Lipschitz on all intervals $[s + ks^2, s + (k + 1)s^2]$ for all $0 \leq k < K$, we know $|\eta_0(r) - \eta_1(r)| \leq 2s^2$ for all $r \in [s,1]$. Moreover, $\eta_0(r) = \frac{1}{2}$ for all $0 \leq r \leq s$ and $\eta_1(r) = \frac{1}{2} + \frac{1}{100}(r\frac{2}{\pi} - s)$. Hence $|\eta_0(r) - \eta_1(r)| \leq s$ for all $0 \leq r \leq s$. Hence,
\begin{align*}
     \int_0^1 d_{\operatorname{KL}}(\mathsf{Bernoulli}(\eta_0(r)), \mathsf{Bernoulli}(\eta_1(r))) r d r
    & \leq \int_0^1 d_{\chi^2}(\mathsf{Bernoulli}(\eta_0(r)), \mathsf{Bernoulli}(\eta_1(r))) r d r\\
    &= \int_0^1 (\eta_1(r) \Big(\frac{\eta_0(r) - \eta_1(r)}{\eta_1(r)}\Big)^2 + (1 - \eta_1(r))\Big(\frac{\eta_0(r) - \eta_1(r)}{1 - \eta_1(r)}\Big)^2) r d r\\
    & \leq \frac{1}{\frac{1}{2}-\frac{3}{100}} \int_0^1 (\eta_0(r) - \eta_1(r))^2 r dr \\
    & \leq \frac{1}{\frac{1}{2}-\frac{3}{100}} \int_0^s s^2 r d r + \frac{1}{\frac{1}{2}-\frac{3}{100}}\int_s^1 (2s^2)^2 r d r\\
    & \leq \frac{5}{\frac{1}{2}-\frac{3}{100}} s^4.
\end{align*}
Moreover, $|\mu_0(0,0) - \mu_1(0,0)| = \frac{1}{100}s$. Hence, by \citet[Theorem 2.2 (iii)]{tsybakov2008introduction}, take $\frac{5}{\frac{1}{2}-\frac{3}{100}} s_{\ast}^4 = \frac{\log 2}{n}$, and conclude that
\begin{align*}
    \inf_{T_n \in \mathcal{T}} \sup_{\mathbb{P} \in \mathcal{P}} \sup_{\mathbf{x} \in \mathcal{B}(\mathbb{P})} \mathbb{E}_{\mathbb{P}} [|T_n(\mathbf{U}_n(\mathbf{x})) - \mu(\mathbf{x})|] \geq \frac{1}{1600}s_{\ast} \gtrsim n^{-\frac{1}{4}}.
\end{align*}
This concludes the proof.
\qed


\bibliography{CTY_2026_JOE--bib}
\bibliographystyle{plainnat}