The exact contents of citations.db main_text.text for this paper — one flattened LaTeX string, title through conclusion, appendix excluded, unmodified except for removing email addresses. This is what our citation measures are computed over.
412,401 characters
Supplemental Appendix to ``Strong Approximations for Empirical Processes Indexed by Lipschitz Functions''
\title{Supplemental Appendix to ``Strong Approximations for Empirical Processes Indexed by Lipschitz Functions''\footnote{Corresponding author: [email removed]}. Support from the National Science Foundation through grants DMS-2210561 and SES-2241575 is gratefully acknowledged.}}
\author{
Matias D. Cattaneo\footnote{Department of Operations Research and Financial Engineering, Princeton University.}
\and
Ruiqi (Rae) Yu\footnote{Department of Operations Research and Financial Engineering, Princeton University.}
}
\maketitle
\setcounter{page}{0}\thispagestyle{empty}
\begin{abstract}
This supplement appendix reports additional theoretical results not discussed in the paper to conserve space, and provides all the technical proofs. Section \ref{sa-sec: Notations and Definitions} introduces additional notation and definitions used in the proofs. Section \ref{sa-sec: General Empirical Process} studies the general empirical process (Section 3 in the paper). Section \ref{sa-sec: Multiplicative Empirical Process} studies the multiplicative-separable empirical process (not discussed in the paper but of independent interest). Section \ref{sa-sec: Residual-Based Empirical Process} studies the residual-based empirical process (Section 4 in the paper). Section \ref{sa-sec: Quasi-Uniform Haar Basis} studies the three empirical processes in the context of quasi-uniform Haar basis (Section 5 in the paper).
\end{abstract}
\thispagestyle{empty}
\clearpage
\onehalfspacing
\setcounter{page}{1}
\pagestyle{plain}
\singlespacing
\setcounter{tocdepth}{3}
\tableofcontents
\clearpage
\onehalfspacing
\section{Additional Notation}\label{sa-sec: Notations and Definitions}
We introduce additional notation and definitions complementing those given in Section 2 of the paper. See \cite{Ambrosio-Fusco-Pallara_2000_book}, \cite{wellner2013weak-SA}, \cite{Gine-Nickl_2016_Book-SA}, and references therein, for background definitions and more details.
Let $\mathcal{U}, \mathcal{V} \subseteq \mathbb{R}^q$. We define $\mathcal{U} - \mathcal{V} = \{\mathbf{u} - \mathbf{v}: \mathbf{u} \in \mathcal{U}, \mathbf{v} \in \mathcal{V}\}$. We define $\mathcal{U} \triangle \mathcal{V} = (\mathcal{U} \setminus \mathcal{V}) \cup (\mathcal{V} \setminus \mathcal{U})$. Let $\operatorname{det}(\mathbf{A})$ be the determinant of the matrix $\mathbf{A}$. Let $\Phi(z)$ be the distribution function of $\mathsf{Normal}(0,1)$, and $\mathsf{Bern}(p)$ denote the Bernoulli distribution with parameter $p\in(0,1)$. For a real-valued random variable $X$, the $L_p$-norm is defined as $\|X\|_p = \mathbbm{E}[|X|^p]^{1/p}$ for $1 \leq p < \infty$. The $\sigma$-algebra generated by $X$ is denoted by $\boldsymbol{\sigma}(X)$. For $\alpha > 0$, the $\psi_\alpha$-norm of $X$ is given by $\|X\|_{\psi_\alpha} = \min\big\{ \lambda > 0 : \mathbbm{E}\big[\exp\big(\big(\frac{|X|}{\lambda}\big)^\alpha\big)\big] \leq 2 \big\}$. For $\mathbf{x} \in \mathbb{R}^q$ and $r > 0$, let $B(\mathbf{x}, r)$ denote the Euclidean ball with radius $r$ centered at $\mathbf{x}$. For a matrix $\mathbf{A} \in \mathbb{R}^{q \times q}$, $\lVert \mathbf{A} \rVert$ denotes its operator norm. Using standard empirical process notation, $\mathbbm{E}_n[f(\mathbf{x}_i)]$ denotes the empirical average $n^{-1} \sum_{i = 1}^n[f(\mathbf{x}_i) - \mathbbm{E}[f(\mathbf{x}_i)]]$ based on random sample $(\mathbf{x}_i: 1 \leq i \leq n)$. For sequences of real numbers, we write $a_n = \Omega(b_n)$ if there exists some constant $C$ and $N > 0$ such that $n > N$ implies $|a_n| \geq C |b_n|$.
Let $\mathscr{S} \subseteq \mathbb{R}^q$ and $Q$ be a measure on $(\mathscr{S}, \mathcal{B}(\mathscr{S}))$. The semi-metric $\mathfrak{d}_Q$ on $L_2(Q)$ is defined by $\mathfrak{d}_Q(f, g) = \big(\|f - g\|_{Q,2}^2 - \big(\int f \, dQ - \int g \, dQ\big)^2\big)^{1/2}$, for $f, g \in L_2(Q)$. For a class $\mathscr{F} \subseteq L_2(Q)$, let $C(\mathscr{F}, \mathfrak{d}_Q)$ denote the class of all continuous functionals on the space $(\mathscr{F}, \mathfrak{d}_Q)$. For $\alpha > 0$, the $C^\alpha$-norm of a real-valued measurable function $f$ on $(\mathscr{S}, \mathcal{B}(\mathscr{S}))$ is given by $\|f\|_{C^\alpha} = \max_{|k| \leq \lfloor \alpha \rfloor} \sup_{\mathbf{x} \in \mathscr{S}} |D^k f(\mathbf{x})| + \max_{|k| = \alpha} \sup_{\mathbf{x} \neq \mathbf{y} \in \mathscr{S}} \frac{|D^k f(\mathbf{x}) - D^k f(\mathbf{y})|}{\|\mathbf{x} - \mathbf{y}\|_2^{\alpha - \lfloor \alpha \rfloor}}$. The space $C^\alpha(\mathscr{S})$ denotes the collection of all real-valued measurable functions on $(\mathscr{S}, \mathcal{B}(\mathscr{S}))$ with $C^\alpha$-norm bounded by 1. For real-valued functions $f,g$ on $(\mathbb{R}^d,\mathcal{B}(\mathbb{R}^d))$, the convolution of $f$ and $g$ is the function $f \ast g$ such that $f \ast g(x) = \int_{-\infty}^{\infty}f(y)g(x - y) d y, x \in \mathbb{R}$. If $\mathscr{F}$ and $\mathscr{G}$ are two sets of functions from measure space $(\mathcal{U},\mathcal{B}(\mathcal{U}))$ to $\mathbb{R}$ and $(\mathcal{V},\mathcal{B}(\mathcal{V}))$ to $\mathbb{R}$, respectively, then $\mathscr{F} \cdot \mathscr{G}$ denotes the class of measurable functions $\{f \cdot g: f \in \mathscr{F}, g \in \mathscr{G}\}$ from $(\mathcal{U} \times \mathcal{V}, \mathcal{B}(\mathcal{U}) \otimes \mathcal{B}(\mathcal{V}))$ to $\mathbb{R}$. For a semi-metric space $(\mathscr{F}, \mathfrak{d})$ of real-valued measurable functions on $(\mathscr{S}, \mathcal{B}(\mathscr{S}))$, $N_{[\,]}(\varepsilon, \mathscr{F}, \mathfrak{d})$ denotes the bracketing number.
For a probability measure $P$ on $(\mathscr{S}, \mathcal{B}(\mathscr{S}))$, a \emph{$P$-Brownian bridge} is a centered Gaussian random function $(W_P(f): f \in L_2(P))$ with covariance given by $\mathbbm{E}[W_P(f) W_P(g)] = P(fg) - P(f)P(g)$ for $f, g \in L_2(P)$. A class $\mathscr{F} \subseteq L_2(P)$ is said to be $P$-pregaussian if there exists a version of the $P$-Brownian bridge $W_P$ such that $W_P \in C(\mathscr{F}; \mathfrak{d}_P)$ almost surely.
Finally, we use $a_n \lesssim b_n$ to denote that $a_n = O(b_n)$ with only a universal constant, not a function of the data generating process or related parameters. For $K \in \mathbb{N}$, we repeatedly employ the index sets $\mathcal{I}_K = \{(j,k) \in \mathbb{N} \times \mathbb{N}: 1 \leq j \leq K, 0 \leq k < 2^{K - j}\}$ and $\mathcal{J}_K = \{(j,k) \in \mathbb{N} \times \mathbb{N}: 0 \leq j \leq K, 0 \leq k < 2^{K - j}\}$.
\subsection{Additional Main Definitions}
Let $\mathscr{F}$ be a class of measurable functions from a probability space $(\mathbb{R}^q, \mathcal{B}(\mathbb{R}^q), \mathbbm{P})$ to $\mathbb{R}$. We introduce several additional definitions that capture properties of $\mathscr{F}$, complementing those in Section 2.1.
\begin{defn}\label{sa-defn: smooth tv}
For a non-empty $\mathcal{C} \subseteq \mathbb{R}^q$, the smoothed uniform total variation of $\mathscr{F}$ over $\mathcal{C}$ is
\begin{align*}
\mathtt{TV}_{\mathscr{F},\mathcal{C}}^* = \sup_{f \in \mathscr{F}} \inf_{(f_\ell)_{\ell \in \mathbb{N}}}\limsup_{\ell \to \infty} \mathtt{TV}_{\{f_\ell\},\mathcal{C}},
\end{align*}
where the infimum is taken over all sequences of functions $(f_\ell)_{\ell \in \mathbb{N}}$ such that $f_\ell \to f \in \mathscr{F}$ a.s.-$\operatorname*{\mathfrak{m}}$ on $(\mathcal{C},\mathcal{B}(\mathcal{C}))$, and $f_\ell$ is differentiable and bounded by $2 \mathtt{M}_{\mathscr{F},\mathcal{C}}$ on $\mathcal{C}$ for all $\ell\geq1$.
\end{defn}
\begin{defn}\label{sa-defn: smooth ktv}
For a non-empty $\mathcal{C} \subseteq \mathbb{R}^q$, the smoothed uniform local total variation of $\mathscr{F}$ over $\mathcal{C}$ is a positive number $\mathtt{K}_{\mathscr{F},\mathcal{C}}^*$ such that for any cube $\mathcal{D} \subseteq \mathbb{R}^q$ with edges of length $\ell$ parallel to the coordinate axises,
\begin{align*}
\mathtt{TV}_{\mathscr{F},\mathcal{D} \cap \mathcal{C}}^* \leq \mathtt{K}_{\mathscr{F},\mathcal{C}}^* \ell^{d-1}.
\end{align*}
\end{defn}
Suppose \(\mathscr{S}\) is also a class of measurable functions from the probability space \((\mathbb{R}^q, \mathcal{B}(\mathbb{R}^q), \mathbbm{P})\) to \(\mathbb{R}\). We generalize the definition of the uniform covering number to \(\mathscr{F} \times \mathscr{S}\).
\begin{defn}\label{sa-defn: uniform covering number for product space}
For a non-empty \(\mathcal{C} \subseteq \mathbb{R}^q\), the uniform covering number of \(\mathscr{F} \times \mathscr{S}\) with envelope \(M_{\mathscr{F},\mathcal{C}} M_{\mathscr{S},\mathcal{C}}\) over \(\mathcal{C}\) is
\begin{align*}
\mathtt{N}_{\mathscr{F} \times \mathscr{S},\mathcal{C}}(\delta, M_{\mathscr{F},\mathcal{C}} M_{\mathscr{S},\mathcal{C}}) = \sup_{\mu} N(\mathscr{F} \times \mathscr{S}, \lambda_{\mu}, \delta \|M_{\mathscr{F},\mathcal{C}} M_{\mathscr{S},\mathcal{C}}\|_{\mu,2}), \qquad \delta \in (0, \infty),
\end{align*}
where the supremum is taken over all finite discrete measures on \((\mathcal{C}, \mathcal{B}(\mathcal{C}))\), and \(\lambda_{\mu}\) is the semi-metric on \(\mathscr{F} \times \mathscr{S}\) defined by
\begin{align*}
\lambda_{\mu}((f_1, g_1), (f_2, g_2))^2 = \int_{\mathcal{C}} (f_1(\mathbf{x})g_1(\mathbf{x}) - f_2(\mathbf{x}) g_2(\mathbf{x}))^2 \, d \mu(\mathbf{x}).
\end{align*}
We assume that \(M_{\mathscr{F},\mathcal{C}}(\mathbf{u})\) and \(M_{\mathscr{S},\mathcal{C}}(\mathbf{u})\) are finite for every \(\mathbf{u} \in \mathcal{C}\).
\end{defn}
\begin{defn}\label{sa-defn: uniform entropy integral for product space}
For a non-empty \(\mathcal{C} \subseteq \mathbb{R}^q\), the uniform entropy integral of \(\mathscr{F} \times \mathscr{S}\) with envelope \(M_{\mathscr{F},\mathcal{C}} M_{\mathscr{S},\mathcal{C}}\) over \(\mathcal{C}\) is
\begin{align*}
J_\mathcal{C}(\delta, \mathscr{F} \times \mathscr{S}, M_{\mathscr{F},\mathcal{C}} M_{\mathscr{S},\mathcal{C}}) = \int_0^{\delta} \sqrt{1 + \log \mathtt{N}_{\mathscr{F} \times \mathscr{S},\mathcal{C}}(\varepsilon, M_{\mathscr{F},\mathcal{C}} M_{\mathscr{S},\mathcal{C}})} \, d \varepsilon,
\end{align*}
where it is assumed that $M_{\mathscr{F},\mathcal{C}}(\mathbf{u})M_{\mathscr{S},\mathcal{C}}(\mathbf{u})$ is finite for every $\mathbf{u} \in \mathcal{C}$.
\end{defn}
\section{General Empirical Process}\label{sa-sec: General Empirical Process}
Recall that $\mathbf{x}_i\in \mathcal{X} \subseteq \mathbb{R}^d$, $i=1,\dots,n$, are i.i.d. random vectors supported on a background probability space $(\Omega,\mathcal{F},\mathbbm{P})$, and the general empirical process is
\begin{align*}
X_n(h) = \frac{1}{\sqrt{n}} \sum_{i=1}^n \big( h(\mathbf{x}_i) - \mathbbm{E}[h(\mathbf{x}_i)] \big), \qquad h \in \mathscr{H},
\end{align*}
where $\mathscr{H}$ is a possibly $n$-varying class of functions. As briefly explained after Theorem 1 is presented in the paper, its proof relies on the following decomposition:
\begin{align*}
&\lVert X_n - Z_n^X \rVert_{\mathscr{H}}\\
&\leq \lVert X_n - X_n\circ\pi_{\mathscr{H}_\delta} \rVert_{\mathscr{H}}
+ \lVert X_n - Z_n^X \rVert_{\mathscr{H}_\delta}
+ \lVert Z_n^X\circ\pi_{\mathscr{H}_\delta}-Z_n^X \rVert_{\mathscr{H}}\\
&\leq \lVert X_n - X_n\circ\pi_{\mathscr{H}_\delta} \rVert_{\mathscr{H}}
+ \lVert X_n - \mathtt{\Pi}_{0} X_n \rVert_{\mathscr{H}_{\delta}}
+ \lVert \mathtt{\Pi}_{0} X_n - \mathtt{\Pi}_{0} Z_n^X \rVert_{\mathscr{H}_{\delta}}
+ \lVert \mathtt{\Pi}_{0} Z_n^X - Z_n^X \rVert_{\mathscr{H}_\delta}
+ \lVert Z_n^X\circ\pi_{\mathscr{H}_\delta}-Z_n^X \rVert_{\mathscr{H}},
\end{align*}
where $\mathscr{H}_\delta$ denotes a discretization (or meshing) of $\mathscr{H}$ (i.e., $\delta$-net of $\mathscr{H}$), and the terms $\lVert X_n - X_n\circ\pi_{\mathscr{H}_\delta} \rVert_{\mathscr{H}}$ and $\lVert Z_n^X\circ\pi_{\mathscr{H}_\delta}-Z_n^X \rVert_{\mathscr{H}}$ capture the fluctuations (or oscillations) of $X_n$ and $Z_n^X$ relative to the meshing for each of the stochastic processes. These terms are handled using standard arguments for empirical processes. Then, following \cite{Rio_1994_PTRF-SA}, the term $\lVert X_n - Z_n^X \rVert_{\mathscr{H}_\delta}$ is further decomposed into three terms: $\lVert \mathtt{\Pi}_{0} X_n - \mathtt{\Pi}_{0} Z_n^X \rVert_{\mathscr{H}_{\delta}}$ and $\lVert \mathtt{\Pi}_{0} Z_n^X - Z_n^X \rVert_{\mathscr{H}_\delta}$ represent a mean square projection onto a Haar function space, where $\mathtt{\Pi}_{0} X_n(h) = X_n \circ \mathtt{\Pi}_{0} h$ with $\mathtt{\Pi}_{0}$ the $L_2$ projection onto piecewise constant functions on a carefully chosen partition of $\mathcal{X}$, while the final term $\lVert \mathtt{\Pi}_{0} X_n - \mathtt{\Pi}_{0} Z_n^X \rVert_{\mathscr{H}_{\delta}}$ captures the coupling between the projected empirical process and the projected Gaussian process (on a $\delta$-net of $\mathscr{H}$, after the $L_2$ projection).
The proof of Theorem 1 first constructs the Gaussian process $(Z_n^X(h): h \in \mathscr{H})$ on a possibly enlarged probability space supporting the empirical process $(X_n(h): h \in \mathscr{H})$, and then bounds each of the five error terms described above. The proof is given in Section \ref{sa-sec: X-Process -- Proof of Main Theorem}, and it exploits the existence of a surrogate measure and normalizing transformation (Section \ref{sa-sec: X-Process -- normalizing transformation}), along with a collection preliminary technical results (Section \ref{sa-sec: X-process -- Preliminary Technical Results}) that may be of independent interest. More specifically, our preliminary technical results are organized as follows:
\begin{itemize}[leftmargin=*]
\item Section \ref{sa-sec: X-process -- Cell Expansions} introduces a class of recursive quasi-dyadic cells expansion of $\mathcal{X}$, which we employ to generalize prior dyadic cell results in the literature.
\item Section \ref{sa-sec: X-process -- Projection onto Piecewise Constant Functions} introduces the $L_2$ projection onto piecewise constant functions, which can be written as a linear combination of the Haar basis based on the cells. As a consequence, the empirical process indexed by $L_2$-projected functions can be written as linear combinations of counts of i.i.d. data.
\item Section~\ref{sa-sec: X-process -- Strong Approximation Constructions} constructs the Gaussian process $(Z_n^X(h): h \in \mathscr{H})$. Since the constant approximation within each recursive partitioning cell generates counts based on i.i.d. data, the construction boils down to coupling binomial random variables with Gaussian random variables. The celebrated Tusn\'ady's inequality couples $\mathsf{Bin}(n,\frac{1}{2})$ with $\mathsf{Normal}(\frac{n}{2},\frac{n}{4})$, and gives an almost sure bound on the coupling error. In particular, the Gaussian random variable is given by a quantile transformation of the binomial random variable. Building on the quantile transformation idea, our Lemma~\ref{sa-lem: binom coupling} studies the coupling between $\mathsf{Bin}(n,p)$ and $\mathsf{Normal}(np,np(1-p))$, with the error bound given on a high probability set. Due to the dyadic correlation structure, a conditional quantile transformation is used to generated the Binomial--Gaussian pairs down the dyadic cells. Since the constructed Gaussian random variables have a joint distribution that coincides with the Brownian bridge integrated on cells, the Skorohod embedding lemma \citep[Lemma 3.35]{dudley2014uniform} is then used to construct the Brownian bridge $(Z_n^X(h): h \in \mathscr{H})$ on a possibly enriched probability space supporting the data distribution.
\item Section~\ref{sa-sec: X-process -- Meshing Error} handles the meshing errors $\lVert X_n - X_n\circ\pi_{\mathscr{H}_\delta} \rVert_{\mathscr{H}}$ and $\lVert Z_n^X\circ\pi_{\mathscr{H}_\delta}-Z_n^X \rVert_{\mathscr{H}}$ using standard empirical process results, which give the contribution $\mathsf{F}(\delta)$ emerging from Talagrand's inequality \citep[Theorem 3.3.9]{Gine-Nickl_2016_Book-SA} combined with a standard maximal inequality \citep[Theorem 5.2]{chernozhukov2014gaussian-SA}. This allows us to focus on the error on the $\delta$-net to study $\lVert X_n - Z_n^X \rVert_{\mathscr{H}_\delta}$.
\item Section~\ref{sa-sec: X-process -- SA Errors} handles the strong approximation error $\lVert \mathtt{\Pi}_{0} X_n - \mathtt{\Pi}_{0} Z_n^X \rVert_{\mathscr{H}_{\delta}}$. Building on the Tusn\'ady's Lemma, \citet[Theorem 2.1]{Rio_1994_PTRF-SA} established a remarkable coupling result for bounded functions $L_2$-projected on a dyadic cells expansion of $\mathcal{X}$. Our Lemma~\ref{sa-lem: X-process -- SA error} builds on his powerful ideas, and establishes an analogous result for the case of Lipschitz functions $L_2$-projected on dyadic cells expansions of $\mathcal{X}$, thereby obtaining a tighter coupling error. A limitation of these results is that they only apply to a dyadic cell expansion due to the specifics of Tusn\'ady's Lemma. Leveraging the coupling between $\mathsf{Bin}(n,p)$ and $\mathsf{Normal}(np,np(1-p))$, our Lemma~\ref{sa-lem: X-process -- SA error quasi dyadic} established a coupling result for bounded functions $L_2$-projected on a quasi-dyadic cells, although the result is restricted to a high probability event.
\item Section~\ref{sa-sec: X-process -- proj error} handles the $L_2$-projection errors $\lVert X_n - \mathtt{\Pi}_{0} X_n \rVert_{\mathscr{H}_{\delta}}$ and $\lVert \mathtt{\Pi}_{0} Z_n^X - Z_n^X \rVert_{\mathscr{H}_\delta}$ using Bernstein inequality, and taking into account explicitly the potential Lipschitz structure of the functions as well as the generic cell structure.
\end{itemize}
Section~\ref{sa-sec: X-Process -- normalizing transformation} introduces a reduction argument via the surrogate measure and the normalizing transformation in order to apply the preliminary technical results from Section \ref{sa-sec: X-process -- Preliminary Technical Results} to prove Theorem 1. Specifically, the surrogate measure and normalizing transformation reduce the problem to the case where $\mathbf{x}_i \thicksim \mathsf{Uniform}([0,1]^d)$. Section~\ref{sa-sec: X-Process -- Proof of Main Theorem} gives the proof of Theorem 1. Section~\ref{sa-sec: X-process -- Additional Results} presents additional results of independent interest, which are used in Section~\ref{sa-sec: X-process -- Proofs of Corollaries} to prove the results discussed in Section 3.2 of the paper. Finally, Section~\ref{sa-sec: X-process -- KDE Example} provides technical details underlying Example 1 in the paper.
\subsection{Preliminary Technical Results}\label{sa-sec: X-process -- Preliminary Technical Results}
This section presents preliminary technical results that are used to prove Theorem 1. Whenever possible, these results are presented at a higher level of generality, and therefore may be of independent theoretical interest. Throughout this section, we employ the following assumption.
\begin{Assumption}\label{sa-assump: X-process -- step}
Suppose $(\mathbf{x}_i: 1 \leq i \leq n)$ are i.i.d. random vectors taking values in $(\mathbb{R}^d, \mathcal{B}(\mathbb{R}^d))$ with common law $\mathbbm{P}_X$ supported on $\mathcal{X} \subseteq \mathbb{R}^d$, and the following conditions hold.
\begin{enumerate}[label=\emph{(\roman*)}]
\item $\mathscr{H}$ is a real-valued pointwise measurable class of functions on $(\mathbb{R}^d,\mathcal{B}(\mathbb{R}^d),\mathbbm{P}_X)$.
\item $\mathtt{M}_{\mathscr{H},\mathcal{X}} < \infty$ and $J_\mathcal{X}(1, \mathscr{H}, \mathtt{M}_{\mathscr{H},\mathcal{X}}) < \infty$.
\end{enumerate}
\end{Assumption}
Compared to the assumptions in Theorem 1, this assumption does not require the existence of a surrogate measure and normalizing transformation. It will be applied in the analysis of each term in the error decomposition, where we work with the $\mathbbm{P}_X$ distribution. Section~\ref{sa-sec: X-Process -- normalizing transformation} illustrates how the normalizing transformation enables the use of the surrogate measure $\mathbb{Q}_{\mathcal{H}}$, providing greater flexibility in the data generating process. This reduction through the normalizing transformation is a crucial step in the proof of Theorem 1 (Section~\ref{sa-sec: X-Process -- Proof of Main Theorem}).
\subsubsection{Cells Expansions}\label{sa-sec: X-process -- Cell Expansions}
We introduce two definitions of quasi-dyadic cells expansions. Recall that $\mathcal{I}_K = \{(j,k) \in \mathbb{N} \times \mathbb{N}: 1 \leq j \leq K, 0 \leq k < 2^{K - j}\}$ and $\mathcal{J}_K = \{(j,k) \in \mathbb{N} \times \mathbb{N}: 0 \leq j \leq K, 0 \leq k < 2^{K - j}\}$.
\begin{defn}[Quasi-Dyadic Expansion] \label{sa-defn: quasi dyadic expansion}
A collection of Borel measurable sets in $\mathbb{R}^d$, $\mathscr{C}_{K}(\mathbbm{P}, \rho) = \{\mathcal{C}_{j,k}: (j,k) \in \mathcal{J}_K \}$, is called a quasi-dyadic expansion of depth $K$ with respect to probability measure $\mathbbm{P}$ if the following three conditions hold:
\begin{enumerate}[label=\emph{(\roman*)}]
\item $\mathbbm{P}(\mathcal{C}_{K,0}) = 1$.
\item $\mathcal{C}_{j,k} = \mathcal{C}_{j-1, 2k} \sqcup \mathcal{C}_{j-1,2k+1}$, for all $(j,k) \in \mathcal{J}_K$.
\item $\max_{0 \leq k < 2^{K}}\mathbbm{P}(\mathcal{C}_{0,k})/ \min_{0 \leq k < 2^{K}} \mathbbm{P}(\mathcal{C}_{0,k}) \leq \rho < \infty$.
\end{enumerate}
When $\rho = 1$, $\mathcal{C}_{K}(\mathbbm{P},1)$ is called a dyadic expansion of depth $K$ with respect to $\mathbbm{P}$.
\end{defn}
This definition implies $\frac{1}{2} \frac{2}{1 + \rho} \leq \mathbbm{P}(\mathcal{C}_{j-1,2k})/\mathbbm{P}(\mathcal{C}_{j,k}) \leq \frac{1}{2} \frac{2 \rho}{1 + \rho}$ for all $(j,k) \in \mathcal{I}_K$, since each $\mathcal{C}_{j-1,l}$ is a disjoint union of $2^{j-1}$ cells of the form $\mathcal{C}_{0,k}$, which implies the third condition in Definition \ref{sa-defn: quasi dyadic expansion}. Furthermore, $\mathbbm{P}(\mathcal{C}_{j-1,2k}) = \mathbbm{P}(\mathcal{C}_{j-1,2k+1}) = \frac{1}{2} \mathbbm{P}(\mathcal{C}_{j,k})$ in the special case $\rho = 1$, that is, the child level cells are obtained by splitting the parent level cells dyadically in probability.
The next definition specializes the dyadic expansion scheme to axis-aligned splits.
\begin{defn}[Axis-Aligned Quasi-Dyadic Expansion]
A collection of Borel measurable sets in $\mathbb{R}^d$, $\mathscr{A}_{K}(\mathbbm{P}, \rho) = \{\mathcal{C}_{j,k}: (j,k) \in \mathcal{J}_K \}$, is an axis-aligned quasi-dyadic expansion of depth $K$ with respect to probability measure $\mathbbm{P}$ if it can be constructed via the following procedure:
\begin{enumerate}[label=\emph{(\roman*)}]
\item \emph{Initialization ($q=0$)}: Take $\mathcal{C}_{K-q, 0} = \operatorname{Supp}(\mathbbm{P})$.
\item \emph{Iteration ($q=1,\dots,K$):} Given $\mathcal{C}_{K - l,k}$ for $0 \leq l \leq q-1, 0 \leq k < 2^l$, take $s = (q\mod d) + 1$, and construct $\mathcal{C}_{K - q, 2k} = \mathcal{C}_{K - q + 1,k} \cap \{\mathbf{x} \in \mathbb{R}^d: \mathbf{e}_s^\top\mathbf{x} \leq c_{K - q + 1,k}\}$ and $\mathcal{C}_{K - q, 2k+1} = \mathcal{C}_{K - q + 1,k} \cap \{\mathbf{x} \in \mathbb{R}^d: \mathbf{e}_s^\top\mathbf{x} > c_{K - q + 1,k}\}$ where $c_{K-q+1,k}$ is a number chosen so that $\mathbbm{P}(\mathcal{C}_{K - q,2k})/\mathbbm{P}(\mathcal{C}_{K-q + 1,k}) \in [\frac{1}{1 + \rho}, \frac{\rho}{1 + \rho}]$ for all $0 \leq k < 2^{q-1}$. Continue until the collection $(\mathcal{C}_{0,k}:0 \leq k < 2^{K})$ has been constructed.
\end{enumerate}
If $\rho = 1$ and $\mathbbm{P}$ is continuous, then $\mathscr{A}_{K}(\mathbbm{P}, \rho)$ is unique.
\end{defn}
\subsubsection{Projection onto Piecewise Constant Functions}\label{sa-sec: X-process -- Projection onto Piecewise Constant Functions}
For a quasi-dyadic expansion $\mathscr{C}_{K}(\mathbbm{P}, \rho)$, the span of the Haar basis based on the terminal cells is
\begin{align*}
\mathscr{E}_K = \operatorname{Span} \{\mathbbm{1}_{\mathcal{C}_{0,k}}: 0 \leq k < 2^{K}\}.
\end{align*}
For $h \in L_2(\mathbbm{P})$, the mean square projection of $h$ onto $\mathscr{E}_K$ is
\begin{align*}
\mathtt{\Pi}_{0}(\mathscr{C}_{K}(\mathbbm{P},\rho))[h] = \sum_{0 \leq k < 2^{K}} \frac{\mathbbm{1}_{\mathcal{C}_{0,k}}}{\mathbbm{P}(\mathcal{C}_{0,k})} \int_{\mathcal{C}_{0,k}} h(\mathbf{u}) d \mathbbm{P}(\mathbf{u}).
\end{align*}
Because $\mathtt{\Pi}_{0}(\mathscr{C}_{K}(\mathbbm{P},\rho))[h]$ is a linear combination of Haar functions, we obtain the following orthogonal decomposition.
\begin{lemma}\label{sa-lem: X-process -- haar approax dyadic}
For a quasi-dyadic expansion $\mathscr{C}_{K}(\mathbbm{P}, \rho)$ and any $h\in L_2(\mathbbm{P})$, the mean square projection $\mathtt{\Pi}_{0}(\mathscr{C}_{K}(\mathbbm{P},\rho))[h]$ satisfies
\begin{align*}
\mathtt{\Pi}_{0}(\mathscr{C}_{K}(\mathbbm{P},\rho))[h] = \beta_{K,0}(h) e_{K,0} + \sum_{1 \leq j \leq K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}(h) \widetilde{e}_{j,k},
\end{align*}
where
\begin{gather*}
\beta_{j,k}(h) = \frac{1}{\mathbbm{P}(\mathcal{C}_{j,k})} \int_{\mathcal{C}_{j,k}}h(\mathbf{u}) d \mathbbm{P}(\mathbf{u}), \qquad \widetilde{\beta}_{j,k}(h) = \beta_{j-1,2k}(h) - \beta_{j-1,2k+1}(h),
\end{gather*}
\begin{gather*}
e_{j,k} = \mathbbm{1}_{\mathcal{C}_{j,k}}, \qquad \widetilde{e}_{j,k} = \frac{\mathbbm{P}(C_{j-1,2k+1})}{\mathbbm{P}(C_{j,k})} e_{j-1,2k} - \frac{\mathbbm{P}(\mathcal{C}_{j-1,2k})}{\mathbbm{P}(\mathcal{C}_{j,k})} e_{j-1,2k+1},
\end{gather*}
for all $(j,k) \in \mathcal{I}_K = \{(j,k) \in \mathbb{N} \times \mathbb{N}: 1 \leq j \leq K, 0 \leq k < 2^{K - j}\}$.
\end{lemma}
\begin{myproof}{Lemma \ref{sa-lem: X-process -- haar approax dyadic}}
First, we show that $\{e_{K,0}\} \cup \{\widetilde{e}_{j,k}: (j,k) \in \mathcal{I}_K\}$ is an orthogonal basis. For each $(j,k) \in \mathcal{I}_K$,
\begin{align*}
\langle e_{K,0}, \widetilde{e}_{j,k} \rangle
&= \int_{\mathbb{R}^d} \frac{\mathbbm{P}(\mathcal{C}_{j-1,2k+1})}{\mathbbm{P}(\mathcal{C}_{j,k})} e_{j-1,2k}(\mathbf{u})d \mathbbm{P}(\mathbf{u}) - \int_{\mathbb{R}^d} \frac{\mathbbm{P}(\mathcal{C}_{j-1,2k})}{\mathbbm{P}(\mathcal{C}_{j,k})}e_{j-1,2k+1}(\mathbf{u}) d \mathbbm{P}(\mathbf{u}) \\
&= \frac{\mathbbm{P}(\mathcal{C}_{j-1,2k+1}) \mathbbm{P}(\mathcal{C}_{j-1,2k})}{\mathbbm{P}(\mathcal{C}_{j,k})} - \frac{\mathbbm{P}(C_{j-1,2k})\mathbbm{P}(\mathcal{C}_{j-1,2k+1})}{\mathbbm{P}(\mathcal{C}_{j,k})} = 0,
\end{align*}
where $\langle \cdot,\cdot \rangle$ denotes the inner product on $L_2(\mathbbm{P})$ given by $\langle f,g \rangle = \int_{\mathbb{R}^d}f(\mathbf{u})g(\mathbf{u})d \mathbbm{P}(\mathbf{u})$, $f,g \in L_2(\mathbbm{P})$.
Let $(j_1, k_1), (j_2, k_2) \in \mathcal{I}_K$ such that $(j_1,k_1) \neq (j_2,k_2)$. We show $\langle e_{j_1,k_1}, e_{j_2,k_2}\rangle = 0$ by considering two cases.
\begin{itemize}
\item \textit{Case 1}: $j_1 = j_2$ and $k_1 \neq k_2$, then $\widetilde{e}_{j_1,k_1}$ and $\widetilde{e}_{j_2,k_2}$ have different support, hence $\langle \widetilde{e}_{j_1,k_1}, \widetilde{e}_{j_2,k_2} \rangle = 0$.
\item \textit{Case 2}: $j_1 \neq j_2$ and, without loss of generality, we assume $j_1 < j_2$. By (1) in Definition~\ref{sa-defn: quasi dyadic expansion}, either $\mathcal{C}_{j_1,k_1} \cap \mathcal{C}_{j_2,k_2} = \emptyset$ or $\mathcal{C}_{j_1,k_1} \subset \mathcal{C}_{j_2,k_2}$.
\end{itemize}
In the first case, we also have $\langle \widetilde{e}_{j_1,k_1}, \widetilde{e}_{j_2,k_2}\rangle = 0$. In the second case, using (1) in Definition~\ref{sa-defn: quasi dyadic expansion} again, either $\mathcal{C}_{j_1,k_1} \subseteq \mathcal{C}_{j_2 - 1, 2 k_2}$ or $\mathcal{C}_{j_1, k_1} \subseteq \mathcal{C}_{j_2-1, 2 k_2 + 1}$. Assume, without loss of generality, that $\mathcal{C}_{j_1,k_1} \subseteq \mathcal{C}_{j_2 - 1, 2 k_2}$. Then, for any $(j_1, k_1), (j_2, k_2) \in \mathcal{I}_K$,
\begin{align*}
& \langle \widetilde{e}_{j_1,k_1}, \widetilde{e}_{j_2,k_2}\rangle \\
&= \langle \widetilde{e}_{j_1,k_1}, \frac{\mathbbm{P}(\mathcal{C}_{j_2 - 1, 2k_2})}{\mathbbm{P}(\mathcal{C}_{j_2, k_2})}e_{j_2 - 1, 2 k_2}\rangle \\
&= \frac{\mathbbm{P}(\mathcal{C}_{j_2 - 1, 2k_2})}{\mathbbm{P}(\mathcal{C}_{j_2, k_2})} \bigg[\int_{\mathbb{R}^d} \frac{\mathbbm{P}(\mathcal{C}_{j_1-1,2k_1+1})}{\mathbbm{P}(\mathcal{C}_{j_1,k_1})} e_{j_1-1,2k_1}(\mathbf{u})d \mathbbm{P}(\mathbf{u}) - \int_{\mathbb{R}^d} \frac{\mathbbm{P}(\mathcal{C}_{j_1-1,2 k_1})}{\mathbbm{P}(\mathcal{C}_{j_1,k_1})}e_{j_1-1,2 k_1+1}(\mathbf{u}) d \mathbbm{P}(\mathbf{u}) \bigg] \\
&= 0.
\end{align*}
Thus, $\{e_{K,0}\} \cup \{\widetilde{e}_{j,k}: (j,k)\in\mathcal{I}_K\}$ is an orthogonal basis for $\mathscr{E}_K$, and the $L_2$ projection for all $h \in L_2(\mathbbm{P})$ is
\begin{align*}
\mathtt{\Pi}_{0}(\mathscr{C}_{K}(\mathbbm{P},\rho))[h] = \frac{\langle h, e_{K,0}\rangle}{\langle e_{K,0}, e_{K,0} \rangle} e_{K,0} + \sum_{1 \leq j \leq K} \sum_{0 \leq k < 2^{K - j}} \frac{\langle h, \widetilde{e}_{j,k}\rangle}{\langle \widetilde{e}_{j,k}, \widetilde{e}_{j,k} \rangle} \widetilde{e}_{j,k}.
\end{align*}
For all $(j,k) \in \mathcal{I}_K$, the coefficients are given by
\begin{align*}
\frac{\langle h, \widetilde{e}_{j,k} \rangle}{\langle \widetilde{e}_{j,k}, \widetilde{e}_{j,k} \rangle}
&= \frac{\int_{\mathbb{R}^d} h(\mathbf{u}) \widetilde{e}_{j,k}(\mathbf{u})d \mathbbm{P}(\mathbf{u})}{\int_{\mathbb{R}^d} \widetilde{e}_{j,k}(\mathbf{u}) \widetilde{e}_{j,k}(\mathbf{u})d \mathbbm{P}(\mathbf{u})}\\
&= \frac{ \mathbbm{P}(\mathcal{C}_{j-1,2k+1}) \mathbbm{P}(\mathcal{C}_{j-1,2k}) \mathbbm{P}(\mathcal{C}_{j,k})^{-1}\beta_{j-1,2k}(h) - \mathbbm{P}(\mathcal{C}_{j-1,2k})\mathbbm{P}(\mathcal{C}_{j-1,2k+1})\mathbbm{P}(\mathcal{C}_{j,k})^{-1}\beta_{j-1,2k+1}(h)}{\mathbbm{P}(\mathcal{C}_{j-1,2k+1})^2 \mathbbm{P}(\mathcal{C}_{j-1,2k}) \mathbbm{P}(\mathcal{C}_{j,k})^{-2} + \mathbbm{P}(\mathcal{C}_{j-1,2k})^2 \mathbbm{P}(\mathcal{C}_{j-1,2k+1}) \mathbbm{P}(\mathcal{C}_{j,k})^{-2}} \\
&= \frac{ \mathbbm{P}(\mathcal{C}_{j-1,2k+1}) \mathbbm{P}(\mathcal{C}_{j-1,2k}) \mathbbm{P}(\mathcal{C}_{j,k})^{-1}\beta_{j-1,2k}(h) - \mathbbm{P}(\mathcal{C}_{j-1,2k})\mathbbm{P}(\mathcal{C}_{j-1,2k+1})\mathbbm{P}(\mathcal{C}_{j,k})^{-1}\beta_{j-1,2k+1}(h)}{\mathbbm{P}(\mathcal{C}_{j-1,2k+1}) \mathbbm{P}(\mathcal{C}_{j-1,2k}) \mathbbm{P}(\mathcal{C}_{j,k})^{-1}} \\
&= \beta_{j-1,2k}(h) - \beta_{j-1,2k+1}(h) = \widetilde{\beta}_{j,k}(h).
\end{align*}
Moreover,
\begin{align*}
\frac{\langle h, e_{K,0} \rangle}{\langle e_{K,0}, e_{K,0} \rangle} = \mathbbm{P}(\mathcal{C}_{K,0})^{-1} \int_{\mathcal{C}_{K,0}} h (\mathbf{u}) d \mathbbm{P}(\mathbf{u}) = \beta_{K,0}(h).
\end{align*}
This concludes the proof.
\end{myproof}
To save notation, we will write $\mathtt{\Pi}_{0}$ for $\mathtt{\Pi}_{0}(\mathscr{C}_{K}(\mathbbm{P},\rho))$ whenever the underlying cells expansion is clear from the context. For a class of functions $\mathscr{H}$ on $(\mathbb{R}^d,\mathcal{B}(\mathbb{R}^d),\mathbbm{P})$ such that $\mathscr{H} \subseteq L_2(\mathbbm{P})$, denote $\mathtt{\Pi}_{0} \mathscr{H} = \{\mathtt{\Pi}_{0} h: h \in \mathscr{H}\}$.
\subsubsection{Strong Approximation Constructions}\label{sa-sec: X-process -- Strong Approximation Constructions}
This section employs the notations and conventions introduced in Sections \ref{sa-sec: X-process -- Cell Expansions} and \ref{sa-sec: X-process -- Projection onto Piecewise Constant Functions}. Unless explicitly stated otherwise, we assume a quasi-dyadic expansion $\mathscr{C}_{K}(\mathbbm{P}_X, \rho)$ is given. Let $(\widetilde{\xi}_{j,k}:(j,k) \in \mathcal{I}_K)$ be i.i.d. standard Gaussian random variables. Let $F_{(j,k),m}$ be the cumulative distribution function of $(S_{j,k} - m p_{j,k})/\sqrt{m p_{j,k}(1 - p_{j,k})}$, where $S_{j,k}$ is a $\mathsf{Bin}(m, p_{j,k})$ random variable with $p_{j,k} = \mathbbm{P}_X(\mathcal{C}_{j-1,2k})/\mathbbm{P}_X(\mathcal{C}_{j,k})$, and $G_{(j,k), m}(t) = \inf \{x: F_{(j,k),m}(x) \geq t\}$.
We define the collection of random variables $(U_{j,k}: (j,k) \in \mathcal{J}_K)$ and $(\widetilde{U}_{j,k}: (j,k) \in \mathcal{I}_K)$ via the following iterative scheme:
\begin{enumerate}
\item \emph{Initialization} ($j=K$): $U_{K, 0} = n$.
\item \emph{Iteration} ($j=K,K-1,\dots,1)$: For each $1 \leq j \leq K$, and given $(U_{l,k}:j < l \leq K, 0 \leq k < 2^{K - l})$, solve for $(U_{j,k}:0 \leq k < 2^{K - j})$ such that
\begin{gather}\label{sa-eq: X-process -- conditional quantile transformation}
\nonumber \widetilde{U}_{j,k} = \sqrt{U_{j,k}p_{j,k}(1 - p_{j,k})} G_{(j,k), U_{j,k}} \circ \Phi(\widetilde{\xi}_{j,k}), \\
\nonumber \widetilde{U}_{j,k} = (1 - p_{j,k}) U_{j-1, 2k} - p_{j,k} U_{j-1, 2k+1} = U_{j-1, 2k} - p_{j,k} U_{j,k}, \\
U_{j-1, 2k} + U_{j-1, 2k+1} = U_{j,k},
\end{gather}
where $0 \leq k < 2^{K - j}$. Continue till $(U_{0,k}:0 \leq k < 2^{K})$ are defined.
\end{enumerate}
Then, $(U_{j,k}: (j,k) \in \mathcal{J}_K)$ has the same joint distribution as $(\sum_{i = 1}^n e_{j,k}(\mathbf{x}_i): (j,k) \in \mathcal{J}_K)$ from Lemma \ref{sa-lem: X-process -- haar approax dyadic}. By the Vorob'ev–Berkes–Philipp theorem \citep[Theorem 1.31]{dudley2014uniform}, $(\widetilde{\xi}_{j,k}: (j,k) \in \mathcal{I}_K)$ can be constructed on a possibly enlarged probability space such that the previously constructed $U_{j,k}$ satisfies $U_{j,k} = \sum_{i = 1}^n e_{j,k}(\mathbf{x}_i)$ almost surely for all $(j,k) \in \mathcal{J}_K$. We will show that the $\widetilde{\xi}_{j,k}$'s can be given as a Brownian bridge indexed by $\widetilde{e}_{j,k}$'s from Lemma \ref{sa-lem: X-process -- haar approax dyadic}. Recall the definitions given in Section \ref{sa-sec: X-process -- Projection onto Piecewise Constant Functions}.
\begin{lemma}\label{sa-lem: X-process -- pregaussian}
Suppose Assumption~\ref{sa-assump: X-process -- step} holds, and a quasi-dyadic expansion $\mathscr{C}_{K}(\mathbbm{P}_X,\rho)$ is given. Then, $\mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H} \subseteq L_2(\mathbbm{P}_X)$ and is $\mathbbm{P}_X$-pregaussian.
\end{lemma}
\begin{myproof}{Lemma \ref{sa-lem: X-process -- pregaussian}}
To simplify notation, the parameters of $\mathscr{H}$ (Definitions 4 to 12) are taken with $\mathcal{C} = \mathcal{X}$, and the index $\mathcal{C}$ is omitted. Since $\mathtt{M}_{\mathscr{H}} < \infty$, $\mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H} \subseteq L_2(\mathbbm{P}_X)$. Definition of $\mathtt{\Pi}_{0}$ from Section~\ref{sa-sec: X-process -- Projection onto Piecewise Constant Functions} implies that $\mathtt{M}_{\mathscr{H}}$ is an envelope for $\mathtt{\Pi}_{0} \mathscr{H}$.
\underline{Claim}: For all $0 < \delta < 1$, $J(\mathtt{\Pi}_{0} \mathscr{H}, \mathtt{M}_{\mathscr{H}}, \delta) \leq J(\mathscr{H}, \mathtt{M}_{\mathscr{H}}, \delta)$.
\underline{Proof of Claim}: Let $Q$ be a finite discrete measure on $\mathcal{X}$. Let $f,g \in \mathscr{H}$. Then, by the definition of $\mathtt{\Pi}_{0}$ and Jensen's inequality,
\begin{align*}
\lVert \mathtt{\Pi}_{0} f - \mathtt{\Pi}_{0} g \rVert_{Q,2}^2
& \leq \sum_{0 \leq k < 2^{K}} Q(\mathcal{C}_{0,k}) 2^{K}\int_{\mathcal{C}_{0,k}} (f - g)^2 d\mathbbm{P}_X.
\end{align*}
Define a measure $\widetilde{Q}$ such that for any $A \in \mathcal{B}(\mathbb{R}^d)$, $\widetilde{Q}(A) = \sum_{0 \leq k < 2^{K}} Q(\mathcal{C}_{0,k}) 2^{K} \mathbbm{P}_X(A \cap \mathcal{C}_{0,k})$, then
\begin{align*}
\lVert \mathtt{\Pi}_{0} f - \mathtt{\Pi}_{0} g \rVert^2_{Q,2} \leq \lVert f - g \rVert_{\widetilde{Q},2}^2.
\end{align*}
Take $\mathscr{L}$ to be a $\delta \mathtt{M}_{\mathscr{H}}$-net of $\mathscr{H}$ over $\mathcal{X}$ with respect to $\lVert \cdot \rVert_{\widetilde{Q},2}$ with cardinality no greater than $\mathtt{N}_{\mathscr{H}}(\delta,\mathtt{M}_{\mathscr{H}})$. Let $\mathtt{\Pi}_{0} f$ be in an arbitrary function in $\mathtt{\Pi}_{0} \mathscr{H}$, there exists $g \in \mathcal{L}$ such that $\lVert \mathtt{\Pi}_{0} f - \mathtt{\Pi}_{0} g \rVert_{Q,2}^2 \leq \lVert f - g \rVert_{\widetilde{Q},2}^2 \leq \delta^{2} \mathtt{M}_{\mathscr{H}}^2$. The claim then follows.
It follows from the claim and (ii) from Assumption~\ref{sa-assump: X-process -- step} that $J(1,\mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H}, \mathtt{M}_{\mathscr{H}}) < \infty$. By Dominated Convergence Theorem, $\lim_{\delta \downarrow 0}J(\delta,\mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H}) < \infty$. Since $\mathtt{M}_{\mathscr{H}} < \infty$, $\mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H}$ is totally bounded with respect to $\lVert \cdot \rVert_{\mathbbm{P}_X,2}$. By separability of $\mathscr{H}$ and \citet[Corollary 2.2.9]{wellner2013weak-SA}, $\mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H}$ is $\mathbbm{P}_X$-pregaussian.
\end{myproof}
Under the conditions of Lemma~\ref{sa-lem: X-process -- pregaussian}, take $(Z_n^X(h): h \in \mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H})$ to be a $\mathbbm{P}_X$-Brownian bridge such that $Z_n^X(\cdot) \in C(\mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H},\mathfrak{d}_{\mathbbm{P}_X})$ almost surely. Since $(Z_n^X(\widetilde{e}_{j,k}):(j,k) \in \mathcal{I}_K)$ are independent random variables with distribution $\mathsf{Normal}(0,\mathbbm{P}_X(\mathcal{C}_{j-1,2k})\mathbbm{P}_X(\mathcal{C}_{j-1,2k+1})\mathbbm{P}_X(\mathcal{C}_{j,k})^{-1})$ for $(j,k) \in \mathcal{I}_K$, by Skorohod Embedding lemma \citep[Lemma 3.35]{dudley2014uniform}, on a possibly enlarged probability space, the Brownian bridge $(Z_n^X(h): h \in \mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H})$ can be constructed such that it satisfies
\begin{align}\label{sa-eq: X-process -- Gaussian Process}
\widetilde{\xi}_{j,k} = \sqrt{\frac{\mathbbm{P}_X(\mathcal{C}_{j,k})}{\mathbbm{P}_X(\mathcal{C}_{j-1,2k})\mathbbm{P}_X(\mathcal{C}_{j-1,2k+1})}} Z_n^X(\widetilde{e}_{j,k}),
\end{align}
for all $(j,k) \in \mathcal{I}_K$. Moreover, for all $g \in \mathtt{\Pi}_{0} \mathscr{H}$,
\begin{align*}
\sqrt{n} X_n(g) & = \sum_{j = 1}^{K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}(g) \widetilde{U}_{j,k} \qquad\text{and}\qquad
\sqrt{n} Z_n^X(g) = \sum_{j = 1}^{K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}(g) \widetilde{V}_{j,k},
\end{align*}
where $\widetilde{V}_{j,k} = \sqrt{n}Z_n^X(\widetilde{e}_{j,k})$ for all $(j,k) \in \mathcal{I}_K$. The difference between $X_n(g)$ and $Z_n^X(g)$, for all $g \in \mathtt{\Pi}_{0} \mathscr{H}$, will rely on the coefficient $(\widetilde{\beta}_{j,k}(g):(j,k)\in \mathcal{I}_K, g \in \mathtt{\Pi}_{0} \mathscr{H})$ and the coupling between $\widetilde{U}_{j,k}$ and $\widetilde{V}_{j,k}$, which is the essence of Theorem 2.1 in \cite{Rio_1994_PTRF-SA}. Although \citet[Theorem 2.1]{Rio_1994_PTRF-SA} is stated for i.i.d. $\mathsf{Uniform}([0,1])$ random variables, the underlying process only depends through the counts of the random variables taking values in each interval of the form $[k2^{-j}, (k+1)2^{-j})$ for $(j,k) \in \mathcal{J}_K$, which have the same distribution as the counts $(\sum_{i=1}^n \mathbbm{1}(\mathbf{x}_i\in\mathcal{C}_{j,k}):(j,k) \in \mathcal{J}_K)$. Therefore, we have the following corollary of \citet[Theorem 2.1]{Rio_1994_PTRF-SA} under Assumption~\ref{sa-assump: X-process -- step}. Recall the definitions given in Section \ref{sa-sec: X-process -- Projection onto Piecewise Constant Functions}.
\begin{lemma}\label{sa-lem: X-process -- sa for pcw-const}
Suppose Assumption~\ref{sa-assump: X-process -- step} holds, a dyadic expansion $\mathscr{C}_{K}(\mathbbm{P}_X, 1)$ is given, and $(Z_n^X(h): h \in \mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H})$ is the Gaussian process constructed as in \eqref{sa-eq: X-process -- Gaussian Process} on a possibly enlarged probability space. Then, for any $g \in \mathtt{\Pi}_{0} \mathscr{H}$ and any $t > 0$,
\begin{align*}
\mathbbm{P} \left(\sqrt{n}\left|X_n(g) - Z_n^X(g)\right|
\geq 24 \sqrt{\lVert g \rVert_{\mathscr{E}_{K}}^2 t} + 4 \sqrt{\mathtt{C}_{\{g\},K}}t \right) \leq 2 \exp(-t),
\end{align*}
where
\begin{align*}
\lVert g \rVert_{\mathscr{E}_{K}}^2 = \sum_{j = 1}^{K} \sum_{0 \leq k < 2^{K - j}} |\widetilde{\beta}_{j,k}(g)|^2
\end{align*}
using the definitions in Lemma \ref{sa-lem: X-process -- haar approax dyadic}, and
\begin{align*}
\mathtt{C}_{\mathscr{F},K} = \sup_{f \in \mathscr{F}}\min\bigg\{\sup_{(j,k) \in \mathcal{I}_K} \bigg[\sum_{1 \leq l < j} (j-l)(j-l+1) 2^{l-j} \sum_{0 \leq m < 2^{K - l}: \mathcal{C}_{l,m} \subseteq \mathcal{C}_{j,k}} \widetilde{\beta}_{l,m}^2(f) \bigg] + \mathtt{M}_{\{f\},\mathcal{C}_{K,0}}^2, K \mathtt{M}_{\{f\},\mathcal{C}_{K,0}}^2 \bigg\},
\end{align*}
for any $\mathscr{F} \subseteq \mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H}$.
\end{lemma}
\begin{myproof}{Lemma \ref{sa-lem: X-process -- sa for pcw-const}}
Let $(w_i:1 \leq i \leq n)$ be i.i.d. $\mathsf{Uniform}([0,1])$, and $I_{j,k} = [k 2^{-j}, (k+1) 2^{-j})$ for $(j,k) \in \mathcal{J}_K$. Take $B$ to be a Brownian bridge on $[0,1]$, that is, there exists a standard Wiener process $W$ such that $B(t) = W(t) - t W(1)$ for all $t \in [0,1]$. Take
\begin{align*}
v_{j,k} = \sqrt{n}\int_{0}^1 \mathbbm{1}(t \in I_{j,k}) d B(t), \qquad & (j,k) \in \mathcal{J}_K, \\
\widetilde{v}_{j,k} = v_{j-1,2k} - v_{j-1,2k+1}, \qquad & (j,k) \in \mathcal{I}_K.
\end{align*}
Take $F_{m}$ to be the cumulative distribution function of $(S_m - \frac{1}{2}m)/\sqrt{m/4}$, where $S_m$ is a $\mathsf{Bin}(m, 1/2)$ random variable, and $G_{m}(t) = \inf \{x: F_{m}(x) \geq t\}$. Define $u_{j,k}$'s and $\widetilde{u}_{j,k}$'s via the iterative quantile transformation:
\begin{enumerate}
\item \emph{Initialization}: $u_{K, 0} = n$.
\item \emph{Iteration}: For each $0\leq j \leq K-1$, and given $(u_{l,k}:0 \leq k < 2^{K - l}, j < l \leq K)$, then solve for $(u_{j,k}:0 \leq k < 2^{K - j})$ such that
\begin{align*}
\widetilde{u}_{j,k} &= \frac{1}{2}\sqrt{u_{j,k}} G_{u_{j,k}} \circ \Phi(\widetilde{\xi}_{j,k}), \\
\widetilde{u}_{j,k} &= \frac{1}{2} u_{j-1, 2k} - \frac{1}{2} u_{j-1, 2k+1} = u_{j-1, 2k} - \frac{1}{2} u_{j,k}, \\
u_{j-1, 2k} + u_{j-1, 2k+1} &= u_{j,k},
\end{align*}
for $0 \leq k < 2^{K - j}$. Continue till $(u_{0,k}:0 \leq k < 2^{K})$ are defined.
\end{enumerate}
Then $u_{j,k}$'s have the same joint distribution as $\sum_{i = 1}^n \mathbbm{1}(w_i \in I_{j,k})$'s. Hence, by Skorohod Embedding lemma \citep[Lemma 3.35]{dudley2014uniform}, on a rich enough probability space, we can take $(B(t): 0 \leq t \leq 1)$ such that $u_{j,k} = \sum_{i = 1}^n \mathbbm{1}(w_i \in I_{j,k})$ for all $(j,k) \in \mathcal{J}_K$, almost surely.
Observe $\{(\widetilde{u}_{j,k}, \widetilde{v}_{j,k}): (j,k) \in \mathcal{I}_K \}$ and $\{(\widetilde{U}_{j,k}, \widetilde{V}_{j,k}): (j,k) \in \mathcal{I}_K \}$ have the same joint distribution, and
\begin{align*}
(X_n(g), Z_n^X(g)) = \bigg(\frac{1}{\sqrt{n}} \sum_{j = 1}^{K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}(g) \widetilde{U}_{j,k}, \frac{1}{\sqrt{n}} \sum_{j = 1}^{K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}(g) \widetilde{V}_{j,k}\bigg),
\end{align*}
for all $g \in \mathtt{\Pi}_{0} \mathscr{H}$. Thus the distribution of the process $\{(X_n(g), Z_n^X(g)): g \in \mathtt{\Pi}_{0} \mathscr{H}\}$ is the same as distribution of
\begin{align*}
\big((x_n(g),z_n(g)): g \in \mathtt{\Pi}_{0} \mathscr{H}\big)
= \bigg(\bigg(\frac{1}{\sqrt{n}} \sum_{j = 1}^{K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}(g) \widetilde{u}_{j,k}, \frac{1}{\sqrt{n}} \sum_{j = 1}^{K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}(g) \widetilde{v}_{j,k}\bigg): g \in \mathtt{\Pi}_{0} \mathscr{H} \bigg),
\end{align*}
We can then apply \citet[Theorem 2.1]{Rio_1994_PTRF-SA} on $((x_n(g),z_n(g)): g \in \mathtt{\Pi}_{0} \mathscr{H})$ and use its equi-distribution as $((X_n(g),Z_n^X(g)): g \in \mathtt{\Pi}_{0} \mathscr{H})$ to get for any $\mathbf{p} = (p_1,\cdots,p_K)$ with positive components such that $\sum_{i = 1}^K p_i \leq 1$, if we take $q_i = (2^i p_i)^{-1}$ and
\begin{align*}
M(\mathbf{p},g) = 4 \sup_{(j,k) \in \mathcal{I}_K} \bigg[\sum_{1 \leq l < j} q_{j-l} \sum_{0 \leq m < 2^{K - l}: \mathcal{C}_{l,m} \subseteq \mathcal{C}_{j,k}} \widetilde{\beta}_{l,m}^2(g) \bigg], \qquad g \in \mathtt{\Pi}_{0} \mathscr{H},
\end{align*}
then for any $t > 0$ and $g \in \mathtt{\Pi}_{0} \mathscr{H}$,
\begin{align*}
\mathbbm{P} \bigg(\sqrt{n}\big|X_n(g) - Z_n^X(g)\big| \geq (\sqrt{M(\mathbf{p},g)} + \mathtt{M}_{\{g\},\mathcal{C}_{K,0}})t + \Big(\Big(\sum_{i = 1}^K q_i/2\Big)^{1/2}+ 3\Big)\lVert g \rVert_{\mathscr{E}_{K}}\sqrt{t} \bigg) \leq 2 \exp(-t).
\end{align*}
Following \citet[Section 3]{Rio_1994_PTRF-SA}, we choose either $p_i = \frac{1}{2}\big(\frac{1}{K} + \frac{1}{i(i+1)}\big)$ to get
\begin{align*}
M(\mathbf{p},g) \leq 8 K \mathtt{M}_{\{g\},\mathcal{C}_{K,0}} \qquad \text{ and } \qquad \sum_{i = 1}^K \frac{q_i}{2} < 8,
\end{align*}
or $p_i = \frac{1}{i(i+1)}$ to get
\begin{align*}
M(\mathbf{p},g) \leq \sup_{(j,k) \in \mathcal{I}_K} \bigg[\sum_{1 \leq l < j} (j-l)(j-l+1) 2^{l-j} \sum_{0 \leq m < 2^{K - l}: \mathcal{C}_{l,m} \subseteq \mathcal{C}_{j,k}} \widetilde{\beta}_{l,m}^2(g) \bigg] \text{ and } \sum_{i = 1}^K \frac{q_i}{2} < 4.
\end{align*}
The conclusion then follows.
\end{myproof}
Lemma \ref{sa-lem: X-process -- sa for pcw-const} relies on a coupling of $\mathsf{Bin}(m,1/2)$ random variables with Gaussian random variables. A weaker coupling also holds for $\mathsf{Bin}(m,p)$ with the error term only depending on how far away $p$ is bounded away from $0$ and $1$, as the following lemma establishes.
\begin{lemma}\label{sa-lem: binom coupling}
Suppose $X \thicksim \mathsf{Bin}(m,p)$ where $0 < \underline{p} < p < \overline{p} < 1$. Then, there exists a a random variable $Z \thicksim \mathsf{Normal}(0,1)$, and constants $c_0, c_1, c_2, c_3 > 0$ only depending on $\underline{p}$ and $\overline{p}$, such that whenever the event $A = \{|X - m p| \leq c_1 m\}$ occurs and $c_0 \sqrt{m} \geq 1$, we have
\begin{align*}
\left|X - m p - \sqrt{m p (1 - p)}Z \right| \leq c_2 Z^2 + c_3
\qquad\text{and}\qquad
\left|X - m p \right| \leq \frac{1}{c_0} + 2 \sqrt{m p (1 - p)}|Z|.
\end{align*}
In particular, we can take $c_0 > 0$ to be the solution of
\begin{align*}
60 c_0 \overline{p} \left(\sqrt{\frac{1 - \underline{p}}{\underline{p}}}\right)^3 \exp\left(2\sqrt{\frac{1 - \underline{p}}{\underline{p}}}c_0\right) + 60 c_0 (1 - \underline{p}) \left(\sqrt{\frac{\overline{p}}{1 - \overline{p}}}\right)^3 \exp\left(2\sqrt{\frac{\overline{p}}{1 - \overline{p}}}c_0\right) = 1,
\end{align*}
and $c_1 = 15 c_0 \sqrt{\underline{p}(1 - \overline{p})}$, $c_2 = 1/(15 c_0)$, $c_3 = 1/ c_0$, and then set
\begin{align*}
Z = \Phi^{-1} \circ F \Big((X - m p)/\sqrt{m p(1-p)}\Big).
\end{align*}
That is, $Z$ can be taken via the quantile transformation based on $F(x) = \mathbbm{P}(X -m p < \sqrt{m p (1-p)}x)$.
\end{lemma}
\begin{myproof}{Lemma \ref{sa-lem: binom coupling}}
Let $(X_j:1\leq j \leq m)$ be i.i.d. $\mathsf{Bern}(p)$ with $0 < \underline{p} < p < \overline{p} < 1$. Take $\xi_j = (X_j - p)/\sqrt{m p(1-p)}$ and $S_m = \sum_{j = 1}^m \xi_j$. Then, for any $a \in \mathbb{R}$,
\begin{align*}
L(a)
&= \sum_{j = 1}^m \mathbbm{E} \left[|\xi_j|^3 \exp(|a \xi_j|) \right]
= \sum_{j = 1}^m \mathbbm{E} \bigg[\bigg|\frac{X_j - p}{\sqrt{m p (1-p)}}\bigg|^3 \exp \bigg(a\Big|\frac{X_j - p}{\sqrt{m p (1 - p)}}\Big|\bigg) \bigg] \\
&= m p \bigg(\frac{1 - p}{\sqrt{m p (1-p)}}\bigg)^3 \exp \bigg(a\frac{1 - p}{\sqrt{m p (1 - p)}}\bigg) + m (1 - p) \bigg(\frac{p}{\sqrt{m p (1-p)}}\bigg)^3 \exp \bigg(a \frac{p}{\sqrt{m p (1 - p)}}\bigg).
\end{align*}
Take $c_0 > 0$ such that
\[60 c_0 \overline{p} \left(\sqrt{\frac{1 - \underline{p}}{\underline{p}}}\right)^3 \exp\left(2\sqrt{\frac{1 - \underline{p}}{\underline{p}}}c_0\right) + 60 c_0 (1 - \underline{p}) \left(\sqrt{\frac{\overline{p}}{1 - \overline{p}}}\right)^3 \exp\left(2\sqrt{\frac{\overline{p}}{1 - \overline{p}}}c_0\right) = 1.\]
Then, for any $m \in \mathbb{N}$ and $\lambda = c_0 \sqrt{m}$, we have $60 \lambda L(2 \lambda) \leq 1$.
\citet[Lemma 2]{sakhanenko1996estimates-SA} implies that, whenever $c_0 \sqrt{m} \geq 1$ and the event $\{|S_m| < c_0 \sqrt{m}\}$ occurs,
\begin{align*}
\left|S_m - Z \right| \leq \frac{1}{c_0 \sqrt{m}} + \frac{S_n^2}{60 c_0 \sqrt{m}}.
\end{align*}
Moreover, $Z$ can be taken such that $Z = \Phi^{-1} \circ F(S_m)$.
We then proceed as in the proof for Lemma 2 in \cite{brown2010nonparametric-SA}, where they show for each $0 < p < 1$, the coupling exits with $c_0$ to $c_3$ not depending on $m$, though they did not give explicit dependency of $c_0$ to $c_3$ on $p$. Take $c_1$ such that $c_1/(60 c_0) < 1/2$. In particular, we can take $c_1 = 15 c_0$. Then, on the event $\{|S_m| < c_1 \sqrt{m}\}$,
\begin{align*}
|S_m - Z| \leq \frac{1}{c_0 \sqrt{m}} + |S_m| \frac{c_1 \sqrt{m}}{60 c_0 \sqrt{m}} \leq \frac{1}{c_0 \sqrt{m}} + \frac{1}{2}|S_m|.
\end{align*}
Hence, by triangle inequality, $|S_m| \leq \frac{2}{c_0 \sqrt{m}} + 2 |Z|$, and
\begin{align*}
|S_m - Z| \leq \frac{1}{c_0 \sqrt{m}} + \frac{1}{60 c_0 \sqrt{m}} \left(\frac{2}{c_0 \sqrt{m}} + 2 |Z|\right)^2 \leq \frac{2}{c_0 \sqrt{m}} + \frac{2}{15 c_0 \sqrt{m}}|Z|^2.
\end{align*}
Since $X = \sum_{j = 1}^m X_j \thicksim \mathsf{Bin}(m,p)$, whenever the event $\Big\{|X - m p| < c_1 m \sqrt{\underline{p}(1 - \overline{p})}\Big\}$ occurs and $c_0 \sqrt{m} \geq 1$,
\[\left|X - mp - \sqrt{m p (1 - p)}Z\right| \leq \frac{2}{c_0} \sqrt{p (1 - p)} + \frac{2}{15 c_0}\sqrt{p(1-p)}|Z|^2 \leq \frac{1}{c_0} + \frac{Z^2}{15 c_0}.\]
Moreover, $|S_m| \leq \frac{2}{c_0 \sqrt{m}} + 2 |Z|$ implies $|X - m p | \leq \frac{1}{c_0} + 2 \sqrt{m p (1 - p)}|Z|$.
\end{myproof}
This generalization of Tusn\'ady's Lemma enables the following strong approximation for the case of a quasi-dyadic cells expansion.
\begin{lemma}\label{sa-lem: X-process -- sa for pcw-const non-dyadic}
Suppose Assumption~\ref{sa-assump: X-process -- step} holds, a quasi-dyadic expansion $\mathscr{C}_{K}(\mathbbm{P}_X, \rho)$ is given with $\rho > 1$, and $(Z_n^X(h): h \in \mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H})$ is the Gaussian process constructed as in \eqref{sa-eq: X-process -- Gaussian Process} on a possibly enlarged probability space. Then, for any $g \in \mathtt{\Pi}_{0} \mathscr{H}$ and for any $t > 0$,
\begin{equation*}
\begin{aligned}
\mathbbm{P} \left(\sqrt{n}\left|X_n(g) - Z_n^X(g)\right| \geq c_{\rho} \sqrt{\lVert g \rVert_{\mathscr{E}_{K}}^2 t} + c_{\rho} \sqrt{\mathtt{C}_{\{g\},K}}t \right)
\leq 2 \exp(-t) + 2^{K + 2} \exp \big(- c_{\rho} n 2^{-K}\big),
\end{aligned}
\end{equation*}
where $c_{\rho}$ is a constant that only depends on $\rho$, and $\lVert g \rVert_{\mathscr{E}_{K}}^2$ and $\mathtt{C}_{\{g\},K}$ are defined in Lemma~\ref{sa-lem: X-process -- sa for pcw-const}.
\end{lemma}
\begin{myproof}{Lemma \ref{sa-lem: X-process -- sa for pcw-const non-dyadic}}
We adopt the coupling method from Section 2 of \cite{Rio_1994_PTRF-SA}, extending it to accommodate quasi-dyadic cells. Instead of applying the well-known Tusnády inequality as in \cite{Rio_1994_PTRF-SA}, which states that for \( X \sim \mathsf{Bin}(m,\frac{1}{2}) \), there exists \( Z \sim \mathsf{Normal}(0,1) \) such that almost surely:
\[
\left|X - \frac{m}{2} - \left(\frac{\sqrt{m}}{2}\right)Z\right| \leq 1 + \frac{Z^2}{8}, \quad \text{and} \quad |X - \frac{m}{2}| \leq 1 + \frac{\sqrt{m}}{2}|Z|,
\]
we rely on Lemma~\ref{sa-lem: binom coupling}, which allows for coupling in the case of \( \mathsf{Bin}(m,p) \) with \( p \neq \frac{1}{2} \), though restricted to a high-probability set. The proof proceeds in two parts: Part 1 establishes an upper bound for the small-probability event where the coupling inequalities from Lemma~\ref{sa-lem: binom coupling} fail to hold; Part 2 decomposes the error \( X_n(g) - Z_n^X(g) \) into the coupling errors corresponding to each pair of cells \( (\mathcal{C}_{j-1,2k}, \mathcal{C}_{j-1,2k+1}) \), following the strategy in \cite{Rio_1994_PTRF-SA}, while accounting for the restriction to the high-probability set.
\textit{Part 1: Strong Approximation Set-up}. By the construction at Equation~\eqref{sa-eq: X-process -- conditional quantile transformation}, condition on $U_{j,k}$, $\widetilde{U}_{j,k}$ has the same distribution as $2 \mathsf{Bin}(U_{j,k},p_{j,k}) - U_{j,k}$, and the conditional quantile transformation relation $\widetilde{U}_{j,k} = \sqrt{U_{j,k}p_{j,k}(1 - p_{j,k})} G_{(j,k), U_{j,k}} \circ \Phi(\widetilde{\xi}_{j,k})$ holds. This allows for application of Lemma~\ref{sa-lem: binom coupling}. Let $\overline{p} = \rho$, $\underline{p} = \rho^{-1}$, $c_0$ to be the positive solution of
\[60 c_0 \overline{p} \left(\sqrt{\frac{1 - \underline{p}}{\underline{p}}}\right)^3 \exp\left(2\sqrt{\frac{1 - \underline{p}}{\underline{p}}}c_0\right) + 60 c_0 (1 - \underline{p}) \left(\sqrt{\frac{\overline{p}}{1 - \overline{p}}}\right)^3 \exp\left(2\sqrt{\frac{\overline{p}}{1 - \overline{p}}}c_0\right) = 1,\]
$c_1 = 15 c_0 \sqrt{\underline{p}(1 - \overline{p})}$, $c_2 = 1/(15 c_0)$, and $c_3 = 1/ c_0$. Consider the small probability set $\mathcal{A}$ where the coupling inequalities from Lemma~\ref{sa-lem: binom coupling} are not guaranteed to hold,
\begin{align*}
\mathcal{A} = \Big\{|\widetilde{U}_{j,k}| \leq c_1 U_{j,k} : (j,k) \in \mathcal{I}_K\Big\},
\end{align*}
and notice that we can always take $c_1 \leq 1$ because $|\widetilde{U}_{j,k}| \leq U_{j,k}$ almost surely. Using Lemma~\ref{sa-lem: binom coupling} conditional on $U_{j,k}$, whenever $\mathcal{A}$ occurs,
\begin{equation}\label{sa-eq: X-process -- coupling}
\begin{aligned}
& \left|\widetilde{U}_{j,k} - \sqrt{U_{j,k} p_{j,k}(1 - p_{j,k})} \widetilde{\xi}_{j,k} \right| < c_2 \widetilde{\xi}_{j,k}^2 + c_3, \\
& \left|\widetilde{U}_{j,k} \right| \leq 1/c_0 + 2 \sqrt{p_{j,k}(1-p_{j,k})}|\widetilde{\xi}_{j,k}|,
\end{aligned}
\end{equation}
for all $(j,k) \in \mathcal{I}_K$.
To bound $\mathbbm{P}(\mathcal{A}^c)$, first notice that by Chernoff's inequality for Binomial distribution, $\mathbbm{P}( U_{j,k} \leq \mathbbm{E}[U_{j,k}]/2) \leq \exp(- \mathbbm{E}[U_{j,k}]/8)$ for all $(j,k) \in \mathcal{I}_K$, and $\mathbbm{P}(U_{j,k} \leq 2^{-1} \rho^{-1} n 2^{j - K}) \leq \exp(- 8^{-1} \rho^{-1} n 2^{j - K})$ for all $(j,k) \in \mathcal{I}_K$ because $\rho^{-1} n 2^{j-K} \leq \mathbbm{E}[U_{j,k}] \leq \rho n 2^{j -K}$. Furthermore, using Hoeffding's inequality and the fact that $\widetilde{U}_{j,k} = U_{j-1,2k} -p_{j,k} U_{j,k} = U_{j-1,2k} - \mathbbm{E}[U_{j-1,2k}|U_{j,k}]$,
\begin{align*}
& \mathbbm{P} \Big(|\widetilde{U}_{j,k}| \geq c_1 U_{j,k} \Big| U_{j,k} \geq \frac{1}{2} \rho^{-1}n 2^{-K + j}\Big)
\leq 2 \exp \left(- \frac{c_1^2 n 2^{-K + j}}{3 \rho}\right).
\end{align*}
Putting these together, and using the union bound,
\begin{align}\label{sa-eq: X-process -- SA quasi residual}
\nonumber \mathbbm{P}(\mathcal{A}^c)
& \leq \sum_{(j,k) \in \mathcal{I}_K} \mathbbm{P}(|\widetilde{U}_{j,k} | > c_1 U_{j,k} ) \\
\nonumber & \leq \sum_{(j,k) \in \mathcal{I}_K} \mathbbm{P} \Big(U_{j,k} \leq \frac{1}{2} \rho^{-1} n 2^{-K + j}\Big)
+ \mathbbm{P} \Big(|\widetilde{U}_{j,k}| \geq c_1 U_{j,k} \Big| U_{j,k} \geq \frac{1}{2} \rho^{-1} n 2^{-K + j}\Big)\\
\nonumber & \leq \sum_{j = 1}^{K} \sum_{0 \leq k < 2^{K - j}} \Big\{\exp(-8^{-1} \rho^{-1} n 2^{j-K}) + 2 \exp(- c_1^2 \rho^{-1} n 2^{j-K}/3) \Big\}\\
& \leq 3 \cdot 2^{K} \exp(- \min\{c_1^2/3 , 1/8\} \rho^{-1} n 2^{-K}).
\end{align}
\textit{Part 2: Bounding Strong Approximation Error}. We show that the proof of Theorem 2.1 in \cite{Rio_1994_PTRF-SA} still goes through for an approximate dyadic scheme. In other words, we show that the approximate dyadic scheme gives essentially the same Gaussian coupling rates as the dyadic scheme (Section \ref{sa-sec: X-process -- Cell Expansions}). We employ the same notation as in \cite{Rio_1994_PTRF-SA}, and for $g \in \mathtt{\Pi}_{0} \mathscr{H}$, define
\begin{align*}
\Delta(g) &= (X - Z)(g),
\qquad
X(g) = \sum_{j = 1}^{K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}(g) \widetilde{U}_{j,k},
\qquad
Z(g) = \sum_{j = 1}^{K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}(g) \widetilde{V}_{j,k},\\
\Delta_1(g) &= (X - Y)(g),\qquad
Y(g) = \sum_{j = 1}^{K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}(g) \sqrt{U_{j,k}\widetilde{p}_{j,k}(1 - \widetilde{p}_{j,k})} \widetilde{\xi}_{j,k},\\
\Delta_2(g) &= (Y - Z)(g)
\end{align*}
It suffices to verify the following two claims.
\underline{Claim 1}: $\mathbbm{E}[\exp(t \Delta_1(g)) \mathbbm{1}(\mathcal{A})] \leq \prod_{j = 1}^{K} \prod_{0 \leq k < 2^{K - j}} \mathbbm{E} [\cosh(t \widetilde{\beta}_{j,k}(g)(2 + \widetilde{\xi}_{j,k}^2/4))]$ for all $g \in \mathtt{\Pi}_{0} \mathscr{H}$. Then, it follows from the proof of Lemma 2.2 in \cite{Rio_1994_PTRF-SA} that
\begin{align*}
\log \mathbbm{E} [\exp(4 t \Delta_1(g))\mathbbm{1}(\mathcal{A})] \leq - \frac{83}{3} c_{\rho}^2 \left(\sum_{j =1}^{K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}^2(g) \right) \log (1 - t^2),
\end{align*}
for all $|t| < 1$.
\underline{Claim 2}: $\mathbbm{E} [\exp(t \Delta_2) \mathbbm{1}(\mathcal{A})] \leq \mathbbm{E}[ \exp(t c_{\rho}\Delta_3)]$ for all $t > 0$, where
\begin{align*}
\Delta_3(g) = \sum_{j = 1}^{K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k} (g) \widetilde{\xi}_{j,k} \left(1 + \sum_{l = j}^{K} \sum_{0 \leq q < 2^{K - l}} 2^{-|j-l|/2} \big|\widetilde{\xi}_{l,q}\big| \mathbbm{1}(\mathcal{C}_{l,q} \supseteq \mathcal{C}_{j,k}) \right),
\end{align*}
for all $g \in \mathtt{\Pi}_{0} \mathscr{H}$, and $c_{\rho}$ a constant that only depends on $\rho$.
\bigskip\noindent\underline{Proof of Claim 1}:
Let $\mathcal{F}_{j} = \boldsymbol{\sigma}(\{\widetilde{\xi}_{l,k}: j < l \leq K, 0 \leq k < 2^{K - l}\})$ for all $1 \leq j < K$. In particular, $\boldsymbol{\sigma}(\{ U_{l,k}: j \leq l \leq K, 0 \leq k < 2^{K - l}\}) \subseteq \mathcal{F}_j$. Then, by Equation~\ref{sa-eq: X-process -- coupling}, for all $t \in \mathbb{R}$,
\begin{align*}
& \mathbbm{E} \left[\exp\left(t \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}(g) \left(\widetilde{U}_{j,k} - \sqrt{U_{j,k}\widetilde{p}_{j,k}(1 - \widetilde{p}_{j,k})} \widetilde{\xi}_{j,k} \right)\right) \mathbbm{1}(\mathcal{A}) \bigg | \mathcal{F}_{j}\right] \\
& \qquad \leq \mathbbm{E} \left[\prod_{0 \leq k < 2^{K - j}} \cosh \left(t \widetilde{\beta}_{j,k}(g) (c_2\widetilde{\xi}_{j,k}^2 + c_3) \right) \mathbbm{1}(\mathcal{A}) \middle| \mathcal{F}_j\right].
\end{align*}
Then, we will use the same induction argument as in the proof of Lemma 2.2 in \cite{Rio_1994_PTRF-SA}: let
\begin{align*}
S_j(t) & = \exp \left(t\sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}(g) \left(\widetilde{U}_{j,k} - \sqrt{U_{j,k}\widetilde{p}_{j,k}(1 - \widetilde{p}_{j,k})} \widetilde{\xi}_{j,k} \right)\right),
\end{align*}
so that $\mathbbm{E}[\exp(t \Delta_1) \mathbbm{1}(\mathcal{A})] = \mathbbm{E}[ \prod_{j = 1}^{K} S_j(t) \mathbbm{1}(\mathcal{A})]$, and
\begin{align*}
T_j(t) = \prod_{0 \leq k < 2^{K - j}} \cosh\left(t \widetilde{\beta}_{j,k}(g)(c_2 \widetilde{\xi}_{j,k}^2 + c_3) \right),
\end{align*}
so that $\prod_{j = 1}^{K} \prod_{0 \leq k < 2^{K - j}} \mathbbm{E} [\cosh(t \widetilde{\beta}_{j,k}(2 + \widetilde{\xi}_{j,k}^2/4))] = \mathbbm{E}[\prod_{j = 1}^{K} T_j(t)]$. By Equation~\ref{sa-eq: X-process -- coupling}, for all $1 \leq j \leq K$,
\begin{align*}
\mathbbm{E} \left[S_j(t) \prod_{l = 1}^{j-1} T_l(t) \mathbbm{1}(\mathcal{A}) \middle | \mathcal{F}_j \right]
\leq \mathbbm{E} \left[\prod_{l = 1}^j T_l(t) \mathbbm{1}(\mathcal{A}) \middle| \mathcal{F}_j \right].
\end{align*}
It follows that
\begin{align*}
\mathbbm{E}[\exp(t \Delta_1) \mathbbm{1}(\mathcal{A})]
& = \mathbbm{E} \left[ \prod_{j = 1}^{K} S_j(t) \mathbbm{1}(\mathcal{A})\right]
= \mathbbm{E} \left[ \mathbbm{E} \left[S_1(t) \mathbbm{1}(\mathcal{A})\middle| \mathcal{F}_1 \right] \prod_{j = 2}^{K} S_j(t)\right]
\leq \mathbbm{E} \left[ \mathbbm{E} \left[T_1(t) \mathbbm{1}(\mathcal{A})\middle| \mathcal{F}_1 \right] \prod_{j = 2}^{K} S_j(t)\right]\\
& = \mathbbm{E} \left[ \mathbbm{E} \left[T_1(t) S_2(t)\mathbbm{1}(\mathcal{A})\middle| \mathcal{F}_2 \right] \prod_{j = 3}^{K} S_j(t)\right]
\leq \mathbbm{E} \left[ \mathbbm{E} \left[T_1(t) T_2(t)\mathbbm{1}(\mathcal{A})\middle| \mathcal{F}_2 \right] \prod_{j = 3}^{K} S_j(t)\right]\\
& \leq \mathbbm{E} \left[\prod_{j = 1}^{K} T_j(t) \mathbbm{1}(\mathcal{A}) \right]
\leq \mathbbm{E} \left[\prod_{j = 1}^{K} T_j(t)\right]
= \prod_{j = 1}^{K} \prod_{0 \leq k < 2^{K - j}} \mathbbm{E} [\cosh(t \widetilde{\beta}_{j,k}(h)(c_2\widetilde{\xi}_{j,k}^2 + c_3))] \\
& \leq \prod_{j = 1}^{K} \prod_{0 \leq k < 2^{K - j}} \mathbbm{E} [\cosh(t c_{\rho} \widetilde{\beta}_{j,k}(h)(\widetilde{\xi}_{j,k}^2/4 + 2))]
\end{align*}
where in the last line, we have used independence of $(\widetilde{\xi}_{j,k}: 1 \leq j \leq K, 0 \leq k < 2^{K - j})$. Without loss of generality, we assume that $c_{\rho} \sup_{\mathbf{x} \in \mathcal{C}_{K,0}} |g(\mathbf{x})| \leq 1$. Since we know $(\widetilde{\xi}_{j,k}, 1 \leq j \leq K, 0 \leq k < 2^{K - j})$ are i.i.d. standard Gaussian, the same upper bound established in \cite{Rio_1994_PTRF-SA} for the right hand side of the last display holds: for all $g \in \mathtt{\Pi}_{0} \mathscr{H}$, $|t| < 1$,
\begin{equation}\label{sa-eq: X-process -- haar delta_1}
\log \mathbbm{E}[\exp(4 t \Delta_1(g)) \mathbbm{1}(\mathcal{A})] \leq - \frac{83}{3} c_{\rho^2} \left(\sum_{j =1}^{K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}^2 (h) \right) \log(1 - t^2) = h_{\Delta_1}(t),
\end{equation}
which concludes the verification of the first claim.
\bigskip\noindent\underline{Proof of Claim 2}:
Denote $q_{j,k} = \mathbbm{P}_X(\mathcal{C}_{j,k})$ for $(j,k) \in \mathcal{J}_K$. By Equation~\eqref{sa-eq: X-process -- conditional quantile transformation}, for any $g \in \mathtt{\Pi}_{0} \mathscr{H}$, we have
\begin{align*}
\Delta_2(g) = \sum_{j = 1}^{K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}(g) \left(\sqrt{U_{j,k}} - \sqrt{\mathbbm{E}[U_{j,k}]} \right) \sqrt{\frac{q_{j-1,2k}q_{j-1,2k+1}}{q_{j,k}^2}} \widetilde{\xi}_{j,k}.
\end{align*}
We will use the same strategy as in \cite{Rio_1994_PTRF-SA} adapted to the quasi-dyadic case. Fix $0 \leq l < 2^{K - j}$ and $0 \leq j \leq K$, and let $k_l$ be the unique integer in $[0,2^{K - l})$ such that $\mathcal{C}_{l,k_l} \supseteq \mathcal{C}_{j,k}$. Then,
\begin{align*}
\sqrt{U}_{j,k} - \sqrt{\mathbbm{E}[U_{j,k}]} & = \sum_{l = j}^{K - 1} \sqrt{U_{l,k_l} \frac{q_{j,k}}{q_{l,k_l}}} - \sqrt{U_{l+1,k_{l+1}}\frac{q_{j,k}}{q_{l+1,k_{l+1}}}} \\
& = \sum_{l = j}^{K - 1} \sqrt{\frac{q_{j,k}}{q_{l+1,k_{l+1}}}} \left( \sqrt{\frac{q_{l+1,k_{l+1}}}{q_{l,k_l}}U_{l,k_l}} - \sqrt{U_{l+1,k_{l+1}}}\right).
\end{align*}
By Equation~\ref{sa-eq: X-process -- coupling}, when the event $\mathcal{A}$ holds,
\begin{align*}
\left|\sqrt{\frac{q_{l+1,k_{l+1}}}{q_{l,k_l}}U_{l,k_l}} - \sqrt{U_{l+1,k_{l+1}}}\right|
& \leq \frac{|\widetilde{U}_{l,k_l}|}{\sqrt{\frac{q_{l+1,k_{l+1}}}{q_{l,k_l}}U_{l,k_l}} + \sqrt{U_{l+1,k_{l+1}}}} \\
& \leq \frac{2 \sqrt{\frac{q_{l+1,2k_{l}}}{q_{l,k_l}} \frac{q_{l+1,2k_l + 1}}{q_{l,k_l}}U_{l,k_l}}|\widetilde{\xi}_{l,k_l}| + \min\{ c_0^{-1}, \widetilde{U}_{l,k_l}\}}{\sqrt{\frac{q_{l+1,k_{l+1}}}{q_{l,k_l}}U_{l,k_l}} + \sqrt{U_{l+1,k_{l+1}}}} \\
& \leq 2 \sqrt{\frac{q_{l+1,2 k_l + 1}}{q_{l,k_l}}} |\widetilde{\xi}_{l,k_l}| + \frac{\min\{ c_0^{-1}, |\widetilde{U}_{l,k_l}|\}}{\sqrt{\frac{q_{l+1,k_{l+1}}}{q_{l,k_l}}U_{l,k_l}} + \sqrt{U_{l+1,k_{l+1}}}}.
\end{align*}
For the first summand,
\begin{align*}
\sum_{l = j}^{K - 1} \sqrt{\frac{q_{j,k}}{q_{l+1,k_{l+1}}}} 2 \sqrt{\frac{q_{l+1,2 k_l + 1}}{q_{l,k_l}}} |\widetilde{\xi}_{l,k_l}|
= \sum_{l = j}^{K - 1} \sqrt{\prod_{j < s \leq l}p_{s,k_s}} 2 \sqrt{p_{l,k_l}} |\widetilde{\xi}_{l,k_l}|
\leq c_{\rho}\sum_{l = j}^{K - 1} 2^{-(l-j)/2} |\widetilde{\xi}_{l,k_l}|.
\end{align*}
For the second summand, we separate it into two terms as in \cite{Rio_1994_PTRF-SA}. For $\mathbbm{1}(\widetilde{U}_{l,k_l} \leq 0)$, we have
\begin{align*}
& \sum_{l = j}^{K - 1} \sqrt{\frac{q_{j,k}}{q_{l+1,k_{l+1}}}} \frac{\min\{ c_0^{-1}, -\widetilde{U}_{l,k_l}\}}{\sqrt{\frac{q_{l+1,k_{l+1}}}{q_{l,k_l}}U_{l,k_l}} + \sqrt{U_{l+1,k_{l+1}}}} \mathbbm{1}(\widetilde{U}_{l,k_l} \leq 0) \\
& \qquad = \sum_{l = j}^{K - 1} \sqrt{\frac{q_{j,k}}{q_{l+1,k_{l+1}}}} \frac{\min\{ c_0^{-1}, -\widetilde{U}_{l,k_l}\}}{\sqrt{U_{l+1,k_{l+1}} - \widetilde{U}_{l,k_l}} + \sqrt{U_{l+1,k_{l+1}}}} \mathbbm{1}(\widetilde{U}_{l,k_l} \leq 0) \leq c_{\rho},
\end{align*}
since $\sup_{0 \leq x \leq u} \min\{c_0^{-1}, x\}/(\sqrt{u} + \sqrt{u + x}) \lesssim 1$. For $\mathbbm{1}(\widetilde{U}_{l,k_l} > 0)$, we have
\begin{align*}
& \sum_{l = j}^{K - 1} \sqrt{\frac{q_{j,k}}{q_{l+1,k_{l+1}}}} \frac{\min\{ c_0^{-1}, \widetilde{U}_{l,k_l}\}}{\sqrt{\frac{q_{l+1,k_{l+1}}}{q_{l,k_l}}U_{l,k_l}} + \sqrt{U_{l+1,k_{l+1}}}} \mathbbm{1}(\widetilde{U}_{l,k_l} > 0) \\
& \leq \sum_{l = j}^{K - 1} \sqrt{\frac{q_{j,k}}{q_{l+1,k_{l+1}}}} \left( \sqrt{U_{l+1,k_{l+1}}} - \sqrt{\frac{q_{l+1,k_{l+1}}}{q_{l,k_l}}U_{l,k_l}}\right)\mathbbm{1}\Big(\frac{q_{l+1,k_{l+1}}}{q_{l,k_l}}U_{l,k_l} \leq U_{l+1,k_{l+1}} \leq \frac{q_{l+1,k_{l+1}}}{q_{l,k_l}}U_{l,k_l} + c_0^{-1}\Big) \\
& \leq \sum_{l = j}^{K - 1} \sqrt{\frac{q_{j,k}}{q_{l+1,k_{l+1}}}} \sqrt{c_0^{-1}} = \sum_{l = j}^{K - 1} \sqrt{\prod_{j < s \leq l}p_{s,k_s}} \sqrt{c_0^{-1}} \leq c_{\rho}.
\end{align*}
It follows that when the event $\mathcal{A}$ holds,
\begin{align*}
\Big|\sqrt{U_{j,k}} - \sqrt{\mathbbm{E}[U_{j,k}]} \Big|
& \leq c_{\rho} \bigg(1 + \sum_{l = j}^{K - 1} 2^{-(l-j)/2} \sum_{0 \leq q < 2^{K - l}} |\widetilde{\xi}_{l,q}| \mathbbm{1}(\mathcal{C}_{l,q} \supseteq \mathcal{C}_{j,k}) \bigg).
\end{align*}
Using an induction argument, for all $g \in \mathtt{\Pi}_{0} \mathscr{H}$, $t > 0$,
\begin{equation}\label{eq: delta_2}
\mathbbm{E}[ \exp(t \Delta_2(g)) \mathbbm{1}(\mathcal{A})] \leq \mathbbm{E}[\exp(t c_{\rho} \Delta_3(g))].
\end{equation}
For any random variable $W$, define $\gamma_{W}(t) = \log(\mathbbm{E}[ \exp(t c_{\rho} W)])$ for all $ t > 0$, and $h_W(u) = \sup_{t > 0}(t u - \gamma_{W}(u))$. Combining Equation~\eqref{sa-eq: X-process -- haar delta_1}, for any $g \in \mathtt{\Pi}_{0} \mathscr{H}$, $t > 0$,
\begin{align*}
\mathbbm{P}(\Delta_1(g) \geq t \text{ and } \mathcal{A})
& \leq \inf_{u > 0}\mathbbm{P} (\exp(\Delta_1(g) u) \geq \exp(tu) \text{ and } \mathcal{A}) \leq \inf_{u > 0} \exp(-tu) \mathbbm{E}[\exp(\Delta_1(g) u) \mathbbm{1}(\mathcal{A})] \\
& \leq \exp\left(-h_{\Delta_1(g)}(t)\right)
= \exp\left(-\sup_{u > 0}\left(t u + \frac{83}{3}c_{\rho}^2 \lVert g \rVert_{\mathscr{E}_K}^2 \log(1 - u^2/16)\right)\right),
\end{align*}
hence for any $t > 0$,
\begin{align}\label{sa-eq: X-process -- SA quasi delta 1}
\mathbbm{P}(|\Delta_1(g)| \geq C c_{\rho} \lVert g \rVert_{\mathscr{E}_K} \sqrt{t} + C t \text{ and } \mathcal{A})
= \mathbbm{P}(\Delta_1(g) \geq h_{\Delta_1(g)}^{-1}(t) \text{ and } \mathcal{A}) \leq 2 \exp(-t).
\end{align}
By Equation~\eqref{eq: delta_2}, for any $t > 0$,
\begin{align}\label{sa-eq: GEP -- SA quasi delta 2}
\mathbbm{P}(\Delta_2(g) \geq t \text{ and } \mathcal{A})
\leq \inf_{u > 0} \exp(-tu) \mathbbm{E}[\exp(\Delta_2(g) u) \mathbbm{1}(\mathcal{A})]
\leq \exp\left(-h_{\Delta_3(g)}(t)\right).
\end{align}
Since $\Delta_3(g)$ only depends on $((\widetilde{\xi}_{j,k},\widetilde{\beta}_{j,k}(g)): (j,k) \in \mathcal{I}_K)$, the rest of the proof follows from Lemma 2.4 in \cite{Rio_1994_PTRF-SA}. In particular, define
\begin{gather*}
\Delta_4(g) = \sum_{j = 1}^K \sum_{0 \leq k < 2^{K-j}} \widetilde{\beta}_{j,k}(g) \widetilde{\xi}_{j,k}, \qquad
\Delta_5(g) = \Delta_3(g) - \Delta_4(g),
\end{gather*}
then identifying that $\Delta_4(g)$ is Gaussian and applying \citet[Lemma 2.4]{Rio_1994_PTRF-SA} with two choices of $p_i$-sequence, $p_i = \frac{1}{2}(\frac{1}{K} + \frac{1}{i(i+1)})$ and $p_i = \frac{1}{i(i+1)}$ separately on $\Delta_5(g)$, we get for any $t > 0$, and $g \in \mathtt{\Pi}_{0} \mathscr{H}$,
\begin{align*}
\mathbbm{P} \left(\left|\Delta_2(g)\right|
\geq c_{\rho} \lVert g \rVert_{\mathscr{E}_K} \sqrt{t} + c_{\rho} \sqrt{\mathtt{C}_{\{g\},K}}t \text{ and } \mathcal{A} \right)
\leq
\mathbbm{P} \left(\left|\Delta_3(g)\right|
\geq c_{\rho} \lVert g \rVert_{\mathscr{E}_K} \sqrt{t} + c_{\rho} \sqrt{\mathtt{C}_{\{g\},K}}t \right) \leq 2 \exp(-t).
\end{align*}
Combining Equation~\eqref{sa-eq: X-process -- SA quasi residual}, \eqref{sa-eq: X-process -- SA quasi delta 1} and \eqref{sa-eq: GEP -- SA quasi delta 2}, we get the stated result.
\end{myproof}
\subsubsection{Meshing Error}\label{sa-sec: X-process -- Meshing Error}
For $0 < \delta \leq 1$, consider the $(\delta \mathtt{M}_{\mathscr{H},\mathcal{X}})$-net of $(\mathscr{H}, \lVert \cdot \rVert_{\mathbbm{P}_X,2})$ over $\mathcal{X}$, $\mathscr{H}_\delta$, with cardinality no larger than $\mathtt{N}_{\mathscr{H},\mathcal{X}}(\delta,\mathtt{M}_{\mathscr{H},\mathcal{X}})$. Define $\pi_{\mathscr{H}_{\delta}}: \mathscr{H} \mapsto \mathscr{H}$ such that $\lVert \pi_{\mathscr{H}_{\delta}}(h) - h \rVert_{\mathbbm{P}_X,2} \leq \delta \mathtt{M}_{\mathscr{H},\mathcal{X}}$ for all $h \in \mathscr{H}$. To simplify notation, in this section the parameters of $\mathscr{H}$ (Definitions 4 to 12) are taken with $\mathcal{C} = \mathcal{X}$, and the index $\mathcal{C}$ is omitted whenever there is no confusion.
\begin{lemma}\label{sa-lem: X-Process -- Fluctuation off the net}
Suppose Assumption~\ref{sa-assump: X-process -- step} holds, a quasi-dyadic expansion $\mathscr{C}_{K}(\mathbbm{P}_X, \rho)$ is given, $(Z_n^X(h): h \in \mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H})$ is the Gaussian process constructed as in \eqref{sa-eq: X-process -- Gaussian Process} on a possibly enlarged probability space, and $\mathscr{H}_\delta$ is chosen in Section~\ref{sa-sec: X-process -- Meshing Error}. Then, for all $t > 0$ and $0<\delta<1$,
\begin{align*}
\mathbbm{P}\big[\lVert X_n - X_n\circ\pi_{\mathscr{H}_\delta} \rVert_{\mathscr{H}} \gtrsim \mathsf{F}_n(t,\delta)\big] &\leq \exp(-t),\\
\mathbbm{P}\big[\lVert Z_n^X\circ\pi_{\mathscr{H}_\delta}-Z_n^X \rVert_{\mathscr{H}} \gtrsim \mathtt{M}_{\mathscr{H}}J(\delta,\mathscr{H}, \mathtt{M}_{\mathscr{H}}) + \delta \mathtt{M}_{\mathscr{H}} \sqrt{t}\big] &\leq \exp(-t).
\end{align*}
\end{lemma}
\begin{myproof}{Lemma \ref{sa-lem: X-Process -- Fluctuation off the net}}
Take $\mathcal{L} = \{h - \pi_{\mathscr{H}_{\delta}}(h): h \in \mathscr{H}\}$ on $(\mathcal{X},\mathcal{B}(\mathcal{X}),\mathbbm{P}_X)$. Then, $\sup_{l \in \mathcal{L}}\lVert l \rVert_{\mathbbm{P}_X,2} \leq \delta \mathtt{M}_{\mathscr{H}}$ and, for all $0 < \varepsilon < \delta$,
\begin{align*}
\mathtt{N}_{\mathcal{L}}(\varepsilon,\mathtt{M}_{\mathscr{H}}) \leq \mathtt{N}_{\mathcal{L}}(\varepsilon,\mathtt{M}_{\mathscr{H}}) \mathtt{N}_{\mathcal{L}}(\delta,\mathtt{M}_{\mathscr{H}}) \leq \mathtt{N}_{\mathcal{L}}(\varepsilon,\mathtt{M}_{\mathscr{H}})^2,
\end{align*}
Hence $J(u,\mathcal{L},\mathtt{M}_{\mathscr{H}}) \leq 2 J(u, \mathscr{H}, \mathtt{M}_{\mathscr{H}})$ for all $0 < u < \delta$. By \citet[Theorem 5.2]{chernozhukov2014gaussian-SA}, we have
\begin{align*}
\mathbbm{E}[\lVert X_n - X_n \circ \pi_{\mathscr{H}_{\delta}} \rVert_{\mathscr{H}} ]
\lesssim J(\delta, \mathscr{H}, \mathtt{M}_{\mathscr{H}}) \mathtt{M}_{\mathscr{H}}
+ \frac{ \mathtt{M}_{\mathscr{H}} J^2(\delta, \mathscr{H}, \mathtt{M}_{\mathscr{H}})}{\delta^2 \sqrt{n}}.
\end{align*}
By Talagrand's inequality \citep[Theorem 3.3.9]{Gine-Nickl_2016_Book-SA}, for all $t > 0$,
\begin{align*}
\mathbbm{P} \left(\lVert X_n - X_n \circ \pi_{\mathscr{H}_{\delta}} \rVert_{\mathscr{H}} \gtrsim J(\delta,\mathscr{H}, \mathtt{M}_{\mathscr{H}}) \mathtt{M}_{\mathscr{H}} + \frac{ \mathtt{M}_{\mathscr{H}} J^2(\delta,\mathscr{H}, \mathtt{M}_{\mathscr{H}})}{\delta^2 \sqrt{n}} + \delta \mathtt{M}_{\mathscr{H}} \sqrt{t} +\frac{ \mathtt{M}_{\mathscr{H}}}{\sqrt{n}}t \right) \leq \exp(-t).
\end{align*}
By \citet[Corollary 2.2.9]{wellner2013weak-SA},
\begin{align*}
\mathbbm{E}[\lVert Z_n - Z_n \circ \pi_{\mathscr{H}_{\delta}} \rVert_{\mathscr{H}}] \lesssim J(\delta, \mathscr{H}, \mathtt{M}_{\mathscr{H}})\mathtt{M}_{\mathscr{H}_{\delta}}.
\end{align*}
By pointwise separability and a concentration inequality for Gaussian suprema, for all $t > 0$,
\begin{align*}
\mathbbm{P} \left(\lVert Z_n - Z_n \circ \pi_{\mathscr{H}_{\delta}} \rVert_{\mathscr{H}} \gtrsim J(\delta, \mathscr{H},\mathtt{M}_{\mathscr{H}})\mathtt{M}_{\mathscr{H}} + \delta \mathtt{M}_{\mathscr{H}}\sqrt{t} \right) \leq \exp(-t),
\end{align*}
which concludes the proof.
\end{myproof}
\subsubsection{Strong Approximation Errors}\label{sa-sec: X-process -- SA Errors}
To simplify notation, in this section the parameters of $\mathscr{H}$ (Definitions 4 to 12) are taken with $\mathcal{C} = \mathcal{X}$, and the index $\mathcal{C}$ is omitted whenever there is no confusion. The next lemma controls the strong approximation error for projected processes.
\begin{lemma}\label{sa-lem: X-process -- SA error}
Suppose Assumption~\ref{sa-assump: X-process -- step} holds, a dyadic expansion \(\mathscr{C}_{K}(\mathbbm{P}_X, 1)\) is given, \((Z_n^X(h): h \in \mathscr{H} \cup \mathscr{E}_{K,\mathtt{M}_{\mathscr{H}}})\) is the Gaussian process constructed as in \eqref{sa-eq: X-process -- Gaussian Process} on a possibly enlarged probability space, and \(\mathscr{H}_\delta\) is chosen as in Section~\ref{sa-sec: X-process -- Meshing Error}. For each \(1 \leq j \leq K\), define the \(j\)-th level difference set
\begin{align*}
\mathcal{U}_j = \cup_{0 \leq k < 2^{K - j}} (\mathcal{C}_{j-1,2k+1} - \mathcal{C}_{j-1,2k}).
\end{align*}
Then, for all \(t > 0\),
\begin{align*}
\mathbbm{P}\Bigg[\lVert X_n \circ \mathtt{\Pi}_{0} - Z_n^X \circ \mathtt{\Pi}_{0} \rVert_{\mathscr{H}_{\delta}}
> 48 \sqrt{\frac{\mathscr{R}_{K}(\mathscr{H}_{\delta})}{n} t}
+ 4 \sqrt{\frac{\mathtt{C}_{\mathscr{H}_{\delta},K} }{n}}t \Bigg]
\leq 2 \mathtt{N}_{\mathscr{H}_{\delta}}(\delta, \mathtt{M}_{\mathscr{H}_{\delta}}) e^{-t},
\end{align*}
where
\begin{align*}
\mathscr{R}_{K}(\mathscr{H}_{\delta}) = \sum_{j = 1}^{K} \min \{\mathtt{M}_{\mathscr{H}_{\delta}}, \lVert \mathcal{U}_j \rVert_{\infty} \mathtt{L}_{\mathscr{H}_{\delta}} \} 2^{K - j}
\min \bigg\{\sqrt{d} \sup_{\mathbf{x} \in \mathcal{X}} f_X^2(\mathbf{x}) 2^{2(K - j)} \lVert \mathcal{U}_j \rVert_{\infty} \mathfrak{m}(\mathcal{U}_j) \mathtt{TV}_{\mathscr{H}_{\delta}}^{\ast}, \lVert \mathcal{U}_j \rVert_{\infty} \mathtt{L}_{\mathscr{H}_{\delta}}, \mathtt{E}_{\mathscr{H}_{\delta}}\bigg\},
\end{align*}
and \(\mathtt{C}_{\mathscr{H}_{\delta},K}\) is defined in Lemma \ref{sa-lem: X-process -- sa for pcw-const}. In the above display, \(f_X\) denotes the Lebesgue density of \(\mathbbm{P}_X\): if it does not exist, the term \(\sqrt{d} \sup_{\mathbf{x} \in \mathcal{X}} f_X^2(\mathbf{x}) 2^{2(K - j)} \lVert \mathcal{U}_j \rVert_{\infty} \mathfrak{m}(\mathcal{U}_j) \mathtt{TV}_{\mathscr{H}_{\delta}}^{\ast}\) is taken to be infinity.
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: X-process -- SA error}}
We employ the same strategy as in the proof of Theorem 1.1 from \cite{Rio_1994_PTRF-SA}, noting that incorporating the Lipschitz condition can lead to a tighter bound for strong approximation error.
A very first bound that we can obtain for is
\begin{align*}
\sum_{0 \leq k < 2^{K-j}} \left|\widetilde{\beta}_{j,k}(h)\right|
& \leq \sum_{\sum_{0 \leq k < 2^{K-j}}} 2^{K - j} \int_{\mathcal{C}_{j,k}} \left|h(\mathbf{x})\right| d \mathbbm{P}_X(\mathbf{x})\\
& \leq 2^{K - j} \int_{\sqcup_{0 \leq k < 2^{K - (j-1)}} \mathcal{C}_{j-1,k}} \left|h(\mathbf{x})\right| d \mathbbm{P}_X(\mathbf{x})
\leq 2^{K - j} \mathtt{E}_{\{h\}}.
\end{align*}
If we further assume $\mathbbm{P}_X$ admits a Lebesgue density $f_X$, then an analysis based on total variation of $h$ can be done as follows. For each $1 \leq j \leq K$, there exists unique integers $j_1, \ldots, j_d$ such that $0 \leq j_1 \leq \ldots \leq j_d \leq j_1 + 1$ and $\sum_{i = 1}^d j_i = j$. In particular, there exists a unique $l = l(j) \in \{1,2,\dots,d\}$ such that either $l \leq d - 1$ and $j_l < j_{l+1}$ or $l = d$ and $j_d < j_1 + 1$.
\begin{align*}
\widetilde{\beta}_{j,k}(h) & = 2^{K - j} \int_{\mathcal{C}_{j-1,2k}} h(\mathbf{x}) f_X(\mathbf{x}) d\mathbf{x} - 2^{K - j}\int_{\mathcal{C}_{j-1,2k+1}} h(\mathbf{y})f_X(\mathbf{y})d \mathbf{y} \\
& = 2^{K - j} \int_{\mathcal{C}_{j-1,2k}} \left(h(\mathbf{x}) - \left(2^{K - j}\int_{\mathcal{C}_{j-1,2k+1}} h(\mathbf{y})f_X(\mathbf{y})d \mathbf{y} \right) \right) f_X(\mathbf{x}) d \mathbf{x} \\
& = 2^{2(K - j)} \int_{\mathcal{C}_{j-1,2k}} \int_{\mathcal{C}_{j-1,2k+1}} (h(\mathbf{x}) - h(\mathbf{y}))f_X (\mathbf{x}) f_X(\mathbf{y}) d \mathbf{y} d \mathbf{x} \\
& = 2^{2(K - j)} \int_{\mathcal{C}_{j-1,2k}} \int_{\mathcal{C}_{j-1,2k+1} - \{\mathbf{x}\}} (h(\mathbf{x}) - h(\mathbf{x} + \mathbf{s}))f_X(\mathbf{x}) f_X(\mathbf{x} + \mathbf{s}) \mathbbm{1}_{\mathcal{C}_{j-1,2k+1}}(\mathbf{x} + \mathbf{s}) d \mathbf{s} d \mathbf{x}.
\end{align*}
Since we have assumed $f$ is bounded from above on $\mathcal{X}$ and hence on $\mathcal{C}_{K,0}$, and $\mathcal{C}_{j-1,2k+1} - \{\mathbf{x}\} \subseteq \mathcal{C}_{j-1,2k+1} - \mathcal{C}_{j-1,2k}$,
\begin{align*}
\left|\widetilde{\beta}_{j,k}(h)\right| \leq 2^{2(K - j)} \int_{\mathcal{C}_{j-1,2k+1} - \mathcal{C}_{j-1,2k}} \int_{\mathcal{C}_{j-1,2k}} |h(\mathbf{x}) - h(\mathbf{x} + \mathbf{s})| f_X(\mathbf{x}) f_X(\mathbf{x} + \mathbf{s})d \mathbf{x} d \mathbf{s}.
\end{align*}
and therefore
\begin{align*}
\sum_{0 \leq k < 2^{K-j}} \left|\widetilde{\beta}_{j,k}(h)\right| \leq 2^{2(K - j)} \int_{\mathcal{U}_j} \int_{\sqcup_{0 \leq k < 2^{K - j}}\mathcal{C}_{j-1,2k}}|h(\mathbf{x}) - h(\mathbf{x} + \mathbf{s})| f_X(\mathbf{x}) f_X(\mathbf{x} + \mathbf{s}) d \mathbf{x} d \mathbf{s},
\end{align*}
where $\mathcal{U}_j = \cup_{0 \leq k < 2^{K - j}} (\mathcal{C}_{j-1,2k+1} - \mathcal{C}_{j-1,2k})$. Let $(h_\ell)_{\ell \in \mathbb{N}}$ be any sequence of real-valued functions on $(\mathcal{X},\mathcal{B}(\mathcal{X}))$ such that
$h_{\ell} \rightarrow h$ $\mathfrak{m}$-almost surely, and are bounded by $2 \mathtt{M}_{\mathscr{H}}$ on $\mathcal{X}$. Since we assumed $\mathtt{M}_\mathscr{H} < \infty$, and $h_\ell$ and $h$ are bounded by $2 \mathtt{M}_\mathscr{H}$ with $\int_{\mathbb{R}^d} 2 \mathtt{M}_{\mathscr{H}}f_X(\mathbf{x})d\mathbf{x} \leq 2 \mathtt{M}_{\mathscr{H}} < \infty$, Dominated Convergence Theorem implies for any $\mathbf{x} \in \mathcal{U}_j$,
\begin{align*}
\int_{\sqcup_{0 \leq k < 2^{K - j}}\mathcal{C}_{j-1,2k}}|h(\mathbf{x}) - h(\mathbf{x} + \mathbf{s})| f_X(\mathbf{x})d \mathbf{x}
& = \lim_{\ell \rightarrow \infty}\int_{\sqcup_{0 \leq k < 2^{K - j}}\mathcal{C}_{j-1,2k}}|h_{\ell}(\mathbf{x}) - h_{\ell}(\mathbf{x} + \mathbf{s})| f_X(\mathbf{x}) d \mathbf{x} \\
& = \lim_{\ell \rightarrow \infty}\int_{\sqcup_{0 \leq k < 2^{K - j}}\mathcal{C}_{j-1,2k}} \int_{0}^{\lVert \mathbf{s} \rVert} \lVert \nabla h_{\ell}(\mathbf{x} + t \mathbf{s}/\lVert \mathbf{s} \rVert) \rVertrVert)} f_X(\mathbf{x}) d t d \mathbf{x} \\
& = \sup_{\mathbf{x} \in \mathcal{X}}f_X(\mathbf{x}) \lVert \mathbf{s} \rVert \limsup_{\ell \rightarrow \infty} \mathtt{TV}_{\{h_{\ell}\}}^*.
\end{align*}
Since the above inequality holds for all sequences $(h_{\ell})_{\ell \in \mathbb{N}}$ such that $h_{\ell} \rightarrow h$ $\mathfrak{m}$-almost surely, and are bounded by $2 \mathtt{M}_{\mathscr{H}}$ on $\mathcal{X}$, Definition~\ref{sa-defn: smooth tv} implies
\begin{align*}
\int_{\sqcup_{0 \leq k < 2^{K - j}}\mathcal{C}_{j-1,2k}}|h(\mathbf{x}) - h(\mathbf{x} + \mathbf{s})| f_X(\mathbf{x})d \mathbf{x} & \leq \sup_{\mathbf{x} \in \mathcal{X}}f_X(\mathbf{x}) \lVert \mathbf{s} \rVert \mathtt{TV}_{\{h\}}^* \\
& \leq \sup_{\mathbf{x} \in \mathcal{X}}f_X(\mathbf{x}) \sqrt{d} \lVert \mathcal{U}_j \rVert_{\infty} \mathtt{TV}_{\{h\}}^*.
\end{align*}
It follows that
\begin{align*}
\sum_{0\leq k < 2^{K - j}} \left|\widetilde{\beta}_{j,k}(h)\right| \leq \sqrt{d} \bigg(\sup_{\mathbf{x} \in \mathcal{X}} f_X(\mathbf{x})\bigg)^2 2^{2(K - j)} \lVert \mathcal{U}_j \rVert_{\infty} \mathfrak{m}(\mathcal{U}_j) \mathtt{TV}_{\{h\}}^*.
\end{align*}
Moreover, $|\widetilde{\beta}_{j,k}(h)| \leq \min \{\mathtt{M}_{\{h\}}, \lVert \mathcal{U}_j \rVert_{\infty}\mathtt{L}_{\{h\}}\}$, hence
\begin{align*}
\sup_{h \in \mathscr{H}_{\delta}} \lVert h \rVert_{\mathscr{E}_K}^2 =
\sup_{h \in \mathscr{H}_{\delta}}\sum_{j = 1}^{K} \sum_{0 \leq k < 2^{K - j}} |\widetilde{\beta}_{j,k}(h)|^2
& \leq \sup_{h \in \mathscr{H}_{\delta}} \sum_{j = 1}^{K} \min \{\mathtt{M}_{\mathscr{H}_{\delta}}, \lVert \mathcal{U}_j \rVert_{\infty}\mathtt{L}_{\mathscr{H}_{\delta}}\}\sum_{0 \leq k < 2^{K - j}} |\widetilde{\beta}_{j,k}(h)|
\leq \mathscr{R}_{K}(\mathscr{H}_{\delta}),
\end{align*}
where $\mathscr{R}_{K}\left(\mathscr{H}_{\delta}\right)$ is defined to be
\begin{align*}
& \sum_{j = 1}^{K} \min \{\mathtt{M}_{\mathscr{H}_{\delta}}, \lVert \mathcal{U}_j \rVert_{\infty}\mathtt{L}_{\mathscr{H}_{\delta}} \} 2^{K - j} \min\left\{\sqrt{d} \bigg(\sup_{\mathbf{x} \in \mathcal{X}} f_X(\mathbf{x})\bigg)^2 2^{2(K - j)} \lVert \mathcal{U}_j \rVert_{\infty} \mathfrak{m}(\mathcal{U}_j)\mathtt{TV}_{\mathscr{H}_{\delta}}^*, \lVert \mathcal{U}_j \rVert_{\infty} \mathtt{L}_{\mathscr{H}_{\delta}}, \mathtt{E}_{\mathscr{H}_{\delta}}\right\}.
\end{align*}
Applying Lemma~\ref{sa-lem: X-process -- sa for pcw-const}, for any $h \in \mathscr{H}_{\delta}$, for any $t > 0$, with probability at least $1 - 2 \exp(-t)$,
\begin{align*}
\left|X_n \circ \mathtt{\Pi}_{0} (h) - Z_n^X \circ \mathtt{\Pi}_{0} (h)\right| \leq 48 \sqrt{\frac{\mathscr{R}_{K}(\mathscr{H}_{\delta})}{n} t} + \sqrt{\frac{\mathtt{C}_{\mathscr{H}_{\delta},K}}{n}}t.
\end{align*}
The result then follows from the fact that $|\mathscr{H}_{\delta}| \leq \mathtt{N}_{\mathscr{H}}(\delta,\mathtt{M}_{\mathscr{H}})$ and a union bound argument.
\end{myproof}
\begin{lemma}\label{sa-lem: X-process -- SA error quasi dyadic}
Suppose Assumption~\ref{sa-assump: X-process -- step} holds, a quasi-dyadic expansion $\mathscr{C}_{K}(\mathbbm{P}_X, \rho)$ is given with $\rho > 1$, $(Z_n^X(h): h \in \mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H})$ is the Gaussian process constructed at Equation~\eqref{sa-eq: X-process -- Gaussian Process} on a possibly enlarged probability space, and $\mathscr{H}_\delta$ is chosen in Section~\ref{sa-sec: X-process -- Meshing Error}. Then, for all $t > 0$,
\begin{equation*}
\mathbbm{P}\Big[\lVert X_n \circ \mathtt{\Pi}_{0} - Z_n^X \circ \mathtt{\Pi}_{0} \rVert_{\mathscr{H}_{\delta}}
> C_{\rho} \sqrt{\frac{\mathscr{R}_{K}(\mathscr{H}_{\delta})}{n} t}
+ C_{\rho} \sqrt{\frac{\mathtt{C}_{\mathscr{H}_{\delta},K}}{n}}t \Big]
\leq 2 \mathtt{N}_{\mathscr{H}}(\delta,\mathtt{M}_{\mathscr{H}}) e^{-t} + 2^{K} \exp \left( - C_{\rho} n 2^{-K} \right),
\end{equation*}
where $C_{\rho}$ is a constant only depending on $\rho$, $\mathscr{R}_{K}(\mathscr{H}_{\delta})$ is defined in Lemma \ref{sa-lem: X-process -- SA error}, and $\mathtt{C}_{\mathscr{H}_{\delta},K}$ is defined in Lemma \ref{sa-lem: X-process -- sa for pcw-const}.
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: X-process -- SA error quasi dyadic}}
This follows from Lemma~\ref{sa-lem: X-process -- sa for pcw-const non-dyadic} and the fact that
\begin{align*}
\sup_{h \in \mathscr{H}_{\delta}}\lVert h \rVert_{\mathscr{E}_K}^2 \leq \mathscr{R}_K(\mathscr{H}_{\delta}), \qquad h \in \mathscr{H}_{\delta},
\end{align*}
from the proof of Lemma~\ref{sa-lem: X-process -- SA error}.
\end{myproof}
\subsubsection{Projection Error}\label{sa-sec: X-process -- proj error}
To simplify notation, in this section the parameters of $\mathscr{H}$ (Definitions 4 to 12) are taken with $\mathcal{C} = \mathcal{X}$, and the index $\mathcal{C}$ is omitted whenever there is no confusion. The following lemma controls the mean square projection onto piecewise constant functions.
\begin{lemma}\label{sa-lem: X-process -- projection error}
Suppose Assumption~\ref{sa-assump: X-process -- step} holds, a dyadic expansion $\mathscr{C}_{K}(\mathbbm{P}_X, 1)$ is given, $(Z_n^X(h): h \in \mathscr{H} \cup \mathtt{\Pi}_{0} \mathscr{H})$ is the Gaussian process constructed as in \eqref{sa-eq: X-process -- Gaussian Process} on a possibly enlarged probability space, and $\mathscr{H}_\delta$ is chosen in Section~\ref{sa-sec: X-process -- Meshing Error}. In addition, assume $\mathbbm{P}_X$ admits a Lebesgue density $f_X$ supported on $\mathcal{X} \subseteq \mathbb{R}^d$. Define quasi-dyadic variation set $\mathcal{V} = \cup_{0 \leq k < 2^{K}} (\mathcal{C}_{0,k} - \mathcal{C}_{0,k})$. Then, for all $t > 0$,
\begin{align*}
\mathbbm{P}\Big[\lVert X_n - X_n \circ \mathtt{\Pi}_{0} \rVert_{\mathscr{H}_{\delta}}
> \sqrt{4 \mathtt{V}_{\mathscr{H}_{\delta}} t}
+ \frac{4 \mathtt{B}_{\mathscr{H}_{\delta}}}{3\sqrt{n}}t \Big]
& \leq 2 \mathtt{N}_{\mathscr{H}_{\delta}}(\delta,\mathtt{M}_{\mathscr{H}_{\delta}}) e^{-t},\\
\mathbbm{P}\Big[\lVert Z_n^X - Z_n^X \circ \mathtt{\Pi}_{0} \rVert_{\mathscr{H}_{\delta}}
> \sqrt{4 \mathtt{V}_{\mathscr{H}_{\delta}} t} \Big]
& \leq 2 \mathtt{N}_{\mathscr{H}_{\delta}}(\delta,\mathtt{M}_{\mathscr{H}_{\delta}}) e^{-t},
\end{align*}
where
\begin{gather*}
\mathtt{V}_{\mathscr{H}_{\delta}} = \min\{2 \mathtt{M}_{\mathscr{H}_{\delta}}, \mathtt{L}_{\mathscr{H}_{\delta}}\lVert \mathcal{V} \rVert_{\infty}\} \left(\sup_{\mathbf{x} \in \mathcal{X}} f_X(\mathbf{x})\right)^2 2^{K} \mathfrak{m}(\mathcal{V}) \lVert \mathcal{V} \rVert_{\infty} \mathtt{TV}_{\mathscr{H}_{\delta}}^{\ast}, \\ \mathtt{B}_{\mathscr{H}_{\delta}} = \min\{2 \mathtt{M}_{\mathscr{H}_{\delta}}, \mathtt{L}_{\mathscr{H}_{\delta}}\lVert \mathcal{V} \rVert_{\infty}\}.
\end{gather*}
\end{lemma}
In particular, if $\mathbbm{P}_X=\mathsf{Uniform}([0,1]^d)$ and $\mathscr{C}_{K}(\mathbbm{P}_X, 1)=\mathscr{A}_{K}(\mathbbm{P}_X, 1)$, then for all $t > 0$,
\begin{gather*}
\mathbbm{P}\Big[\lVert X_n - X_n \circ \mathtt{\Pi}_{0} \rVert_{\mathscr{H}_{\delta}}
> \sqrt{4d \min\{2\mathtt{M}_{\mathscr{H}_\delta},\mathtt{L}_{\mathscr{H}_\delta}2^{-K}\} 2^{-K} \mathtt{TV}_{\mathscr{H}_\delta}^{\ast} t}
+ \frac{4\min\{2\mathtt{M}_{\mathscr{H}_\delta}, \mathtt{L}_{\mathscr{H}_\delta}2^{-K}\}}{3\sqrt{n}} t \Big]
\leq 2 \mathtt{N}_{\mathscr{H}_{\delta}}(\delta,\mathtt{M}_{\mathscr{H}_{\delta}}) e^{-t}, \\
\mathbbm{P}\Big[\lVert Z_n^X - Z_n^X \circ \mathtt{\Pi}_{0} \rVert_{\mathscr{H}_{\delta}}
> \sqrt{4d \min\{2\mathtt{M}_{\mathscr{H}_\delta},\mathtt{L}_{\mathscr{H}_\delta}2^{-K}\} 2^{-K} \mathtt{TV}_{\mathscr{H}_\delta}^{\ast} t} \Big]
\leq 2 \mathtt{N}_{\mathscr{H}_{\delta}}(\delta,\mathtt{M}_{\mathscr{H}_{\delta}}) e^{-t}.
\end{gather*}
\begin{myproof}{Lemma~\ref{sa-lem: X-process -- projection error}} Let $h \in \mathscr{H}$. Then, $|h(\mathbf{x}_i) - \mathtt{\Pi}_{0} h(\mathbf{x}_i)| \leq \min\{2 \mathtt{M}_{\mathscr{H}_{\delta}}, \mathtt{L}_{\mathscr{H}_{\delta}}\lVert \mathcal{V} \rVert_{\infty}\} = \mathtt{B}_{\mathscr{H}_{\delta}}$,
\begin{align*}
\mathbbm{E} \left[|h(\mathbf{x}_i) - \mathtt{\Pi}_{0} h(\mathbf{x}_i)|\right]
& = \sum_{0 \leq k < 2^{K}} \int_{\mathcal{C}_{0,k}} \Big| h(\mathbf{x}) - 2^{K} \int_{\mathcal{C}_{0,k}} h(\mathbf{y}) f_X(\mathbf{y}) d\mathbf{y} \Big| f_X(\mathbf{x}) d \mathbf{x} \\
& \leq \sum_{0 \leq k < 2^{K}} 2^{K} \int_{\mathcal{C}_{0,k}} \int_{\mathcal{C}_{0,k}}|h(\mathbf{x}) - h(\mathbf{y})| f_X(\mathbf{y}) f_X(\mathbf{x})d \mathbf{y} d \mathbf{x}.
\end{align*}
Using a change of variables $\mathbf{s} = \mathbf{y} - \mathbf{x}$ and the fact that $f_X$ is bounded above, we have
\begin{align*}
& \mathbbm{E} [|h(\mathbf{x}_i) - \mathtt{\Pi}_{0} h(\mathbf{x}_i)|] \\
&\leq \sum_{0 \leq k < 2^{K}} 2^{K} \int_{\mathcal{C}_{0,k} - \mathcal{C}_{0,k}} \int_{\mathcal{C}_{0,k}} |h(\mathbf{x}) - h(\mathbf{x} + \mathbf{s})| f_X(\mathbf{x} + \mathbf{s}) f_X(\mathbf{x}) \mathbbm{1}_{\mathcal{C}_{0,k}}(\mathbf{x} + \mathbf{s}) d \mathbf{x} d \mathbf{s} \\
&\leq 2^{K} \int_{\mathcal{V}} \int_{\mathcal{C}_{K,0}} |h(\mathbf{x}) - h(\mathbf{x} + \mathbf{s})| f_X(\mathbf{x} + \mathbf{s}) f_X(\mathbf{x}) d \mathbf{x} d \mathbf{s}.
\end{align*}
Let $(h_\ell)_{\ell \in \mathbb{N}}$ be any sequence of real-valued functions on $(\mathcal{X},\mathcal{B}(\mathcal{X}))$ such that
$h_{\ell} \rightarrow h$ $\mathfrak{m}$-almost surely, and are bounded by $2 \mathtt{M}_{\mathscr{H}}$ on $\mathcal{X}$. Since we assumed $\mathtt{M}_\mathscr{H} < \infty$, and $h_\ell$ and $h$ are bounded by $2 \mathtt{M}_\mathscr{H}$, by Dominated Convergence Theorem we have that
\begin{align*}
\int_{\mathcal{C}_{K,0}} \left|h(\mathbf{x}) - h(\mathbf{x} + \mathbf{s})\right| f_X(\mathbf{x}) d \mathbf{x}
&= \lim_{\ell \to \infty} \int_{\mathcal{C}_{K,0}} |h_\ell(\mathbf{x}) - h_\ell(\mathbf{x} + \mathbf{s})| f_X(\mathbf{x}) d \mathbf{x} \\
& \leq \sup_{\mathbf{x} \in \mathcal{X}}f_X(\mathbf{x}) \cdot \lim_{\ell \to \infty} \int_{\mathcal{X}} \int_0^{\lVert \mathbf{s} \rVert} \lVert \nabla h_{\ell}(\mathbf{x} + t \mathbf{s} / \lVert \mathbf{s} \rVert) \rVertrVert)}d t d \mathbf{x} \\
& \leq \sup_{\mathbf{x} \in \mathcal{X}}f_X(\mathbf{x}) \cdot \int_0^{\lVert \mathbf{s} \rVert} \lim_{\ell \to \infty} \int_{\mathcal{X}} \lVert \nabla h_{\ell}(\mathbf{x} + t \mathbf{s} / \lVert \mathbf{s} \rVert) \rVertrVert)} d \mathbf{x} d t \\
& \leq \sup_{\mathbf{x} \in \mathcal{X}}f_X(\mathbf{x}) \cdot \lVert \mathbf{s} \rVert \limsup_{\ell \to \infty}\mathtt{TV}_{\{h_\ell\}}.
\end{align*}
Since this holds for any sequence $(h_\ell)_{\ell \in \mathbb{N}}$ $h_{\ell} \rightarrow h$ $\mathfrak{m}$-almost surely, and are bounded by $2 \mathtt{M}_{\mathscr{H}}$ on $\mathcal{X}$, hence $\int_{\mathcal{X}} \left|h(\mathbf{x}) - h(\mathbf{x} + \mathbf{s})\right| d \mathbf{x} \leq \lVert \mathbf{s} \rVert \mathtt{TV}_{\{h\}}^*$. It follows that
\begin{align*}
& \mathbbm{E}[|h(\mathbf{x}_i) - \mathtt{\Pi}_{0} h(\mathbf{x}_i)|] \leq \left(\sup_{\mathbf{x} \in \mathcal{X}}f_X(\mathbf{x})\right)^2 2^{K} \mathfrak{m}(\mathcal{V}) \lVert \mathcal{V} \rVert_{\infty} \mathtt{TV}_{\{h\}}^*,
\end{align*}
and
\begin{align*}
\mathbbm{V} [h(\mathbf{x}_i) - \mathtt{\Pi}_{0} h (\mathbf{x}_i) ] \leq \min\{2 \mathtt{M}_{\mathscr{H}_{\delta}}, \mathtt{L}_{\mathscr{H}_{\delta}}\lVert \mathcal{V} \rVert_{\infty}\} \left(\sup_{\mathbf{x} \in \mathcal{X}}f_X(\mathbf{x})\right)^2 2^{K} \mathfrak{m}(\mathcal{V}) \lVert \mathcal{V} \rVert_{\infty} \mathtt{TV}_{\mathscr{H}_{\delta}}^* = \mathtt{V}_{\mathscr{H}_{\delta}},
\end{align*}
for all $h \in \mathscr{H}_{\delta}$. Then, by Bernstein inequality, for any $t > 0$,
\begin{align*}
\mathbbm{P}(|X_n(h) - X_n(\mathtt{\Pi}_{0} h)| \geq t)
& \leq 2 \exp \left(-\frac{\frac{1}{2}t^2 n}{n \mathtt{V}_{\mathscr{H}_{\delta}}+ \frac{1}{3} \mathtt{B}_{\mathscr{H}_{\delta}}t \sqrt{n}}\right)
\leq 2 \exp \left(- \frac{1}{2} \min \left\{\frac{\frac{1}{2}t^2 n}{n \mathtt{V}_{\mathscr{H}_{\delta}}}, \frac{\frac{1}{2}t^2 n}{\frac{1}{3} \mathtt{B}_{\mathscr{H}_{\delta}}t \sqrt{n}} \right\} \right).
\end{align*}
Set $u = \frac{1}{2} \min \left\{\frac{\frac{1}{2}t^2 n}{n \mathtt{V}_{\mathscr{H}_{\delta}}}, \frac{\frac{1}{2}t^2 n}{\frac{1}{3} \mathtt{B}_{\mathscr{H}_{\delta}}t \sqrt{n}} \right\} > 0$, then either $t = 2 \sqrt{\mathtt{V}_{\mathscr{H}_{\delta}}} \sqrt{u}$ or $t = \frac{4}{3} \frac{\mathtt{B}}{\sqrt{n}}u$. Hence $t \leq 2 \sqrt{\mathtt{V}_{\mathscr{H}_{\delta}}} \sqrt{u} + \frac{4}{3} \frac{\mathtt{B}_{\mathscr{H}_{\delta}}}{\sqrt{n}}u$. For any $u > 0$, $\mathbbm{P}(|X_n(h) - X_n(\mathtt{\Pi}_{0} h)| \geq 2 \sqrt{\mathtt{V}_{\mathscr{H}_{\delta}}} \sqrt{u} + \frac{4}{3} \frac{\mathtt{B}_{\mathscr{H}_{\delta}}}{\sqrt{n}}u) \leq 2 \exp(-u)$. The result for $\lVert X_n - X_n \circ \mathtt{\Pi}_{0} \rVert_{\mathscr{H}_{\delta}}$ then follows from a union bound. The result for $\lVert Z_n - Z_n \circ \mathtt{\Pi}_{0} \rVert_{\mathscr{H}_{\delta}}$ follows from the fact that $Z_n(h) - Z_n(\mathtt{\Pi}_{0} h)$ is a mean-zero Gaussian with variance $\mathbbm{V}[X_n(h) - X_n(\mathtt{\Pi}_{0} h)]$ and a union bound argument.
\end{myproof}
\subsection{Surrogate Measure and Normalizing Transformation}\label{sa-sec: X-Process -- normalizing transformation}
This section studies the properties of the surrogate measure $\mathbb{Q}_\mathscr{H}$ and normalizing transformation $\phi_\mathscr{H}$ introduced in condition (ii) of Theorem 1. The following lemma characterizes the connections between the original and the transformed parameters of $\mathscr{H}$ (Definitions 4 to 12) when deploying $\mathbb{Q}_\mathscr{H}$ and $\phi_\mathscr{H}$.
\begin{lemma}\label{sa-lem: normalizing transformation}
Suppose following conditions hold.
\begin{enumerate}[label=(\roman*)]
\item $\mathscr{H}$ is a real-valued pointwise measurable class of functions on $(\mathbb{R}^d,\mathcal{B}(\mathbb{R}^d),\mathbbm{P}_X)$.
\item There exists a surrogate measure $\mathbb{Q}_\mathscr{H}$ for $\mathbbm{P}_X$ with respect to $\mathscr{H}$ such that $\mathbb{Q}_\mathscr{H} = \operatorname*{\mathfrak{m}} \circ \phi_\mathscr{H}$, where the \textit{normalizing transformation} $\phi_{\mathscr{H}}: \mathcal{Q}_\mathscr{H} \mapsto [0,1]^d$ is a diffeomorphism.
\end{enumerate}
Let $\widetilde{\mathscr{H}} = \{h \circ \phi_{\mathscr{H}}^{-1}: h \in \mathscr{H}\}$. Then,
\begin{gather*}
\mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d} = \mathtt{M}_{\mathscr{H},\mathcal{Q}_\mathscr{H}},
\qquad
\mathtt{E}_{\widetilde{\mathscr{H}},[0,1]^d} = \mathtt{E}_{\mathscr{H},\mathcal{Q}_\mathscr{H}}, \\
\mathtt{N}_{\widetilde{\mathscr{H}},[0,1]^d}(\varepsilon,\mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d})
= \mathtt{N}_{\mathscr{H},\mathcal{Q}_\mathscr{H}}(\varepsilon,\mathtt{M}_{\mathscr{H},\mathcal{Q}_\mathscr{H}}),
\qquad
\varepsilon \in(0,1), \\
\mathtt{L}_{\widetilde{\mathscr{H}},[0,1]^d} \leq \mathtt{c}_2 \mathtt{L}_{\mathscr{H},\mathcal{Q}_{\mathscr{H}}}, \qquad \mathtt{c}_2 = \sup_{\mathbf{x} \in \mathcal{Q}_{\mathscr{H}}} \frac{1}{\sigma_{d}(\nabla \phi_{\mathscr{H}}(\mathbf{x}))}, \\
\mathtt{TV}_{\widetilde{\mathscr{H}},[0,1]^d}^* \leq d^{-1}\mathtt{c}_1 \mathtt{TV}_{\mathscr{H},\mathcal{Q}_{\mathscr{H}}},
\qquad
\mathtt{c}_1 = d \sup_{\mathbf{x} \in \mathcal{Q}_{\mathscr{H}}} \prod_{j = 1}^{d-1} \sigma_j(\nabla \phi_{\mathscr{H}}(\mathbf{x})), \\
\mathtt{K}_{\widetilde{\mathscr{H}},[0,1]^d}^* \leq d^{-1/2} \mathtt{c}_3 \mathtt{K}_{\mathscr{H},\mathcal{Q}_{\mathscr{H}}},
\qquad \mathtt{c}_3 = 2^{d-1} d^{d/2-1} \mathtt{c}_1 \mathtt{c}_2^{d-1}.
\end{gather*}
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: normalizing transformation}}
The first three identities are self-evident. Consider next the relation between $\mathtt{L}_{\widetilde{\mathscr{H}},[0,1]^d}$ and $\mathtt{L}_{\mathscr{H},\mathcal{Q}_\mathscr{H}}$: for any $h \in \mathscr{H}$, using a change of variables and the differentiability of $\phi_{\mathscr{H}}$,
\begin{align*}
\mathtt{L}_{\{h \circ \phi_{\mathscr{H}}^{-1} \},[0,1]^d}
& = \sup_{\mathbf{u},\mathbf{u}^{\prime} \in [0,1]^d} \frac{|h \circ \phi_{\mathscr{H}}^{-1}(\mathbf{u}) - h \circ \phi_{\mathscr{H}}^{-1}(\mathbf{u}^{\prime})|}{\lVert \mathbf{u} - \mathbf{u}^{\prime} \rVert} \\
& \leq \sup_{\mathbf{x},\mathbf{x}^{\prime} \in \mathcal{Q}_\mathscr{H}} \frac{|h(\mathbf{x}) - h(\mathbf{x}^{\prime})|}{\lVert \mathbf{x} - \mathbf{x}^{\prime} \rVert} \frac{\lVert \mathbf{x} - \mathbf{x}^{\prime} \rVert}{\lVert \phi_{\mathscr{H}}(\mathbf{x}) - \phi_{\mathscr{H}}(\mathbf{x}^{\prime}) \rVert} \\
& \leq \mathtt{L}_{\{h \},\mathcal{Q}_{\mathscr{H}}} \sup_{\mathbf{u}, \mathbf{u}^{\prime} \in [0,1]^d} \frac{|\phi_{\mathscr{H}}^{-1}(\mathbf{u}) - \phi_{\mathscr{H}}^{-1}(\mathbf{u}^{\prime})|}{\lVert \mathbf{u} - \mathbf{u}^{\prime} \rVert} \\
& \leq \mathtt{L}_{\{h \},\mathcal{Q}_{\mathscr{H}}} \sup_{\mathbf{z} \in [0,1]^d}\sigma_{1}(\nabla \phi_{\mathscr{H}}^{-1}(\mathbf{z})) \\
& = \mathtt{L}_{\{h \},\mathcal{Q}_{\mathscr{H}}} \sup_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}} \sigma_d(\nabla \phi_{\mathscr{H}}(\mathbf{x}))^{-1},
\end{align*}
and the result follows.
Now consider the relation between $\mathtt{TV}_{\widetilde{\mathscr{H}},[0,1]^d}$ and $\mathtt{TV}_{\mathscr{H},\mathcal{Q}_{\mathscr{H}}}$. First suppose all functions in $\mathscr{H}$ are differentiable, an integration by parts based on the definition of uniform total variation (Definition 5) and a change of variables calculation gives
\begin{align*}
\mathtt{TV}_{\{h \circ \phi_{\mathscr{H}}^{-1} \},[0,1]^d}
& = \sup_{\varphi \in \mathscr{D}_d([0,1]^d)} \int_{[0,1]^d} h \circ \phi_{\mathscr{H}}^{-1} (\mathbf{x}) \operatorname{div}(\varphi)(\mathbf{x})d \mathbf{x}/ \lVert \lVert \varphi \rVert_2 \rVert_{\infty}\\
& = \int_{\mathbf{u} \in [0,1]^d} \lVert \nabla ( h \circ \phi_{\mathscr{H}}^{-1}) (\mathbf{u}) \rVert d \mathbf{u} \\
& = \int_{\mathbf{u} \in [0,1]^d} \lVert \nabla \phi_{\mathscr{H}}^{-1}(\mathbf{u})^{\top} \nabla h(\phi_{\mathscr{H}}^{-1}(\mathbf{u})) \rVert d \mathbf{u} \\
& = \int_{\mathcal{Q}_\mathscr{H}} \lVert \nabla \phi_{\mathscr{H}}^{-1}(\phi_{\mathscr{H}}(\mathbf{x}))^{\top} \nabla h(\mathbf{x}) \rVert \cdot |\operatorname{det}(\nabla \phi_{\mathscr{H}}(\mathbf{x}))| d \mathbf{x}\\
& \leq \int_{\mathcal{Q}_\mathscr{H}} \lVert \nabla h(\mathbf{x}) \rVert d \mathbf{x} \sup_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}} |\operatorname{det}(\nabla \phi_{\mathscr{H}}(\mathbf{x}))|\cdot \lVert \nabla \phi_{\mathscr{H}}^{-1}(\phi_{\mathscr{H}}(\mathbf{x})) \rVert \\
& \leq \mathtt{c}_1 \mathtt{TV}_{\{h\},\mathcal{Q}_{\mathscr{H}}},
\end{align*}
where in the last line we have used $|\operatorname{det}(\nabla \phi_{\mathscr{H}}(\mathbf{x}))| = \prod_{j = 1}^d \sigma_j(\nabla \phi_{\mathscr{H}}(\mathbf{x}))$, and since $\phi_{\mathscr{H}}$ is a diffeomorphism, $\lVert \nabla \phi_{\mathscr{H}}^{-1}(\phi_{\mathscr{H}}(\mathbf{x})) \rVert = \sigma_1(\nabla \phi_{\mathscr{H}}^{-1}(\phi_{\mathscr{H}}(\mathbf{x}))) = \sigma_d(\nabla \phi_{\mathscr{H}}(\mathbf{x}))^{-1}$. Now consider $\mathscr{H}$ which contains possibly non-differentiable functions. Take $\psi: \mathbb{R}^d \to \mathbb{R}$ to be any smooth function with compact support such that $\int_{\mathbb{R}^d} \psi(\mathbf{z}) d \mathbf{z} = 1$, and take $\psi_\varepsilon(\cdot) = \varepsilon^{-d} \psi(\cdot/\varepsilon)$. For each $\ell \in \mathbb{N}$, define $h_\ell = h \ast \psi_{\varepsilon_\ell}$, where $(\varepsilon_\ell)_{\ell \in \mathbb{N}}$ is a sequence of non-increasing real positive numbers converging to zero with $\varepsilon_1$ small enough. Then
\begin{align*}
\mathtt{TV}_{\widetilde{\mathscr{H}},[0,1]^d}^\ast & \leq \sup_{h \in \mathscr{H}} \limsup_{\ell \to \infty} \mathtt{TV}_{\{h_\ell \circ \phi_\mathscr{H}^{-1}\},[0,1]^d}
= \sup_{h \in \mathscr{H}} \limsup_{\ell \to \infty} \mathtt{c}_1 \mathtt{TV}_{\{h_\ell\},\mathcal{Q}_{\mathscr{H}}}
\leq \mathtt{c}_1 \mathtt{TV}_{\mathscr{H},\mathcal{Q}_{\mathscr{H}}},
\end{align*}
where the first inequality is due to $(h_\ell)_{\ell \in \mathbb{N}}$ being a particular sequence satisfying Definition~\ref{sa-defn: smooth tv}, the second inequality from Lemma~\ref{sa-lem: normalizing transformation}, the third inequality due to $\mathtt{TV}_{\{h \ast \psi\},\mathcal{Q}_{\mathscr{H}}} \leq \mathtt{TV}_{\{h\},\mathcal{Q}_{\mathscr{H}}}$ for any smooth $\psi$.
Moreover, let $\mathcal{C} \subseteq \mathbb{R}^d$ be a cube with edges of length $\mathtt{a}$ parallel to the coordinate axises. Then, $\phi_{\mathscr{H}}^{-1}(\mathcal{C})$ is contained in another cube $\mathcal{C}^{\prime}$ with edges of length at most $2 \sqrt{d} \sup_{\mathbf{x} \in [0,1]^d}\lVert \nabla \phi_{\mathscr{H}}^{-1}(\mathbf{x}) \rVert \mathtt{a}$. Again, we first assume that each $h \in \mathscr{H}$ is differentiable. Using a change of variables for the total variation calculation and the definition of $\mathtt{K}_{\{h\},\mathcal{Q}_{\mathscr{H}}}$ (Definition 5), for any $h \in \mathscr{H}$,
\begin{align*}
\mathtt{TV}_{\{h \circ \phi_{\mathscr{H}}^{-1}\},\mathcal{C}}
& = \int_{\mathcal{C}} \lVert \nabla(h \circ \phi_{\mathscr{H}}^{-1})(\mathbf{u}) \rVert d \mathbf{u} \\
& \leq \int_{\mathcal{C}^{\prime}} \lVert \nabla(h \circ \phi_{\mathscr{H}}^{-1})(\phi_{\mathscr{H}}(\mathbf{x})) \rVert \operatorname{det}(\nabla \phi_{\mathscr{H}}(\mathbf{x}))d \mathbf{x} \\
& \leq \int_{\mathcal{C}^{\prime}} \lVert \nabla h(\mathbf{x}) \rVert d \mathbf{x} \sup_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}} |\operatorname{det}(\nabla \phi_{\mathscr{H}}(\mathbf{x}))| \lVert \nabla \phi_{\mathscr{H}}^{-1}(\phi_{\mathscr{H}}(\mathbf{x})) \rVert \\
& \leq \mathtt{K}_{\{h\},\mathcal{Q}_{\mathscr{H}}} \Big(2 \sqrt{d} \sup_{\mathbf{x} \in [0,1]^d}\lVert \nabla \phi_{\mathscr{H}}^{-1}(\mathbf{x}) \rVert \mathtt{a}\Big)^{d-1} \sup_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}} |\operatorname{det}(\nabla \phi_{\mathscr{H}}(\mathbf{x}))| \lVert \nabla \phi_{\mathscr{H}}^{-1}(\phi_{\mathscr{H}}(\mathbf{x})) \rVert\\
& = d^{-1} (2 \sqrt{d})^{d-1} \mathtt{c}_1 \mathtt{c}_2^{d-1}
\mathtt{K}_{\{h\},\mathcal{Q}_{\mathscr{H}}} \mathtt{a}^{d-1} \\
& = d^{-1/2} \mathtt{c}_3 \mathtt{K}_{\{h\},\mathcal{Q}_{\mathscr{H}}} \mathtt{a}^{d-1} ,
\end{align*}
which implies
\begin{align*}
\mathtt{K}_{\{\widetilde{h}\},[0,1]^d} \leq d^{-1/2} \mathtt{c}_3 \mathtt{K}_{\{h\},\mathcal{Q}_{\mathscr{H}}}.
\end{align*}
By similar smoothing arguments as for the $\mathtt{TV}$ terms, we can also show that $\mathtt{K}_{\widetilde{\mathscr{H}},[0,1]^d}^* \leq d^{-1/2} \mathtt{c}_3 \mathtt{K}_{\mathscr{H},[0,1]^d}$ even when $\mathscr{H}$ contains possibly non-differentiable functions.
\end{myproof}
\begin{lemma}\label{sa-lem: statements 3.1}
We recap the statements in Section 3.1 and present their proofs.
\begin{itemize}
\item \textbf{Case 1: Uniform on Rectangle}. Suppose that $\mathbf{x}_i\thicksim\mathsf{Uniform}(\mathcal{X})$ with $\mathcal{X}=\times_{l = 1}^d[\mathsf{a}_l, \mathsf{b}_l]$, where $-\infty < \mathsf{a}_l < \mathsf{b}_l < \infty$, $l=1,2,\dots,d$. Setting $\mathbb{Q}_\mathscr{H} = \mathbbm{P}_X$, a valid normalizing transformation is $\phi_{\mathscr{H}}(x_1, \cdots, x_d) = ((\mathsf{b}_1 - \mathsf{a}_1)^{-1}(x_1 - \mathsf{a}_1), \cdots, (\mathsf{b}_d - \mathsf{a}_d)^{-1}(x_d - \mathsf{a}_d))$, which verifies assumption (ii) in Theorem 1. In this case, $\mathtt{c}_1 = d \max_{1 \leq l \leq d}|\mathsf{b}_l - \mathsf{a}_l| \prod_{l = 1}^d |\mathsf{b}_l - \mathsf{a}_l|^{-1}$, $\mathtt{c}_2 = \max_{1 \leq l \leq d}|\mathsf{b}_l - \mathsf{a}_l|$ and $\mathtt{c}_3 = 2^{d-1} d^{d/2} \max_{1 \leq l \leq d}|\mathsf{b}_l - \mathsf{a}_l|^d \prod_{l = 1}^d |\mathsf{b}_l - \mathsf{a}_l|^{-1}$.
\item \textbf{Case 2: Rectangular $\mathcal{Q}_{\mathscr{H}}$}. Suppose that $\mathbb{Q}_{\mathscr{H}}$ admits a Lebesgue density $f_Q$ supported on $\mathcal{Q}_{\mathscr{H}} = \times_{l = 1}^d [\mathsf{a}_l, \mathsf{b}_l]$, $- \infty \leq \mathsf{a}_l < \mathsf{b}_l \leq \infty$. Then, the Rosenblatt transformation $\phi_{\mathscr{H}} = T_{\mathbb{Q}_\mathscr{H}}$ is a normalizing transformation, and we obtain
\begin{align*}
\mathtt{c}_1 & = d \sup_{\mathbf{u} \in \mathcal{Q}_{\mathscr{H}}} \frac{f_Q(\mathbf{u})}{\min \{f_{Q,1}(u_1), f_{Q,2|1}(u_2|u_1), \cdots, f_{Q,d|-d}(u_d|u_1,\cdots,u_{d-1})\}}, \\
\mathtt{c}_2 & = \sup_{\mathbf{u} \in \mathcal{Q}_{\mathscr{H}}} \frac{1}{\min \{f_{Q,1}(u_1), f_{Q,2|1}(u_2|u_1), \cdots, f_{Q,d|-d}(u_d|u_1,\cdots,u_{d-1})\}},
\end{align*}
and $\mathtt{c}_3 = 2^{d-1} d^{d/2-1} \mathtt{c}_1 \mathtt{c}_2^{d-1}$.
This case covers several examples of interest, which give primitive conditions for assumption (ii) in Theorem 1:
\begin{enumerate}[label=(\alph*)]
\item Suppose $\mathcal{Q}_\mathscr{H} = \times_{l = 1}^d [\mathsf{a}_l, \mathsf{b}_l]$ is bounded. Then,
\begin{align*}
\mathtt{c}_1 \leq d \frac{\overline{f}_Q^2}{\underline{f}_Q} \overline{\mathcal{Q}}_{\mathscr{H}}
\qquad\text{and}\qquad
\mathtt{c}_2 \leq \frac{\overline{f}_Q}{\underline{f}_Q} \overline{\mathcal{Q}}_{\mathscr{H}}.
\end{align*}
\item Suppose $\mathcal{Q}_\mathscr{H} = \times_{l = 1}^d [\mathsf{a}_l, \mathsf{b}_l]$ is unbounded. To fix ideas, let $\mathbf{x}_i \thicksim \mathsf{Normal} (\boldsymbol{\mu}, \boldsymbol{\Sigma})$. Then, setting $\mathbb{Q}_\mathscr{H} = \mathbbm{P}_X$ and $\phi_{\mathscr{H}} = T_{\mathbbm{P}_X}$ also gives a valid normalizing transformation, with
\begin{align*}
\mathtt{c}_1
& \leq d \sup_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}} \max \{f_{X,1}(x_1), f_{X,2|1}(x_2|x_1), \cdots, f_{X,d|-d}(x_d|x_{-d})\}^{d-1}\\
& \leq d \min_{1 \leq k \leq d} \{\boldsymbol{\Sigma}_{k,k} - \boldsymbol{\Sigma}_{k,1:k-1} \boldsymbol{\Sigma}_{1:k-1,1:k-1}^{-1}\boldsymbol{\Sigma}_{1:k-1,k}\}^{-(d-1)/2}\nonumber
\end{align*}
bounded, but $\mathtt{c}_2$ (and hence $\mathtt{c}_3$) unbounded.
\end{enumerate}
\item \textbf{Case 3: Non-Rectangular $\mathcal{Q}_\mathscr{H}$}.
Suppose that $\mathbb{Q}_{\mathscr{H}}$ admits a Lebesgue density $f_Q$ supported on $\mathcal{Q}_{\mathscr{H}}$, and there exists a diffeomorphism $\chi:\mathcal{Q}_{\mathscr{H}}\mapsto[0,1]^d$. Setting $\phi_{\mathscr{H}} = T_{\mathbb{Q}_\mathscr{H} \circ \chi^{-1}} \circ \chi$ gives a valid normalizing transformation, with
\begin{align*}
\mathtt{c}_1 \leq d\frac{\overline{f}_Q^2}{\underline{f}_Q} \mathtt{S}_\chi
\qquad\text{and}\qquad
\mathtt{c}_2 \leq \frac{\overline{f}_Q}{\underline{f}_Q} \mathtt{S}_\chi,
\end{align*}
where $\mathtt{S}_\chi = \frac{\sup_{\mathbf{x} \in [0,1]^d} |\operatorname{det}(\nabla\chi^{-1}(\mathbf{x}))|}{\inf_{\mathbf{x} \in [0,1]^d} |\operatorname{det}(\nabla\chi^{-1}(\mathbf{x}))|}\lVert \lVert \nabla \chi^{-1} \rVert_2 \rVert_{\infty}$.
\end{itemize}
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: statements 3.1}}
We consider the three cases separately.\\
\noindent \textbf{Case 1: Uniform on Rectangle}. For every $\mathbf{x} \in \mathcal{Q}_\mathscr{H}$, the singular values of $\nabla \phi_\mathscr{H}(\mathbf{x})$ are $(\mathtt{b}_1 - \mathtt{a}_1)^{-1}, \cdots, (\mathtt{b}_d - \mathtt{a}_d)^{-1}$. The values of $\mathtt{c}_1$ and $\mathtt{c}_2$ (and hence $\mathtt{c}_3$) then follow.\\
\noindent \textbf{Case 2: Rectangular $\mathcal{Q}_\mathscr{H}$}. We start with a proof for a general result for $\mathtt{c}_1,\mathtt{c}_2,\mathtt{c}_3$, and then prove upper bounds for (a) and (b).
\begin{enumerate}
\item \underline{The General Case}. Since $\mathbb{Q}$ has a Lebesgue density $f_{Q}$,
\begin{align*}
\nabla T_{\mathbb{Q}}(\mathbf{x}) =
\begin{bmatrix}
f_{Q,1}(x_1) & 0 & \cdots & 0 \\
\ast & f_{Q,2|1}(x_2|x_1) & \cdots & 0 \\
\ast & \ast & \vdots & 0 \\
\ast & \ast & \cdots & f_{Q,d|1,\cdots,d-1}(x_d|x_1, \cdots, x_{d-1})
\end{bmatrix}, \qquad \mathbf{x} \in \mathcal{Q}_\mathscr{H}.
\end{align*}
Because the singular values of $\nabla \phi_{\mathscr{H}}(\mathbf{x}) = \nabla T_{\mathbb{Q}}(\mathbf{x})$ are the values on the diagonal,
\begin{align*}
\mathtt{c}_1 & = d \sup_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}} \frac{f_Q(\mathbf{x})}{\min \{f_{Q,1}(x_1), f_{Q,2|1}(x_2|x_1), \cdots, f_{Q,d|-d}(x_d|x_{-d})\}}, \\
\mathtt{c}_2 & = \sup_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}} \max \{f_{Q,1}(x_1)^{-1}, f_{Q,2|1}(x_2|x_1)^{-1}, \cdots, f_{Q,d|-d}(x_d|x_{-d})^{-1}\}.
\end{align*}
\item \underline{Case (a): $\mathcal{Q}_\mathscr{H} = \times_{l = 1}^d [\mathsf{a}_l, \mathsf{b}_l]$ is bounded}. Since we assumed the existence of an $\mathcal{Q}_\mathscr{H}$ that is compact and $\inf_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}}f_Q(\mathbf{x}) > 0$, integrating on the rectangle gives
\begin{align*}
\frac{\underline{f}_Q}{\overline{f}_Q} \frac{1}{\overline{L}} \leq f_{Q,k|1,\cdots,k-1}(x_k|x_1, \cdots,x_{k-1})
& = \frac{\int_{\prod_{l = k+1}^{d}[a_l,b_l]}f_Q(x_1,\cdots,x_k,\mathbf{z})d \mathbf{z}}{\int_{\prod_{l = k}^{d}[a_l,b_l]}f_Q(x_1, \cdots,x_{k-1},\mathbf{u}) d\mathbf{u}} \leq \frac{\overline{f}_Q}{\underline{f}_Q} \frac{1}{\underline{L}},
\end{align*}
where $\overline{L} = \max_{1 \leq l \leq d} (b_l - a_l)$ and $\underline{L} = \min_{1 \leq l \leq d} (b_l - a_l)$. Plugging in the generic bounds for $\mathtt{c}_1$ and $\mathtt{c}_2$,
\begin{gather*}
\mathtt{c}_1 \leq d \bigg(\frac{\overline{f}_Q}{\underline{f}_Q}\max_{1 \leq k \leq d} \frac{1}{b_k - a_k}\bigg)^{d-1}
\qquad\text{and}\qquad
\mathtt{c}_2 \leq \frac{\overline{f}_Q}{\underline{f}_Q}\max_{1 \leq k \leq d} |b_k - a_k|.
\end{gather*}
\item \underline{Case (b): $\mathbf{x}_i \thicksim \mathsf{Normal}(\boldsymbol{\mu}, \boldsymbol{\Sigma})$}. The bound on $\mathtt{c}_1$ follows from properties of the conditional distribution of multivariate Gaussian distribution. Since $\inf_{\mathbf{x} \in \mathbb{R}^{k}}f_{k|1,\cdots,k-1}(x_k|x_1,\cdots,x_{k-1}) = 0$ for $1 \leq k \leq d$, $\mathtt{c}_2$ (and hence $\mathtt{c}_3$) are unbounded.
\end{enumerate}
\noindent \textbf{Case 3: Non-rectangular $\mathcal{Q}_\mathscr{H}$}. Since both $T_{\mathbb{Q}_\mathscr{H}}$ and $\chi$ are diffeomorphisms, we can use chain rule to get,
\begin{align*}
\mathtt{c}_1 & = \sup_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}} \prod_{j = 1}^{d-1} \sigma_j (\nabla \phi_{\mathscr{H}}(\mathbf{x})) \\
& = \sup_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}} \operatorname{det}(\nabla \phi_{\mathscr{H}}(\mathbf{x})) \lVert \nabla \phi_{\mathscr{H}}^{-1}(\phi_{\mathscr{H}}(\mathbf{x})) \rVert \\
& \leq \sup_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}}\det(\nabla T_{\mathbb{Q}_\mathscr{H}}(\chi(\mathbf{x}))) \det(\nabla \chi(\mathbf{x})) \lVert \nabla \chi^{-1}(\chi(\mathbf{x})) \rVert_{2} \lVert \nabla T_{\mathbb{Q}_\mathscr{H}}^{-1}(\phi_{\mathscr{H}}(\mathbf{x})) \rVert_{2}.
\end{align*}
Take $\mathbf{w}_i = \chi(\mathbf{x}_i)$, and denote by $f_W$ the density of $\mathbf{w}_i$. Then
\begin{align*}
\nabla T_{\mathbb{Q}_\mathscr{H}}(x_1, \cdots, x_d)
& =
\begin{bmatrix}
f_{W_1}(x_1) & 0 & \cdots & 0\\
\ast & f_{W_2|W_1}(x_2|x_1) & \cdots & 0\\
\vdots & \vdots & & \vdots \\
\ast & \ast & \cdots & f_{W_d|W_1,\cdots,W_{d-1}}(x_d|x_1,\cdots,x_{d-1})
\end{bmatrix},
\end{align*}
where $\ast$ denotes values that won't affect determinant or operator norm of the matrix $\nabla T_{\mathbb{Q}_\mathscr{H}}$. Hence,
\begin{align*}
\det(\nabla T_{\mathbb{Q}_\mathscr{H}}(\chi(\mathbf{x}))) = f_{W}(\chi(\mathbf{x})) = f_X(\mathbf{x}) |\det(\nabla \chi(\mathbf{x}))|^{-1}
\end{align*}
and
\begin{align*}
\lVert \nabla T_{\mathbb{Q}_\mathscr{H}}^{-1}(\phi_{\mathscr{H}}(\mathbf{x})) \rVert_{2}
& = \sigma_d(\nabla T_{\mathbb{Q}_\mathscr{H}}(\chi(\mathbf{x})))^{-1}
\leq \frac{\sup_{\mathbf{w} \in [0,1]^d}f_W(\mathbf{w})}{\inf_{\mathbf{w} \in [0,1]^d} f_W(\mathbf{w})} \\
& \leq \frac{\sup_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}}f_X(\mathbf{x})}{\inf_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}} f_X(\mathbf{x})} \cdot \frac{\sup_{\mathbf{x} \in [0,1]^d}|\det(\nabla \chi^{-1}(\mathbf{x}))|}{\inf_{\mathbf{x} \in [0,1]^d}|\det(\nabla \chi^{-1}(\mathbf{x}))|}.
\end{align*}
Putting together, we have
\begin{align*}
\mathtt{c}_1 \leq \frac{\overline{f}_X^2}{\underline{f}_X} \mathtt{S}_{\chi},
\end{align*}
with $\overline{f}_X = \sup_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}}f_X(\mathbf{x})$, $\underline{f}_X = \inf_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}}f_X(\mathbf{x})$, $\mathtt{S}_\chi = \frac{\sup_{\mathbf{x} \in [0,1]^d} |\operatorname{det}(\nabla\chi^{-1}(\mathbf{x}))|}{\inf_{\mathbf{x} \in [0,1]^d} |\operatorname{det}(\nabla\chi^{-1}(\mathbf{x}))|}\lVert \lVert \nabla\chi^{-1} \rVert_{2} \rVert_{\infty}$. Also,
\begin{align*}
\mathtt{c}_2 & = \sup_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}} \lVert \nabla \phi_{\mathscr{H}}^{-1}(\phi_{\mathscr{H}}(\mathbf{x})) \rVert_{2}
\leq \sup_{\mathbf{u} \in [0,1]^d} \lVert \nabla \chi^{-1}(\mathbf{u}) \rVert_{2} \sup_{\mathbf{u} \in [0,1]^d} \lVert \nabla T_{\mathbb{Q}_\mathscr{H}}^{-1}(\mathbf{u}) \rVert_{2} \\
& \leq \sup_{\mathbf{u} \in [0,1]^d} \lVert \nabla \chi^{-1}(\mathbf{u}) \rVert_{2} \frac{\sup_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}}f_X(\mathbf{x})}{\inf_{\mathbf{x} \in \mathcal{Q}_\mathscr{H}} f_X(\mathbf{x})} \cdot \frac{\sup_{\mathbf{x} \in [0,1]^d}|\det(\nabla \chi^{-1}(\mathbf{x}))|}{\inf_{\mathbf{x} \in [0,1]^d}|\det(\nabla \chi^{-1}(\mathbf{x}))|} \leq \frac{\overline{f}_X}{\underline{f}_X} \mathtt{S}_{\chi}.
\end{align*}
This completes the proof.
\end{myproof}
\subsection{General Result: Proof of Theorem 1}\label{sa-sec: X-Process -- Proof of Main Theorem}
First, we make a reduction through the surrogate measure and normalizing transformation. We want to show that under assumption (ii) in Theorem 1, the empirical process $(X_n(h): h \in \mathscr{H})$ can be written as an empirical process based on i.i.d $\mathsf{Uniform}([0,1]^d)$ random variables. Let $\mathcal{Z}_{\mathscr{H}} = \mathcal{X} \cap \operatorname{Supp}(\mathscr{H})$. Since $\mathbb{Q}_\mathscr{H} = \operatorname*{\mathfrak{m}} \circ \phi_\mathscr{H}$ by Assumption (ii) in Theorem 1, and $\mathbb{Q}_\mathscr{H}|_{\mathcal{Z}_\mathscr{H}} = \mathbbm{P}_X|_{\mathcal{Z}_\mathscr{H}}$,
\begin{align*}
\mathbbm{P}_X|_{\mathcal{Z}_\mathscr{H}} = \operatorname*{\mathfrak{m}} \circ \phi_{\mathscr{H}}|_{\mathcal{Z}_\mathscr{H}}.
\end{align*}
To define the $\mathsf{Uniform}([0,1]^d)$ random variables on the probability space that $\mathbf{x}_i$'s live in, we define a joint probability measure $\mathbb{O}$ on $(\mathbb{R}^d \times \mathbb{R}^d, \mathcal{B}(\mathbb{R}^{2d}))$ such that for all $A \in \mathcal{B}(\mathbb{R}^{2d})$:
\begin{align*}
\mathbb{O}(A \cap (\mathcal{Z}_{\mathscr{H}} \times \mathcal{Z}_{\mathscr{H}}))
& = \mathbbm{P}_X(\Pi_{1:d}(A \cap \{(\mathbf{x},\mathbf{x}): \mathbf{x} \in \mathcal{Z}_{\mathscr{H}}\})), \\
\mathbb{O}(A \cap (\mathcal{Z}_{\mathscr{H}} \times \mathcal{Z}_{\mathscr{H}}^c))
& = \mathbb{O}(A \cap (\mathcal{Z}_{\mathscr{H}}^c \times \mathcal{Z}_{\mathscr{H}})) = 0, \\
\mathbb{O}(A \cap (\mathcal{Z}_{\mathscr{H}}^c \times \mathcal{Z}_{\mathscr{H}}^c))
& = \int_{\mathcal{Z}_{\mathscr{H}}^c \cap \Pi_{d+1:2d}(A)} \frac{\mathbbm{P}_X(A^{\mathbf{u}} \cap \mathcal{Z}_{\mathscr{H}}^c)}{\mathbbm{P}_X(\mathcal{Z}_{\mathscr{H}}^c)} d (\mathfrak{m} \circ \phi_{\mathscr{H}})(\mathbf{u}),
\end{align*}
where $\Pi_{1:d}(A) = \{\mathbf{x} \in \mathbb{R}^d: (\mathbf{x},\mathbf{u}) \in A \text{ for some } \mathbf{u} \in \mathbb{R}^{d}\}$, $\Pi_{d+1:2d}(A) = \{\mathbf{u} \in \mathbb{R}^d: (\mathbf{x},\mathbf{u}) \in A \text{ for some } \mathbf{x} \in \mathbb{R}^{d}\}$, and $A^{\mathbf{u}} = \{\mathbf{x} \in \mathbb{R}^d: (\mathbf{x},\mathbf{u}) \in A\}$. See Figure~\ref{sa-fig: O measure} for a graphical illustration.
\begin{figure}
\centering
\includegraphics[width=0.5\linewidth]{graphs/Qplot.png}
\caption{Illustration of $\mathbb{O}$. $\mathbb{O}$ concentrates on $\{(\mathbf{x},\mathbf{x}): \mathbf{x} \in \mathcal{Z}_{\mathscr{H}}\}$ in $\mathcal{Z}_{\mathscr{H}} \times \mathcal{Z}_{\mathscr{H}}$, agrees with the zero measure on $\mathcal{Z}_{\mathscr{H}} \times \mathcal{Z}_{\mathscr{H}}^c$ and $\mathcal{Z}_{\mathscr{H}}^c \times \mathcal{Z}_{\mathscr{H}}$, and agrees with the product measure of $\mathbbm{P}_X \otimes (\mathfrak{m} \circ \phi_{\mathscr{H}})$ on $\mathcal{Z}_{\mathscr{H}}^c \times \mathcal{Z}_{\mathscr{H}}^c$.}
\label{sa-fig: O measure}
\end{figure}
Then we can check that (i) the marginals of $\mathbb{O}$ are $\mathbbm{P}_X$ and $\mathfrak{m} \circ \phi_{\mathscr{H}}$, respectively; (ii) $\mathbb{O}|_{\mathcal{Z}_{\mathscr{H}} \times \mathbb{R}^d \cup \mathbb{R}^d \times \mathcal{Z}_{\mathscr{H}}}$ is supported on $\{(\mathbf{x},\mathbf{x}): \mathbf{x} \in \mathcal{Z}_{\mathscr{H}}\}$. By Skorohod embedding \citep[Lemma 3.35]{dudley2014uniform}, on a possibly enlarged probability space, there exists a $\mathbf{u}_i, 1 \leq i \leq n$ i.i.d. $\operatorname{Uniform}([0,1]^d)$ such that $(\mathbf{x}_i,\phi_\mathscr{H}^{-1}(\mathbf{u}_i))$ has joint law $\mathbb{O}$. In particular, if $\mathbf{x}_i \in \mathcal{Z}_{\mathscr{H}}$, then $\mathbf{x}_i = \phi_{\mathscr{H}}^{-1}(\mathbf{u}_i)$; if $\mathbf{x}_i \in \mathcal{Z}_{\mathscr{H}}^c$, then $\phi_{\mathscr{H}}^{-1}(\mathbf{u}_i) \in \mathcal{Z}_{\mathscr{H}}^c$, and since $\mathcal{Q}_\mathscr{H} \subseteq \mathcal{X} \cup (\cap_{h \in \mathscr{H}} \operatorname{Supp}(h)^c)$, $\phi_{\mathscr{H}}^{-1}(\mathbf{u}_i) \in \cap_{h \in \mathscr{H}} \operatorname{Supp}(h)^c$.
Thus, we take $\widetilde{h} = h \circ \phi_{\mathscr{H}}^{-1}$, and consider the new class of functions $\widetilde{\mathscr{H}} = \{\widetilde{h}: h \in \mathscr{H}\}$. For any $h \in \mathscr{H}$,
\begin{align*}
X_n(h) & = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \big[h(\mathbf{x}_i) - \mathbbm{E}[h(\mathbf{x}_i)]\big] \\
& = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \big[h(\phi_{\mathscr{H}}^{-1}(\mathbf{u}_i)) - \mathbbm{E}[h(\phi_{\mathscr{H}}^{-1}(\mathbf{u}_i))]\big]\\
& = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \big[\widetilde{h}(\mathbf{u}_i) - \mathbbm{E}[\widetilde{h}(\mathbf{u}_i)]\big],
\end{align*}
where the second equality follows because $\mathbf{x}_i=\phi_{\mathscr{H}}^{-1}(\mathbf{u}_i)$ on the event $\{\mathbf{x}_i \in \mathcal{Z}_{\mathscr{H}}\}$, and $h(\mathbf{x}_i) = h(\phi_{\mathscr{H}}^{-1}(\mathbf{u}_i)) = 0$ (a.s.) on the event $\{\mathbf{x}_i \in \mathcal{Z}_{\mathscr{H}}^c\}$. Hence, we work with an equivalent empirical process
\begin{align*}
\widetilde{X}_n(\widetilde{h}) = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \big[\widetilde{h}(\mathbf{u}_i) - \mathbbm{E}[\widetilde{h}(\mathbf{u}_i)] \big], \qquad \widetilde{h} \in \widetilde{\mathscr{H}}.
\end{align*}
In particular, $\mathbf{u}_i$ has the uniform distribution $\mathbbm{P}_U$ which has a Lebesgue density $f_U$ that is bounded from above and below on its support $[0,1]^d$,
\begin{align*}
(X_n(h): h \in \mathscr{H}) = (\widetilde{X}_n(\widetilde{h}): \widetilde{h} \in \widetilde{\mathscr{H}}) \qquad \text{almost surely},
\end{align*}
and Assumption~\ref{sa-assump: X-process -- step} is satisfied with the random sample $(\mathbf{u}_i: 1 \leq i \leq n)$ with $\mathbf{u}_i\thicksim\mathbbm{P}_U$ and the class of functions $\widetilde{\mathscr{H}}$. We thus consider $\mathscr{A}_{K}(\mathbbm{P}_U,1)$, an axis aligned dyadic expansion of depth $K$ with respect to probability measure $\mathbbm{P}_U = \mathsf{Uniform}([0,1]^d)$. Suppose $\mathscr{E}_K$ ($\mathtt{M}_{\mathscr{H},\mathcal{Q}_{\mathscr{H}}} = \mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d}$) and $\mathtt{\Pi}_{0} = \mathtt{\Pi}_{0}[\mathscr{A}_{K}(\mathbbm{P}_U,1)]$ are defined based on $\mathscr{A}_K(\mathbbm{P}_U,1)$ as in Section~\ref{sa-sec: X-process -- Projection onto Piecewise Constant Functions}. By Lemma~\ref{sa-lem: X-process -- pregaussian} and Lemma~\ref{sa-lem: normalizing transformation}, $\widetilde{\mathscr{H}} \cup \mathtt{\Pi}_{0} \widetilde{\mathscr{H}} $ is $\mathbbm{P}_U$-pregaussian, hence by the same construction given in Section~\ref{sa-sec: X-process -- Strong Approximation Constructions}, on a possibly enlarged probability space, there exists a mean-zero Gaussian process $\widetilde{Z}_n^X$ indexed by $\widetilde{\mathscr{H}} \cup \mathtt{\Pi}_{0} \widetilde{\mathscr{H}}$ such that with almost sure continuous sample path such that
\begin{align*}
\mathbbm{E}[\widetilde{Z}_n^X(g) \widetilde{Z}_n^X(f)] = \mathbbm{E}[\widetilde{X}_n(g) \widetilde{X}_n(f)], \qquad \forall g,f \in \widetilde{\mathscr{H}} \cup \mathtt{\Pi}_{0} \widetilde{\mathscr{H}},
\end{align*}
and $U_{j,k} = \sum_{i = 1}^n e_{j,k}(\mathbf{u}_i)$ for all $(j,k) \in \mathcal{J}_K$. Let $\widetilde{\mathscr{H}}_{\delta}$ be a $\delta \mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d} = \delta \mathtt{M}_{\mathscr{H},\mathcal{Q}_{\mathscr{H}}}$-net of $\widetilde{\mathscr{H}}$ with cardinality no greater than $\mathtt{N}_{\widetilde{\mathscr{H}},[0,1]^d}(\delta,\mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d})$.
The proof proceeds by bounding each of the terms in the decomposition
\begin{gather*}
\lVert \widetilde{X}_n - \widetilde{Z}_n^X \rVert_{\widetilde{\mathscr{H}}} \leq \underbrace{\lVert \widetilde{X}_n - \widetilde{X}_n\circ\pi_{\widetilde{\mathscr{H}}_{\delta}} \rVert_{\widetilde{\mathscr{H}}} + \lVert \widetilde{Z}_n^X\circ\pi_{\widetilde{\mathscr{H}}_{\delta}}-\widetilde{Z}_n^X \rVert_{\widetilde{\mathscr{H}}}}_\text{meshing error} + \underbrace{\lVert \widetilde{X}_n - \widetilde{Z}_n^X \rVert_{\widetilde{\mathscr{H}}_{\delta}}}_\text{error on net}, \\
\lVert \widetilde{X}_n - \widetilde{Z}_n^X \rVert_{\widetilde{\mathscr{H}}_{\delta}}
\leq \underbrace{\lVert \mathtt{\Pi}_{0} \widetilde{X}_n - \mathtt{\Pi}_{0} \widetilde{Z}_n^X \rVert_{\widetilde{\mathscr{H}}_{\delta}}}_\text{approximation error} + \underbrace{\lVert \widetilde{X}_n - \mathtt{\Pi}_{0} \widetilde{X}_n \rVert_{\widetilde{\mathscr{H}}_{\delta}} + \lVert \mathtt{\Pi}_{0} \widetilde{Z}_n^X - \widetilde{Z}_n^X \rVert_{\widetilde{\mathscr{H}}_{\delta}}}_\text{projection error},
\end{gather*}
and then balancing their contributions.
Given the cells $\mathscr{A}_{K}(\mathbbm{P}_U,1)$, we have $\mathcal{U}_j \subseteq [-2^{-\frac{K - j}{d}+1}, 2^{-\frac{K - j}{d}+1}]^d$. Then, by Lemma~\ref{sa-lem: X-process -- SA error}, for all $t > 0$,
\begin{align*}
\mathbbm{P} \bigg[\lVert \widetilde{X}_n \circ \mathtt{\Pi}_{0} - \widetilde{Z}_n^X \circ \mathtt{\Pi}_{0} \rVert_{\widetilde{\mathscr{H}}_{\delta}} > 48 \sqrt{\frac{\mathscr{R}_{K}(\widetilde{\mathscr{H}}_{\delta})}{n}t} + 4\sqrt{\frac{\mathtt{C}_{\widetilde{\mathscr{H}}_{\delta},K}}{n}}t\bigg] \leq 2 \mathtt{N}_{\widetilde{\mathscr{H}},[0,1]^d}(\delta,\mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d}) e^{-t},
\end{align*}
where
\begin{align*}
\mathscr{R}_{K}(\widetilde{\mathscr{H}}_{\delta}) \leq \begin{cases}
\min\{\mathtt{TV}_{\widetilde{\mathscr{H}}_{\delta},[0,1]^d}^* \mathtt{M}_{\widetilde{\mathscr{H}}_{\delta},[0,1]^d}, \mathtt{TV}_{\widetilde{\mathscr{H}}_{\delta},[0,1]^d}^* \mathtt{L}_{\widetilde{\mathscr{H}}_{\delta},[0,1]^d}\}, & \text{if } d = 1, \\
\min \{2^K\mathtt{TV}_{\widetilde{\mathscr{H}}_{\delta},[0,1]^d}^* \mathtt{M}_{\widetilde{\mathscr{H}}_{\delta},[0,1]^d},K \mathtt{TV}_{\widetilde{\mathscr{H}}_{\delta},[0,1]^d}^* \mathtt{L}_{\widetilde{\mathscr{H}}_{\delta},[0,1]^d}\}, & \text{if } d = 2, \\
\min\{2^{K(d-1)}\mathtt{TV}_{\widetilde{\mathscr{H}}_{\delta},[0,1]^d}^* \mathtt{M}_{\widetilde{\mathscr{H}}_{\delta},[0,1]^d},2^{K(d-2)} \mathtt{TV}_{\widetilde{\mathscr{H}}_{\delta},[0,1]^d}^* \mathtt{L}_{\widetilde{\mathscr{H}}_{\delta},[0,1]^d}\}& \text{if } d \geq 3.
\end{cases}
\end{align*}
Now we calculate the $\mathtt{C}_{\widetilde{\mathscr{H}}_{\delta},K}$ term. Let $\widetilde{h} \in \widetilde{\mathscr{H}}$ and take $(\widetilde{h}_\ell)_{\ell \in \mathbb{N}}$ be any sequence of real-valued functions on $([0,1]^d,\mathcal{B}([0,1]^d))$ such that $\widetilde{h}_{\ell} \rightarrow \widetilde{h}$ $\mathfrak{m}$-almost surely, and are bounded by $2 \mathtt{M}_{\mathscr{H}}$ on $\mathcal{X}$. Moreover, by Dominated Convergence Theorem, the definition of $\mathtt{K}_{\widetilde{\mathscr{H}},[0,1]^d}^*$ and similar arguments as in the proof of Lemma~\ref{sa-lem: X-process -- SA error}, for each $(j,k) \in \mathcal{I}_K$,
\begin{align*}
\sum_{m: \mathcal{C}_{l,m} \subseteq \mathcal{C}_{j,k}} \left|\widetilde{\beta}_{j,k}(\widetilde{h})\right|
& \leq 2^{2(K - l)} \int_{\mathcal{U}_l} \int_{\mathcal{C}_{j,k}} |\widetilde{h}(\mathbf{x}) - \widetilde{h}(\mathbf{x} + \mathbf{s})|d \mathbf{x} d \mathbf{s} \\
& = \lim_{\ell \rightarrow \infty} 2^{2(K - l)} \int_{\mathcal{U}_l} \int_{\mathcal{C}_{j,k}} |\widetilde{h}_{\ell}(\mathbf{x}) - \widetilde{h}_{\ell}(\mathbf{x} + \mathbf{s})|d \mathbf{x} d \mathbf{s} \\
& \leq \limsup_{\ell \rightarrow \infty} 2^{2(K-l)} \int_{\mathcal{U}_l} \lVert \mathbf{s} \rVert \mathtt{TV}_{\{\widetilde{h}_{\ell}\},\mathcal{C}_{j,k}}d \mathbf{s}.
\end{align*}
Since the above inequality holds for all sequences $(h_{\ell})_{\ell \in \mathbb{N}}$ such that $\widetilde{h}_{\ell} \rightarrow \widetilde{h}$ $\mathfrak{m}$-almost surely, and are bounded by $2 \mathtt{M}_{\mathscr{H}}$ on $[0,1]^d$, Definitions~\ref{sa-defn: smooth tv} and \ref{sa-defn: smooth ktv} implies
\begin{align*}
\sum_{m: \mathcal{C}_{l,m} \subseteq \mathcal{C}_{j,k}} \left|\widetilde{\beta}_{j,k}(\widetilde{h})\right|
&\leq 2^{2(K - l)} \int_{\mathcal{U}_l} \lVert \mathbf{s} \rVert \mathtt{TV}_{\widetilde{\mathscr{H}},\mathcal{C}_{j,k}}^{\ast} d \mathbf{s} \\
&\leq 2^{2(K - l)} \int_{\mathcal{U}_l} \lVert \mathbf{s} \rVert \mathtt{K}_{\widetilde{\mathscr{H}},[0,1]^d}^* \lVert \mathcal{C}_{j,k} \rVert_{\infty}^{d-1}d \mathbf{s} \\
&\leq \sqrt{d} 2^{2(K - l)} \operatorname*{\mathfrak{m}}(\mathcal{U}_l) \lVert \mathcal{U}_l \rVert_{\infty} \lVert \mathcal{C}_{j,k} \rVert_{\infty}^{d-1} \mathtt{K}_{\widetilde{\mathscr{H}},[0,1]^d}^* \\
& \leq \sqrt{d} 2^{\frac{d-1}{d}(j - l)} \mathtt{K}_{\widetilde{\mathscr{H}},[0,1]^d}^*.
\end{align*}
It follows from the definition of $\mathtt{C}_{\widetilde{\mathscr{H}},K}$ in Lemma~\ref{sa-lem: X-process -- sa for pcw-const} that
\begin{align*}
\mathtt{C}_{\widetilde{\mathscr{H}},K} \leq \min\{\sqrt{K \mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d}^2}, \sqrt{\sqrt{d} \mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d} \mathtt{K}_{\widetilde{\mathscr{H}},[0,1]^d}^* + \mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d}^2}\}.
\end{align*}
For projection error, by Lemma~\ref{sa-lem: X-process -- projection error}, for all $t > 0$, with probability at least $1 - 2 \mathtt{N}_{\widetilde{\mathscr{H}},[0,1]^d}(\delta,\mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d}) e^{-t}$,
\begin{align*}
\lVert \widetilde{X}_n - \widetilde{X}_n \circ \mathtt{\Pi}_{0} \rVert_{\widetilde{\mathscr{H}}_{\delta}}
& \leq \sqrt{4d \min\{2\mathtt{M}_{\widetilde{\mathscr{H}}_\delta,[0,1]^d},\mathtt{L}_{\widetilde{\mathscr{H}}_\delta,[0,1]^d}2^{-K}\} 2^{-K} \mathtt{TV}_{\widetilde{\mathscr{H}}_\delta,[0,1]^d}^* t}
\\
& \qquad + \frac{4\min\{2\mathtt{M}_{\widetilde{\mathscr{H}}_\delta,[0,1]^d},\mathtt{L}_{\widetilde{\mathscr{H}}_\delta,[0,1]^d}2^{-K}\}}{3\sqrt{n}}t, \\
\lVert \widetilde{Z}_n^X - \widetilde{Z}_n^X \circ \mathtt{\Pi}_{0} \rVert_{\widetilde{\mathscr{H}}_{\delta}}
& \leq \sqrt{4d \min\{2\mathtt{M}_{\widetilde{\mathscr{H}}_\delta,[0,1]^d},\mathtt{L}_{\widetilde{\mathscr{H}}_\delta,[0,1]^d}2^{-K}\} 2^{-K} \mathtt{TV}_{\widetilde{\mathscr{H}}_\delta,[0,1]^d}^* t} .
\end{align*}
We balance the previous two errors by choosing $K = \lfloor d^{-1} \log_2 n\rfloor$ and get for all $t >0$, with probability at least $1 - 2 \exp(-t)$,
\begin{align*}
\lVert \widetilde{X}_n - \widetilde{Z}_n^X \rVert_{\widetilde{\mathscr{H}}_{\delta}}
\leq \widetilde{\mathsf{A}}_n(t,\delta),
\end{align*}
where
\begin{align*}
\widetilde{\mathsf{A}}_n(t,\delta) =
& \min \left\{\mathsf{m}_{n,d} \sqrt{\mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d}}, \mathtt{l}_{n,d}\sqrt{\mathtt{L}_{\widetilde{\mathscr{H}},[0,1]^d}}\right\} \sqrt{d \mathtt{TV}_{\widetilde{\mathscr{H}},[0,1]^d}^*} \sqrt{(t + \log \widetilde{\mathtt{N}}_{\widetilde{\mathscr{H}},[0,1]^d}(\delta,\mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d}))} \\
& \quad + \sqrt{\frac{\mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d}}{n}} \min\{\sqrt{\log n} \sqrt{\mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d}},\sqrt{\sqrt{d}\mathtt{K}_{\widetilde{\mathscr{H}},[0,1]^d}^*+\mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d}}\} (t + \log \widetilde{\mathtt{N}}_{\widetilde{\mathscr{H}},[0,1]^d}(\delta, \mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d})).
\end{align*}
By Lemma~\ref{sa-lem: X-Process -- Fluctuation off the net} we bound the meshing error by, for all $t > 0$,
\begin{align*}
\mathbbm{P}\big[\lVert \widetilde{X}_n - \widetilde{X}_n\circ\pi_{\widetilde{\mathscr{H}}_\delta} \rVert_{\widetilde{\mathscr{H}}} > C \mathsf{F}_n(t,\delta)\big] & \leq \exp(-t), \\
\mathbbm{P}\big[\lVert \widetilde{Z}_n^X\circ\pi_{\widetilde{\mathscr{H}}_\delta}-\widetilde{Z}_n^X \rVert_{\widetilde{\mathscr{H}}} > C (\mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d}J(\delta,\widetilde{\mathscr{H}}, \mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d}) + \delta \mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d} \sqrt{t})\big] & \leq \exp(-t),
\end{align*}
where \[\mathsf{F}_n(t,\delta) = J(\delta, \widetilde{\mathscr{H}}, \mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d}) \mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d} + \frac{\mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d} J^2(\delta, \widetilde{\mathscr{H}}, \mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d})}{\delta^2 \sqrt{n}} + \delta \mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d} \sqrt{t} + \frac{\mathtt{M}_{\widetilde{\mathscr{H}},[0,1]^d}}{\sqrt{n}} t.\]
Take the Gaussian process $(Z_n(h): h \in \mathscr{H})$ such that, almost surely, $Z_n(h) = \widetilde{Z}_n(\widetilde{h})$ for all $h \in \mathscr{H}$. The result then follows from the decomposition that
\begin{equation*}
\begin{split}
\lVert X_n - Z_n^X \rVert_{\mathscr{H}} = \lVert \widetilde{X}_n - \widetilde{Z}_n^X \rVert_{\widetilde{\mathscr{H}}}
\leq \lVert \widetilde{X}_n - \widetilde{X}_n \circ \pi_{\widetilde{\mathscr{H}}_{\delta}} \rVert_{\widetilde{\mathscr{H}}} + \lVert \widetilde{Z}_n^X - \widetilde{Z}_n^X \circ \pi_{\widetilde{\mathscr{H}}_{\delta}} \rVert_{\widetilde{\mathscr{H}}} +
\lVert \widetilde{X}_n - \widetilde{Z}_n^X \rVert_{\widetilde{\mathscr{H}}_{\delta}},
\end{split}
\end{equation*}
and Lemma~\ref{sa-lem: normalizing transformation} to establish the relationships between the parameters of $\mathscr{H}$ over $\mathcal{Q}_{\mathscr{H}}$ and those of $\widetilde{\mathscr{H}}$ over $[0,1]^d$.
\qed
\subsection{Additional Results}\label{sa-sec: X-process -- Additional Results}
In what follows, we drop the dependence on $\mathcal{C} = \mathcal{Q}_{\mathscr{F}}$ for all quantities in Definitions 4-12. That is, to save notation, we set $\mathtt{TV}_{\mathscr{F}}=\mathtt{TV}_{\mathscr{F},\mathcal{Q}_{\mathscr{F}}}$, $\mathtt{K}_{\mathscr{F}}=\mathtt{K}_{\mathscr{F},\mathcal{Q}_{\mathscr{F}}}$, $\mathtt{M}_{\mathscr{F},\mathcal{X}}=\mathtt{M}_{\mathscr{F},\mathcal{Q}_{\mathscr{F}}}$, $M_{\mathscr{F},\mathcal{X}}(\mathbf{u})=M_{\mathscr{F},\mathcal{Q}_{\mathscr{F}}}(\mathbf{u})$, $\mathtt{L}_{\mathscr{F}}=\mathtt{L}_{\mathscr{F},\mathcal{Q}_{\mathscr{F}}}$, and so on, whenever there is no confusion.
\begin{coro}[VC-Type Bounded Functions]\label{sa-coro: X-process -- vc bdd}
Suppose the conditions of Corollary 1 hold. Then,
\begin{equation*}
\mathsf{S}_n(t) = \mathsf{m}_{n,d}\sqrt{\mathtt{c}_1 \mathtt{M}_{\mathscr{H}}\mathtt{TV}_{\mathscr{H}}} \sqrt{t + \mathtt{d}_{\mathscr{H}}\log(\mathtt{c}_{\mathscr{H}}n)} + \sqrt{\frac{\mathtt{M}_{\mathscr{H}}}{n}}\min\{\sqrt{\log n} \sqrt{\mathtt{M}_{\mathscr{H}}},\sqrt{\mathtt{c}_3 \mathtt{K}_{\mathscr{H}} + \mathtt{M}_{\mathscr{H}}}\}(t + \mathtt{d}_{\mathscr{H}}\log(\mathtt{c}_{\mathscr{H}}n))
\end{equation*}
in Theorem 1.
\end{coro}
\begin{proof}[Proof of Corollary~\ref{sa-coro: X-process -- vc bdd}]
Take $\delta = n^{-1/2}$. Under the VC-type class condition, $\log \mathtt{N}_{\mathscr{H}}(n^{-1},\mathtt{M}_{\mathscr{H}}) \leq \log(\mathtt{c}_{\mathscr{H}}) + \mathtt{d}_{\mathscr{H}}\log(n) \leq \mathtt{d}_{\mathscr{H}}\log(\mathtt{c}_{\mathscr{H}}n)$, where the last inequality holds since $\mathtt{c}_{\mathscr{H}} \geq e$ and $\mathtt{d}_{\mathscr{H}} > 0$. This gives
\begin{align*}
\mathsf{A}_n(t,n^{-1/2}) & \leq \mathsf{m}_{n,d}\sqrt{ \mathtt{c}_1 (t + \mathtt{d}_{\mathscr{H}}\log(\mathtt{c}_{\mathscr{H}}n))\mathtt{M}_{\mathscr{H}} \mathtt{TV}_{\mathscr{H}}} \\
& \qquad + \min\big\{ \sqrt{\log (n) \mathtt{M}_{\mathscr{H}}} , \sqrt{\mathtt{c}_3 \mathtt{K}_{\mathscr{H}}+\mathtt{M}_{\mathscr{H}}}\big\} \sqrt{\frac{\mathtt{M}_{\mathscr{H}}}{n}} (t + \mathtt{d}_{\mathscr{H}}\log(\mathtt{c}_{\mathscr{H}}n)).
\end{align*}
Moreover, $J(\delta,\mathscr{H}, \mathtt{M}_{\mathscr{H}}) \leq \int_{0}^{\delta}\sqrt{1 + \mathtt{d}_{\mathscr{H}}\log(\mathtt{c}_{\mathscr{H}}\varepsilon^{-1})}d \varepsilon \leq 3 \delta \sqrt{\mathtt{d}_{\mathscr{H}}\log(\mathtt{c}_{\mathscr{H}}/\delta)}$. It follows that
\begin{align*}
\mathsf{F}_{n}(t,n^{-1/2}) \leq \frac{3 \mathtt{M}_{\mathscr{H}}}{\sqrt{n}}\mathtt{d}_{\mathscr{H}}\log(\mathtt{c}_{\mathscr{H}}n) + \frac{\mathtt{M}_{\mathscr{H}}}{\sqrt{n}}(\sqrt{t} + t).
\end{align*}
The result then follows from Theorem 1.
\end{proof}
\begin{coro}[VC-Type Lipschitz Functions]\label{sa-coro: X-process -- vc lip}
Suppose the conditions of Corollary 2 hold. Then,
\begin{align*}
\mathsf{S}_n(t)
& = \min\big\{\mathsf{m}_{n,d} \sqrt{\mathtt{M}_{\mathscr{H}}},
\mathsf{l}_{n,d} \sqrt{\mathtt{c}_2 \mathtt{L}_{\mathscr{H}}}\big\} \sqrt{\mathtt{TV}_{\mathscr{H}}} \sqrt{t + \mathtt{d}_{\mathscr{H}}\log(\mathtt{c}_{\mathscr{H}}n)} \\
& \phantom{=} + \sqrt{\frac{\mathtt{M}_{\mathscr{H}}}{n}} \min\{\sqrt{\log n} \sqrt{\mathtt{M}_{\mathscr{H}}},\sqrt{\mathtt{c}_3 \mathtt{K}_{\mathscr{H}} + \mathtt{M}_{\mathscr{H}}}\} (t + \mathtt{d}_{\mathscr{H}}\log(\mathtt{c}_{\mathscr{H}}n))
\end{align*}
in Theorem 1.
\end{coro}
\begin{proof}[Proof of Corollary~\ref{sa-coro: X-process -- vc lip}]
The result follows by taking $\delta = n^{-1/2}$ and apply Theorem 1, with calculations similar to Corollary~\ref{sa-coro: X-process -- vc bdd}.
\end{proof}
\begin{coro}[Polynomial-Entropy Functions]\label{sa-coro: X-process -- poly entropy gen}
Suppose the conditions of Corollary 2 hold. Then,
\begin{align*}
\mathsf{S}_n(t) = \mathtt{a}_{\mathscr{H}}(2 - \mathtt{b}_{\mathscr{H}})^{-2}\min\{\mathsf{S}_n^{bdd}(t), \mathsf{S}_n^{lip}(t), \mathsf{S}_n^{err}(t)\}
\end{align*}
in Theorem 1, where
\begin{align*}
\mathsf{S}_n^{bdd}(t) & = \mathsf{m}_{n,d}\sqrt{ \mathtt{c}_1\mathtt{M}_{\mathscr{H}}\mathtt{TV}_{\mathscr{H}}}(\sqrt{t}+(\mathsf{m}_{n,d}^2\mathtt{M}_{\mathscr{H}}^{-1}\mathtt{TV}_{\mathscr{H}})^{-\frac{\mathtt{b}_{\mathscr{H}}}{4}}) \\
& \phantom {= } + \sqrt{\frac{\mathtt{M}_{\mathscr{H}}}{n}}\min\{\sqrt{\log n}\sqrt{\mathtt{M}_{\mathscr{H}}}, \sqrt{\mathtt{c}_3 \mathtt{K}_{\mathscr{H}} + \mathtt{M}_{\mathscr{H}}}\}(t +(\mathsf{m}_{n,d}^2 \mathtt{M}_{\mathscr{H}}^{-1} \mathtt{TV}_{\mathscr{H}})^{-\frac{\mathtt{b}_{\mathscr{H}}}{2}}), \\
\mathsf{S}_n^{lip}(t) & = \mathsf{l}_{n,d}\sqrt{\mathtt{c}_1 \mathtt{c}_2\mathtt{L}_{\mathscr{H}}\mathtt{TV}_{\mathscr{H}}}(\sqrt{t}+(\mathsf{l}_{n,d}^2\mathtt{M}_{\mathscr{H}}^{-2}\mathtt{L}_{\mathscr{H}}\mathtt{TV}_{\mathscr{H}})^{-\frac{\mathtt{b}_{\mathscr{H}}}{4}}) \\
& \phantom {= } + \sqrt{\frac{\mathtt{M}_{\mathscr{H}}}{n}}\min\{\sqrt{\log n}\sqrt{\mathtt{M}_{\mathscr{H}}}, \sqrt{\mathtt{c}_3 \mathtt{K}_{\mathscr{H}} + \mathtt{M}_{\mathscr{H}}}\}(t +(\mathsf{l}_{n,d}^2 \mathtt{M}_{\mathscr{H}}^{-2} \mathtt{L}_{\mathscr{H}}\mathtt{TV}_{\mathscr{H}})^{-\frac{\mathtt{b}_{\mathscr{H}}}{2}}), \\
\mathsf{S}_n^{err}(t) & = \min\{\mathsf{m}_{n,d} \sqrt{\mathtt{M}_{\mathscr{H}}}, \mathsf{l}_{n,d} \sqrt{\mathtt{c}_2 \mathtt{L}_{\mathscr{H}}}\}\sqrt{ \mathtt{c}_1 \mathtt{TV}_{\mathscr{H}}}(\sqrt{t}+n^{\frac{\mathtt{b}_{\mathscr{H}}}{2(\mathtt{b}_{\mathscr{H}}+2)}})\\
& \phantom{ = } + \sqrt{\frac{\mathtt{M}_{\mathscr{H}}}{n}} \min\{\sqrt{\log n}\sqrt{\mathtt{M}_{\mathscr{H}}}, \sqrt{\mathtt{c}_3 \mathtt{K}_{\mathscr{H}} + \mathtt{M}_{\mathscr{H}}}\}(t + n^{\frac{\mathtt{b}_{\mathscr{H}}}{\mathtt{b}_{\mathscr{H}}+2}}) + n^{-\frac{1}{b+2}}\mathtt{M}_{\mathscr{H}}\sqrt{t}.
\end{align*}
\end{coro}
\begin{proof}[Proof of Corollary~\ref{sa-coro: X-process -- poly entropy gen}]
Under the polynomial entropy condition, $\log \mathtt{N}_{\mathscr{H}}(\delta) \leq \mathtt{a}_{\mathscr{H}}\delta^{-\mathtt{b}_{\mathscr{H}}}$, $J(\delta, \mathscr{H},\mathtt{M}_{\mathscr{H}}) \leq \sqrt{\mathtt{a}_{\mathscr{H}}}(2 - \mathtt{b}_{\mathscr{H}})^{-1}\delta^{-\mathtt{b}_{\mathscr{H}}/2+1}$,
\begin{align*}
\mathsf{A}_n(t,\delta) & \leq \min\{\mathsf{m}_{n,d}\sqrt{\mathtt{M}_{\mathscr{H}}}, \mathsf{l}_{n,d}\sqrt{\mathtt{c}_2 \mathtt{L}_{\mathscr{H}}}\} \sqrt{\mathtt{TV}_{\mathscr{H}}(t + \mathtt{a}_{\mathscr{H}}\delta^{-\mathtt{b}_{\mathscr{H}}})} \\
& \qquad + \sqrt{\frac{\mathtt{M}_{\mathscr{H}}}{n}}\min\{\sqrt{\log n}\sqrt{\mathtt{M}_{\mathscr{H}}}, \sqrt{\mathtt{c}_3 \mathtt{K}_{\mathscr{H}} + \mathtt{M}_{\mathscr{H}}} \}(t + \mathtt{a}_{\mathscr{H}}\delta^{-\mathtt{b}_{\mathscr{H}}}), \\
\mathsf{F}_n(t,\delta) & \leq \mathtt{a}_{\mathscr{H}}(2 - \mathtt{b}_{\mathscr{H}})^{-2} \bigg(\mathtt{M}_{\mathscr{H}} \delta^{-\mathtt{b}_{\mathscr{H}}/2+1} + \frac{\mathtt{M}_{\mathscr{H}}}{\sqrt{n}}\delta^{-\mathtt{b}_{\mathscr{H}}} + \delta \mathtt{M}_{\mathscr{H}}\sqrt{t} + \frac{\mathtt{M}_{\mathscr{H}}}{\sqrt{n}}t\bigg).
\end{align*}
Notice that the two terms $\frac{\mathtt{M}_{\mathscr{H}}}{\sqrt{n}}\delta^{-\mathtt{b}_{\mathscr{H}}}$ and $\frac{\mathtt{M}_{\mathscr{H}}}{\sqrt{n}}t$ in $\mathsf{F}_n(t,\delta)$ are dominated by terms in $\mathsf{A}_n(t,\delta)$. And when $\delta \leq n^{-1/2}$, the third term $\delta \mathtt{M}_{\mathscr{H}}\sqrt{t}$ is also dominated by terms in $\mathsf{A}_n(t,\delta)$. To choose $\delta$ that balance $\mathsf{A}_n$ and $\mathsf{F}_n$, we consider the following three cases:
\textbf{Case 1:} Choosing $\delta$ such that $\mathsf{m}_{n,d}\sqrt{\mathtt{M}_{\mathscr{H}} \mathtt{TV}_{\mathscr{H}} \delta^{-\mathtt{b}_{\mathscr{H}}}} = \mathtt{M}_{\mathscr{H}}\delta^{-\mathtt{b}_{\mathscr{H}}/2+1}$, gives $\delta_{\ast} = \mathsf{m}_{n,d}\sqrt{\mathtt{TV}_{\mathscr{H}}/\mathtt{M}_{\mathscr{H}}}$. Notice that this choice also makes $\delta \mathtt{M}_{\mathscr{H}}\sqrt{t} \leq \sqrt{\frac{\mathtt{M}_{\mathscr{H}}}{n}}\min\{\sqrt{\log n}\sqrt{\mathtt{M}_{\mathscr{H}}}, \sqrt{\mathtt{c}_3 \mathtt{K}_{\mathscr{H}} + \mathtt{M}_{\mathscr{H}}} \}(t + \mathtt{a}_{\mathscr{H}}\delta^{-\mathtt{b}_{\mathscr{H}}})$. Thus, we get $\mathsf{A}_n(t,\delta_{\ast}) + \mathsf{F}_n(t,\delta_{\ast}) \leq \mathsf{S}_{n}^{bdd}(t)$.
\textbf{Case 2:} Choosing $\delta$ such that $\mathsf{l}_{n,d}\sqrt{\mathtt{L}_{\mathscr{H}}\mathtt{TV}_{\mathscr{H}}\delta^{-\mathtt{b}_{\mathscr{H}}}} = \mathtt{M}_{\mathscr{H}}\delta^{-\mathtt{b}_{\mathscr{H}}/2+1}$, gives $\delta_{\ast} = \mathsf{l}_{n,d}\sqrt{\mathtt{L}_{\mathscr{H}}\mathtt{TV}_{\mathscr{H}}/\mathtt{M}_{\mathscr{H}}^2}$. Again, this choice of $\delta$ makes $\delta \mathtt{M}_{\mathscr{H}}\sqrt{t} \leq \sqrt{\frac{\mathtt{M}_{\mathscr{H}}}{n}}\min\{\sqrt{\log n}\sqrt{\mathtt{M}_{\mathscr{H}}}, \sqrt{\mathtt{c}_3 \mathtt{K}_{\mathscr{H}} + \mathtt{M}_{\mathscr{H}}} \}(t + \mathtt{a}_{\mathscr{H}}\delta^{-\mathtt{b}_{\mathscr{H}}})$. Thus, we get $\mathsf{A}_n(t,\delta_{\ast}) + \mathsf{F}_n(t,\delta_{\ast}) \leq \mathsf{S}_{n}^{lip}(t)$.
\textbf{Case 3:} Choosing $\delta$ such that $\mathtt{M}_{\mathscr{H}} n^{-1/2} \delta^{-\mathtt{b}_{\mathscr{H}}} = \mathtt{M}_{\mathscr{H}} \delta^{-\mathtt{b}_{\mathscr{H}}/2+1}$, gives $\delta_* = n^{-1/(\mathtt{b}_{\mathscr{H}}+2)}$. Thus, we get $\mathsf{A}_n(t,\delta_{\ast}) + \mathsf{F}_n(t,\delta_{\ast}) \leq \mathsf{S}_n^{err}(t)$.
\end{proof}
\subsection{Proofs of Corollaries 1, 2, and 3}\label{sa-sec: X-process -- Proofs of Corollaries}
\begin{proof}[Proof of Corollary 1]
Take $t = C\log n$ with $C>1$ in Corollary~\ref{sa-coro: X-process -- vc bdd}.
\end{proof}
\begin{proof}[Proof of Corollary 2]
Take $t = C\log n$ with $C>1$ in Corollary~\ref{sa-coro: X-process -- vc lip}.
\end{proof}
\begin{proof}[Proof of Corollary 3]
Take $t = C\log n$ with $C>1$ in Corollary~\ref{sa-coro: X-process -- poly entropy gen}.
\end{proof}
\subsection{Example 1: Kernel Density Estimation}\label{sa-sec: X-process -- KDE Example}
To simplify notation, in this section the parameters of $\mathscr{H}$ (Definitions 4 to 12) are taken with $\mathcal{C} = \mathcal{Q}_\mathscr{H}$, and the index $\mathcal{C}$ is omitted whenever there is no confusion.
\subsubsection{Surrogate Measure and Normalizing Transformation}
We show that the two sets of primitive conditions discussed in the paper imply condition (ii) in Theorem 1.
First, consider the case $\mathcal{X} = \times_{l = 1}^d[\mathsf{a}_l, \mathsf{b}_l]$, $- \infty \leq \mathsf{a}_l < \mathsf{b}_l \leq \infty$ and $\mathcal{W}$ is arbitrary. Observe that $\mathbb{Q}_{\mathscr{H}} = \mathbbm{P}_X$ is always a valid surrogate measure for $\mathbbm{P}_X$ with respect to $\mathscr{H}$, according to Definition 2. The conclusion then follows from Case 1 from Section 3.1 with $f_Q = f_X$.
Second, consider the case when $\mathcal{X}$ may be unbounded. We present a general construction, and then specialize it to the example discussed in the paper. Suppose we can find $\mathcal{Q}_\mathscr{H}$ diffeomorphic to $[0,1]^d$ such that $\mathcal{X} \cap \operatorname{Supp}(\mathscr{H}) \subseteq \mathcal{Q}_\mathscr{H} \subseteq \mathcal{X} \cup \operatorname{Supp}(\mathscr{H})^c$, with $\mathbbm{P}_X(\mathcal{X} \cap \operatorname{Supp}(\mathscr{H})) < 1$ and $\operatorname*{\mathfrak{m}}(\mathcal{Q}_\mathscr{H} \setminus (\mathcal{X} \cap \operatorname{Supp}(\mathscr{H}))) > 0$. Setting $\mathbb{Q}_\mathscr{H}$ to be the probability measure with Lebesgue density $f_Q$ such that
\begin{align*}
f_Q(\mathbf{x}) =
\begin{cases}
f_X(\mathbf{x}), & \text{ if } \mathbf{x} \in \mathcal{X} \cap \operatorname{Supp}(\mathscr{H}), \\
(1 - \mathbbm{P}_X(\mathcal{X} \cap \operatorname{Supp}(\mathscr{H})))/\operatorname*{\mathfrak{m}}(\mathcal{Q}_\mathscr{H} \setminus (\mathcal{X} \cap \operatorname{Supp}(\mathscr{H}))), & \text{ if } \mathbf{x} \in \mathcal{Q}_\mathscr{H} \setminus (\mathcal{X} \cap \operatorname{Supp}(\mathscr{H})), \\
0, & \text{ otherwise.}
\end{cases}
\end{align*}
then $\mathbb{Q}_\mathscr{H}$ is a surrogate measure of $\mathbbm{P}_X$ with respect to $\mathscr{H}$. Suppose $\chi$ is a diffeomorphism from $\mathcal{Q}_\mathscr{H}$ to $[0,1]^d$. Since we assumed $\mathcal{X} \cap \operatorname{Supp}(\mathscr{H}) \subseteq \mathcal{Q}_\mathscr{H} \subseteq \mathcal{X} \cup \operatorname{Supp}(\mathscr{H})^c$, with $\mathbbm{P}_X(\mathcal{X} \cap \operatorname{Supp}(\mathscr{H})) < 1$ and $\operatorname*{\mathfrak{m}}(\mathcal{Q}_\mathscr{H} \setminus (\mathcal{X} \cap \operatorname{Supp}(\mathscr{H}))) > 0$, we can check that (1) $f_Q$ is supported and positive on $\mathcal{Q}_\mathscr{H}$, (2) $f_Q$ agrees with $f_X$ on $\mathcal{X} \cap \operatorname{Supp}(\mathscr{H})$. Then Case 2 in Section 3.1 implies $\phi_{\mathscr{H}} = T_{\mathbb{Q}_\mathscr{H} \circ \chi^{-1}} \circ \chi$ is a valid normalizing transformation, and condition (ii) in Theorem 1 holds. Suppose $0 < \inf_{\mathbf{x} \in \mathcal{X} \cap \operatorname{Supp}(\mathscr{H})} f_X(\mathbf{x})< \sup_{\mathbf{x} \in \mathcal{X} \cap \operatorname{Supp}(\mathscr{H})} f_X(\mathbf{x})< \infty$ and $\frac{\sup_{\mathbf{x} \in [0,1]^d} |\operatorname{det}(\nabla\chi^{-1}(\mathbf{x}))|}{\inf_{\mathbf{x} \in [0,1]^d} |\operatorname{det}(\nabla\chi^{-1}(\mathbf{x}))|}\lVert \lVert \nabla \chi^{-1} \rVert_2 \rVert_{\infty} < \infty$, then we have $\mathtt{c}_1 = O(1)$ and $\mathtt{c}_2 = O(1)$ (and hence $\mathtt{c}_3 = O(1)$).
For a concrete example, consider the case $\mathcal{X} = \mathbb{R}^d_+$, $\mathcal{W} = \times_{l = 1}^d [\mathsf{a}_l,\mathsf{b}_l]$, $0 \leq \mathsf{a}_l < \mathsf{b}_l < \infty$, and $\mathcal{K} = [-1,1]^d$. Observe that $\operatorname{Supp}(\mathscr{H}) \cap \mathcal{X} = \times_{l = 1}^d [(\mathsf{a}_l - b)_+, \mathsf{b}_l + b] = \times_{l=1}^d [\overline{\mathsf{a}}_l, \overline{\mathsf{b}}_l]$. Since $\mathcal{X} = \mathbb{R}^d_+$, $\mathbbm{P}_X(\times_{l=1}^d [\overline{\mathsf{a}}_l, \overline{\mathsf{b}}_l]) < 1$. Moreover, we can check that $\mathcal{X} \cap \operatorname{Supp}(\mathscr{H}) \subseteq \mathcal{Q}_\mathscr{H} \subseteq \mathcal{X} \cup \operatorname{Supp}(\mathscr{H})^c$ and $\mathbb{Q}_{\mathscr{H}}$ agrees with $\mathbbm{P}_X$ on $\mathcal{X} \cap \operatorname{Supp}(\mathscr{H})$. The rest then follows from the general construction above.
\subsubsection{Class \texorpdfstring{$\mathscr{H}$}{H} and Its Corresponding Constants}
Let $\mathscr{H} = \{h_{\mathbf{w}}: \mathbf{w} \in \mathcal{W} \}$ with $h_{\mathbf{w}}(\cdot) = b^{-d/2}K(b^{-1}(\mathbf{w} - \cdot))$. Since $K$ is compactly supported and Lipschitz, $\mathtt{M}_{\{K\}} < \infty$. Hence, $\mathtt{M}_{\mathscr{H}} = b^{-d/2} \mathtt{M}_{\{K\}} \leq C_K b^{-d/2}$ and $\mathtt{L}_{\mathscr{H}} \leq b^{-\frac{d}{2}-1}\mathtt{L}_{\{K\}} \leq C_K b^{-d/2-1}$, where $C_K$ is a constant that only depends on the kernel function $K$. Since $\sup_{\mathbf{w} \in \mathcal{W}} \operatorname*{\mathfrak{m}}(\operatorname{Supp}(h_{\mathbf{w}})) \leq C_K b^d$ and each $h_{\mathbf{w}}$ is differentiable,
\begin{align*}
\mathtt{TV}_{\mathscr{H}} = \sup_{\mathbf{w} \in \mathcal{W}} \int \lVert \nabla h_{\mathbf{w}}(\mathbf{u}) \rVertd \mathbf{u} \leq \sup_{\mathbf{w} \in \mathcal{W}} \operatorname*{\mathfrak{m}}(\operatorname{Supp}(h_{\mathbf{w}})) \mathtt{L}_{\mathscr{H}} \leq C_K b^{d/2-1}.
\end{align*}
To upper bound $\mathtt{K}_{\mathscr{H}}$, let $\mathcal{D} \subseteq \mathcal{Q}_\mathscr{H}$ be a cube with edges of length $\mathtt{a}$ parallel to the coordinate axises. Consider the following two cases: (i) if $\mathtt{a} < b$, then $\mathtt{TV}_{\mathscr{H},\mathcal{D}} \leq C_K b^{-d/2-1}\mathtt{a}^d \leq C_K b^{-d/2} \mathtt{a}^{d-1}$; (ii) if $\mathtt{a} > b$, then $\mathtt{TV}_{\mathscr{H},\mathcal{D}} \leq C_K \sup_{\mathbf{w} \in \mathcal{W}} \operatorname*{\mathfrak{m}}(\operatorname{Supp}(h_{\mathbf{w}})) \mathtt{L}_{\mathscr{H}}
\leq C_K b^{d} b^{-d/2-1}
\leq C_K b^{-d/2} b^{d-1}
\leq C_K b^{-d/2} \mathtt{a}^{d-1}$.
This shows $$\mathtt{K}_{\mathscr{H}} \leq C_K b^{-d/2}.$$ Next, by a change of variables,
\begin{align*}
\mathtt{E}_{\mathscr{H}} &
= \sup_{\mathbf{w} \in \mathcal{W}} \int b^{-\frac{d}{2}} |K(b^{-1}(\mathbf{w} - \mathbf{u}))| f_X(\mathbf{u}) d \mathbf{u}
= \sup_{\mathbf{w} \in \mathcal{W}} \int b^{-\frac{d}{2}}|K(\mathbf{z})|f_{X}(\mathbf{w} - h \mathbf{z}) b^d d \mathbf{z}
\leq C_K b^{d/2}.
\end{align*}
Finally, we check that $\mathscr{H}$ is a VC-type class. We will apply Lemma 7 from \cite{Cattaneo-Chandak-Jansson-Ma_2024_Bernoulli-SA} on the class $\mathtt{M}_{\mathscr{H}}^{-1} \mathscr{H}$. To check the conditions in this lemma, define $g_{\mathbf{w}}(\cdot) = b^{-\frac{d}{2}}\mathtt{M}_{\mathscr{H}}^{-1}K(\cdot)$ for all $\mathbf{w} \in \mathcal{W}$. Note that $g_{\mathbf{w}}$ is the same function for all $\mathbf{w} \in \mathcal{W}$ in this setting (but, more generally, our results allow for functions varying with the evaluation point such as in the case of boundary adaptive kernels). Then $\mathtt{M}_{\mathscr{H}}^{-1}\mathscr{H} = \{g_{\mathbf{w}}(\frac{\mathbf{w} - \cdot}{b}): \mathbf{w} \in \mathcal{W}\}$, and there exists a constant $c_K$, only depending on $\mathtt{M}_{\{K\}}$ and $\mathtt{L}_{\{K\}}$, such that
\begin{align*}
\sup_{\mathbf{w} \in \mathcal{W}} \lVert g_{\mathbf{w}} \rVert_{\infty} \leq c_K,
\qquad
\sup_{\mathbf{w} \in \mathcal{W}} \sup_{\mathbf{u}, \mathbf{v} \in \mathcal{Q}_\mathscr{H}} \frac{|g_{\mathbf{w}}(\mathbf{u}) - g_{\mathbf{w}}(\mathbf{v})|}{\lVert \mathbf{u} - \mathbf{v} \rVert_{\infty}} \leq c_K,
\qquad
\sup_{\mathbf{w},\mathbf{w}^{\prime} \in \mathcal{W}} \sup_{\mathbf{u} \in \mathcal{Q}_\mathscr{H}} \frac{|g_{\mathbf{w}}(\mathbf{u}) - g_{\mathbf{w}^{\prime}}(\mathbf{u})|}{\lVert \mathbf{w} - \mathbf{w}^{\prime} \rVert_{\infty}} \leq c_K.
\end{align*}
We can apply Lemma 7 from \cite{Cattaneo-Chandak-Jansson-Ma_2024_Bernoulli-SA}, which is modified upon Lemma 4.1 from \cite{Rio_1994_PTRF-SA}, to show that for all $0 < \varepsilon < 1$, $\mathtt{N}_{\mathtt{M}_{\mathscr{H}}^{-1} \mathscr{H}}(\varepsilon, 1) \leq c_K \varepsilon^{-d-1} + 1,$
and hence
\begin{align*}
\mathtt{N}_{\mathscr{H}}(\varepsilon, \mathtt{M}_{\mathscr{H}}) \leq c_K \varepsilon^{-2d-2} + 1,
\end{align*}
The conclusions on uniform Gaussian strong approximation rates then follow from Corollaries 1--3.
\section{Multiplicative-Separable Empirical Process}\label{sa-sec: Multiplicative Empirical Process}
Let $\mathbf{z}_i=(\mathbf{x}_i, y_i)\in \mathcal{X}\times\mathcal{Y} \subseteq \mathbb{R}^d\times\mathbb{R}$, $i=1,\dots,n$, be i.i.d. random vectors supported on a background probability space $(\Omega,\mathcal{F},\mathbbm{P})$. The multiplicative-separable empirical process is
\begin{align*}
G_n(g,r) = \frac{1}{\sqrt{n}} \sum_{i=1}^n \big( g(\mathbf{x}_i)r(y_i) - \mathbbm{E}[g(\mathbf{x}_i)r(y_i)] \big), \qquad g \in \mathscr{G}, r \in \mathscr{R},
\end{align*}
where $\mathscr{G}$ and $\mathscr{R}$ are possibly $n$-varying classes of functions. Notably, if we take $\mathscr{H} = \mathscr{G} \cdot \mathscr{R} = \{g \cdot r: g \in \mathscr{G}, r \in \mathscr{R}\}$, then the above process can also be written as a generic empirical process based on $(\mathbf{z}_i: 1 \leq i \leq n)$ because
\begin{align*}
X_n(h) = X_n(g \cdot r) = \frac{1}{\sqrt{n}} \sum_{i = 1}^n ((g \cdot r)(\mathbf{z}_i) - \mathbbm{E}[(g \cdot r)(\mathbf{z}_i)]),
\qquad h= g \cdot r \in \mathscr{H}=\mathscr{G}\cdot\mathscr{R}.
\end{align*}
Hence, the same decomposition for the $X_n$ process also applies for the $G_n$ process:
\begin{align*}
\lVert G_n - Z_n^G \rVert_{\mathscr{G} \times \mathscr{R}}
& \leq \lVert G_n - Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_\delta} + \lVert G_n - G_n\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta} \rVert_{\mathscr{G} \times \mathscr{R}}
+ \lVert Z_n^G\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta}-Z_n^G \rVert_{\mathscr{G} \times \mathscr{R}}\\
& \leq \lVert \mathtt{\Pi}_{1} Z_n^G - Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}
+ \lVert G_n - \mathtt{\Pi}_{1} G_n \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
+ \lVert \mathtt{\Pi}_{1} G_n - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}\\
& \qquad + \lVert G_n - G_n\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta} \rVert_{\mathscr{G} \times \mathscr{R}}
+ \lVert Z_n^G\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta}-Z_n^G \rVert_{\mathscr{G} \times \mathscr{R}},
\end{align*}
where $(\mathscr{G} \times \mathscr{R})_\delta$ denotes a discretization (or meshing) of $\mathscr{G} \times \mathscr{R}$ (i.e., $\delta$-net of $\mathscr{G} \times \mathscr{R}$), and the terms $\lVert G_n - G_n\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta} \rVert_{\mathscr{G} \times \mathscr{R}}$ and $\lVert Z_n^G\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta}-Z_n^G \rVert_{\mathscr{G} \times \mathscr{R}}$ capture the fluctuations (or oscillations) of $G_n$ and $Z_n^G$ relative to the meshing for each of the stochastic processes. $\lVert \mathtt{\Pi}_{1} G_n - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$ and $\lVert \mathtt{\Pi}_{1} Z_n^G - Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$ represent projections onto a Haar function space, where $\mathtt{\Pi}_{1} G_n(h) = G_n \circ \mathtt{\Pi}_{1} h$. The operator $\mathtt{\Pi}_{1}$ is a projection onto piecewise constant functions that respects the multiplicative structure of the $G_n$ process. The final term $\lVert \mathtt{\Pi}_{1} G_n - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$ captures the coupling between the empirical process and the Gaussian process (on a $\delta$-net of $\mathscr{G} \times \mathscr{R}$, after the projection $\mathtt{\Pi}_{1}$).
A general result under uniform entropy integral conditions is presented in Section~\ref{sa-sec: MULT -- general result}, and a corollary under a VC-type condition is presented in Section~\ref{sa-sec: MULT -- additional result}. The proofs exploit the existence of a surrogate measure and normalizing transformation of $\mathscr{G}$ with respect to $\mathbb{P}_X$, the law of $\mathbf{x}_1$, as developed in Section~\ref{sa-sec: X-Process -- normalizing transformation}. The preliminary technical results differ from those in Section~\ref{sa-sec: General Empirical Process} by explicitly leveraging the multiplicative structure of the empirical process, and are organized as follows.
\begin{itemize}[leftmargin=*]
\item Section \ref{sa-sec: MULT -- cell expansions} introduces the class of \emph{cylindered quasi-dyadic cell expansions} based on $\mathbbm{P}_Z$, which can be viewed as a special case of the \emph{quasi-dyadic cell expansions} from Definition~\ref{sa-defn: quasi dyadic expansion} that leverages the multiplicative structure. This cell expansion is tailored to the multiplicative structure, with the upper layers corresponding to splits in the $\mathbf{x}_i$-direction and the lower layers handling divisions along the $y_i$-direction.
\item Section \ref{sa-sec: MULT -- proj} introduces an alternative to the $L_2$ projection onto piecewise constant functions on the chosen cells: the \emph{product-factorized projection}, $\mathtt{\Pi}_{1}$. This projection exploits the multiplicative structure of $G_n$, allowing the empirical process to treat $\mathbf{x}_i$ and $y_i$ as independent in layers where cells divide along $\mathcal{Y}$, thereby isolating contributions from $\mathscr{G}$ and $\mathscr{R}$. To analyze the projection errors $\lVert G_n - \mathtt{\Pi}_{1} G_n \rVert_{(\mathscr{G} \times \mathscr{R}){\delta}}$ and $\lVert Z_n^G - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})\delta}$, we also define the $L_2$ projection onto piecewise constant functions on the chosen cells, $\mathtt{\Pi}_{0}$.
\item Section~\ref{sa-sec: MULT -- sa construction} constructs the Gaussian process $(Z_n^G(g, r): (g, r) \in \mathscr{G} \times \mathscr{R})$. These constructions are essentially the same as those in Section~\ref{sa-sec: X-process -- Strong Approximation Constructions}, relying on coupling binomial random variables with Gaussian random variables.
\item Section~\ref{sa-sec: MULT -- meshing error} handles the meshing errors $\lVert G_n - G_n\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta} \rVert_{\mathscr{G} \times \mathscr{R}}$ and $\lVert Z_n^G\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta}-Z_n^G \rVert_{\mathscr{G} \times \mathscr{R}}$ using standard empirical process results, which give the contribution $\mathsf{F}(\delta)$ emerging from Talagrand's inequality \citep[Theorem 3.3.9]{Gine-Nickl_2016_Book-SA} combined with a standard maximal inequality \citep[Theorem 5.2]{chernozhukov2014gaussian-SA}. This allows us to focus on the error on the $\delta$-net to simply study $\lVert G_n - Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$.
\item Section~\ref{sa-sec: MULT -- sa error} addresses the strong approximation error $\lVert \mathtt{\Pi}_{1} G_n - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}$. The multiplicative structure of $G_n$ and the pre-factorization of coefficients in $\mathtt{\Pi}_{1} G_n$ and $\mathtt{\Pi}_{1} Z_n^G$ enable a new bound on the strong approximation error for the empirical process indexed by piecewise constant functions. Specifically, we establish a bound on $\mathbbm{E}[\lVert \mathtt{\Pi}_{1} G_n - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}^2]$ that is polynomial in the number of splits along the $y_i$-direction and exponential in the number of splits along the $\mathbf{x}_i$-direction. This is a key step in achieving a Gaussian strong approximation rate that treats splits along the $y_i$-dimension as residual contributions.
\item Section~\ref{sa-sec: MULT -- proj error} addresses the projection errors $\lVert G_n - \mathtt{\Pi}_{1} G_n \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$ and $\lVert Z_n^G - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$. We begin by comparing the two projections, bounding the differences $\lVert \mathtt{\Pi}_{1} G_n - \mathtt{\Pi}_{0} G_n \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$ and $\lVert \mathtt{\Pi}_{1} Z_n^G - \mathtt{\Pi}_{0} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$. Next, we control the $L_2$ projection errors $\lVert G_n - \mathtt{\Pi}_{0} G_n \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$ and $\lVert Z_n^G - \mathtt{\Pi}_{0} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$ using Bernstein inequality and similar arguments as in Section~\ref{sa-sec: X-process -- proj error}.
\end{itemize}
\subsection{Preliminary Technical Results}\label{sa-sec: MULT -- preliminary}
This section presents preliminary technical results that are used to prove Theorem~\ref{sa-thm: M-process -- main theorem}. Whenever possible, these results are presented at a higher level of generality, and therefore may be of independent theoretical interest. Throughout this section, we employ the following assumption.
\begin{Assumption}\label{sa-assump: MULT & REG -- step}
Suppose $(\mathbf{z}_i=(\mathbf{x}_i, y_i): 1 \leq i \leq n)$ are i.i.d. random vectors taking values in $(\mathbb{R}^{d+1}, \mathcal{B}(\mathbb{R}^{d+1}))$, where $(\mathbf{x}_1,y_1)$ has joint distribution $\mathbbm{P}_{Z}$. Suppose $\mathbf{x}_1$ has distribution $\mathbbm{P}_X$ supported on $\mathcal{X}\subseteq \mathbb{R}^d$, $y_1$ has distribution $\mathbbm{P}_Y$ supported on $\mathcal{Y}\subseteq\mathbb{R}$, and the following conditions hold.
\begin{enumerate}[label=\emph{(\roman*)}]
\item $\mathscr{G}$ is a real-valued pointwise measurable class of functions on $(\mathcal{X}, \mathcal{B}(\mathcal{X}), \mathbbm{P}_X)$.
\item $\mathtt{M}_{\mathscr{G},\mathcal{X}} < \infty$ and $J_{\mathcal{X}}(1,\mathscr{G},\mathtt{M}_{\mathscr{G},\mathcal{X}}) < \infty$.
\item $\mathscr{R}$ is a real-valued pointwise measurable class of functions on $(\mathcal{Y}, \mathcal{B}(\mathcal{Y}),\mathbbm{P}_Y)$.
\item $M_{\mathscr{R},\mathcal{Y}}(y) + \mathtt{pTV}_{\mathscr{R},(-|y|,|y|)} \leq \mathtt{v} (1 + |y|^{\alpha})$ for all $y \in \mathcal{Y}$, for some $\mathtt{v}>0$, and for some $\alpha\geq0$. Furthermore, if $\alpha>0$, then $\sup_{\mathbf{x} \in \mathcal{X}}\mathbbm{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] \leq 2$.
\item $J_{\mathcal{Y}}(\mathscr{R},M_{\mathscr{R},\mathcal{Y}},1) < \infty$.
\end{enumerate}
\end{Assumption}
Compared to the assumptions in Theorem 2, this assumption does not require the existence of a surrogate measure or a normalizing transformation. It will be applied in the analysis of terms in the error decomposition, where we work with the distribution $\mathbbm{P}_Z$, and an extra condition on the existence of Lebesgue density of $\mathbbm{P}_X$ is assumed whenever necessary (Section~\ref{sa-sec: MULT -- proj error}). The surrogate measure and the normalizing transformation will be used in the proof of Theorem~\ref{sa-thm: M-process -- main theorem} with the help of Section~\ref{sa-sec: X-Process -- normalizing transformation}, providing greater flexibility in the data generating process.
\subsubsection{Cells Expansions}\label{sa-sec: MULT -- cell expansions}
\begin{defn}[Cylindered Quasi-Dyadic Expansion of $\mathbb{R}^d$] \label{sa-defn: cylindered quasi dyadic expansion}
Let $\mathbbm{P}$ denote the joint distribution of $(\mathbf{X}, Y)$, a random vector taking values in $(\mathbb{R}^d \times \mathbb{R}, \mathcal{B}(\mathbb{R}^d) \otimes \mathcal{B}(\mathbb{R}))$, and let $\mathbbm{P}_X$ be the marginal distribution of $\mathbf{X}$. For a given $\rho \geq 1$, a collection of Borel measurable sets in $\mathbb{R}^{d+1}$, $\mathscr{C}_{M,N}(\mathbbm{P}, \rho) = \{\mathcal{C}_{j,k}: 0 \leq k < 2^{M + N - j}, 0 \leq j \leq M + N\}$, is called a cylindered quasi-dyadic expansion of $\mathbb{R}^{d+1}$ of depth $M$ for the main subspace $\mathbb{R}^d$ and depth $N$ for the multiplier subspace $\mathbb{R}$ with respect to $\mathbbm{P}$ if the following conditions hold:
\begin{enumerate}
\item For all $N \leq j \leq M + N$, $0 \leq k < 2^{M + N - j}$, there exists a set $\mathcal{X}_{j-N,k} \subseteq \mathbb{R}^d$ such that $\mathcal{C}_{j,k} = \mathcal{X}_{j-N,k} \times \mathcal{Y}_{*,N,0}$, with $\mathcal{Y}_{*,N,0}$ a subset of $\mathbb{R}$ and $\mathbbm{P}(\mathcal{C}_{M+N,0}) = 1$. The collection $\mathscr{C}_M(\mathbbm{P}_X,\rho) = \{\mathcal{X}_{l,k}: 0 \leq l \leq M, 0 \leq k < 2^{M-l}\}$ forms a quasi-dyadic expansion of depth $M$ with respect to $\mathbbm{P}_X$.
\item For all $0 \leq j < N$ and $0 \leq k < 2^{M + N - j}$, let $l$ and $m$ be the unique non-negative integers such that $k = 2^{N-j}l + m$. Then there exists a set $\mathcal{Y}_{l,j,m} \subseteq \mathbb{R}$ such that $\mathcal{C}_{j,k} = \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}$. Moreover, for each $0 \leq l < 2^M$, the collection $\{\mathcal{Y}_{l,j,m}: 0 \leq j < N, 0 \leq m < 2^{N - j}\}$ forms a dyadic expansion of depth $N$ with respect to the conditional distribution $\mathbbm{P}(Y \in \cdot \mid \mathbf{X} \in \mathcal{X}_{0,l})$, and $\mathcal{Y}_{l,N,0} = \mathcal{Y}_{*,N,0}$.
\end{enumerate}
When $\rho = 1$, $\mathscr{C}_{M,N}(\mathbbm{P},1)$ is called a \emph{cylindered dyadic expansion}. For notational simplicity, we write $\mathtt{p}_X[\mathscr{C}_{M,N}(\mathbbm{P}, \rho)] = \{\mathcal{X}_{l,k}: 0 \leq l \leq M, 0 \leq k < 2^{M-l}\}$.
\end{defn}
\begin{defn}[Axis-Aligned Quasi-Dyadic Expansion of $\mathbb{R}^d$]\label{sa-defn: cylindered axis aligned splitting}
Let $\mathbbm{P}$ denote the joint distribution of $(\mathbf{X}, Y)$, a random vector taking values in $(\mathbb{R}^d \times \mathbb{R}, \mathcal{B}(\mathbb{R}^d) \otimes \mathcal{B}(\mathbb{R}))$, and let $\mathbbm{P}_X$ be the marginal distribution of $\mathbf{X}$. For a given $\rho \geq 1$, a collection of Borel measurable sets in $\mathbb{R}^{d+1}$, $\mathscr{A}_{M,N}(\mathbbm{P}, \rho) = \{\mathcal{C}_{j,k}: 0 \leq k < 2^{M + N - j}, 0 \leq j \leq M + N\}$, is called an axis-aligned cylindered quasi-dyadic expansion of $\mathbb{R}^{d+1}$ of depth $M$ in the main subspace $\mathbb{R}^d$ and depth $N$ in the multiplier subspace $\mathbb{R}$ with respect to $\mathbbm{P}$ if the following conditions hold:
\begin{enumerate}
\item $\mathscr{A}_{M,N}(\mathbbm{P},\rho)$ is a cylindered quasi-dyadic expansion of $\mathbb{R}^{d+1}$, of depth $M$ for the main subspace $\mathbb{R}^d$ and depth $N$ for the multiplier subspace $\mathbb{R}$, with respect to $\mathbbm{P}$.
\item $\mathtt{p}_X[\mathscr{A}_{M,N}(\mathbbm{P}, \rho)] = \{\mathcal{X}_{l,k}: 0 \leq l \leq M, 0 \leq k < 2^{M-l}\}$ forms an axis-aligned quasi-dyadic expansion of depth $M$ with respect to $\mathbbm{P}_X$.
\end{enumerate}
When $\rho = 1$, $\mathscr{A}_{M,N}(\mathbbm{P},1)$ is called an \emph{axis-aligned cylindered dyadic expansion}.
\end{defn}
\subsubsection{Projection onto Piecewise Constant Functions}\label{sa-sec: MULT -- proj}
Consider a cylindered quasi-dyadic expansion $\mathcal{C}_{M,N}(\mathbbm{P}, \rho)$ where $\mathbbm{P}$ is the joint distribution of a random vector $(\mathbf{X}, Y)$ taking values in $(\mathbb{R}^d \times \mathbb{R}, \mathcal{B}(\mathbb{R}^d) \otimes \mathcal{B}(\mathbb{R}))$. Define the span of the Haar basis over the terminal cells as described in Section~\ref{sa-sec: X-process -- Projection onto Piecewise Constant Functions}, specifically
\begin{align*}
\mathscr{E}_{M + N} = \operatorname{Span}\{\mathbbm{1}_{\mathcal{C}_{0,k}}: 0 \leq k < 2^{M + N}\}.
\end{align*}
For $h \in L_2(\mathbbm{P})$, recall that the mean square projection of $h$ onto $\mathscr{E}_{M+N}$ is given by
\begin{align*}
\mathtt{\Pi}_{0}(\mathscr{C}_{M,N}(\mathbbm{P},\rho))[h] = \sum_{0 \leq k < 2^{M+N}} \frac{\mathbbm{1}_{\mathcal{C}_{0,k}}}{\mathbbm{P}(\mathcal{C}_{0,k})} \int_{\mathcal{C}_{0,k}} h(\mathbf{u}) \, d \mathbbm{P}(\mathbf{u}).
\end{align*}
and the $\beta$-coefficients are defined by
\begin{gather*}
\beta_{j,k}(h) = \frac{1}{\mathbbm{P}(\mathcal{C}_{j,k})} \int_{\mathcal{C}_{j,k}}h(\mathbf{u}) d \mathbbm{P}(\mathbf{u}), \qquad \widetilde{\beta}_{j,k}(h) = \beta_{j-1,2k}(h) - \beta_{j-1,2k+1}(h).
\end{gather*}
Then we still have
\begin{align*}
\mathtt{\Pi}_{0}(\mathscr{C}_{M,N}(\mathbbm{P},\rho))[h] = \beta_{K,0}(h) e_{K,0} + \sum_{1 \leq j \leq K} \sum_{0 \leq k < 2^{K - j}} \widetilde{\beta}_{j,k}(h) \widetilde{e}_{j,k},
\end{align*}
where
\begin{align*}
e_{j,k} = \mathbbm{1}_{\mathcal{C}_{j,k}}, \qquad
\widetilde{e}_{j,k} = \frac{\mathbbm{P}(C_{j-1,2k+1})}{\mathbbm{P}(C_{j,k})} e_{j-1,2k} - \frac{\mathbbm{P}(\mathcal{C}_{j-1,2k})}{\mathbbm{P}(\mathcal{C}_{j,k})} e_{j-1,2k+1},
\end{align*}
for all $(j,k) \in \mathcal{I}_{M+N} = \{(j,k) \in \mathbb{N} \times \mathbb{N}: 1 \leq j \leq M+N, 0 \leq k < 2^{M + N - j}\}$. We refer to $\mathtt{\Pi}_{0}(\mathcal{C}_{M,N}(\mathbbm{P}, \rho))$ as $\mathtt{\Pi}_{0}$ for simplicity.
To address the separable structure of $g(\mathbf{X}) r(Y)$, we define the \textit{product-factorized projection} from $L_2(\mathbbm{P})$ to $\mathscr{E}_{M+N} = \operatorname{Span}\{\mathcal{C}_{0,k} = \mathcal{X}_{0,l} \times \mathcal{Y}_{l,0,m} : 0 \leq l < 2^M, 0 \leq m < 2^N, k = 2^N l + m\}$, defined as
\begin{align}\label{sa-eq: projmult}
\mathtt{\Pi}_1(\mathcal{C}_{M,N}(\mathbbm{P},\rho))[g, r] = \gamma_{M+N,0}(g, r) e_{M+N,0} + \sum_{1 \leq j \leq M + N} \sum_{0 \leq k < 2^{M + N - j}} \widetilde{\gamma}_{j,k}(g, r) \widetilde{e}_{j,k},
\end{align}
and
\begin{align*}
\gamma_{j,k}(g, r)
& =
\begin{cases}
\mathbbm{E}[g(\mathbf{X}) r(Y) | \mathbf{X} \in \mathcal{X}_{j-N,k}], & \text{if } N \leq j \leq M+N, \\
\mathbbm{E}[g(\mathbf{X}) | \mathbf{X} \in \mathcal{X}_{0,l}] \cdot \mathbbm{E}[r(Y) | \mathbf{X} \in \mathcal{X}_{0,l}, Y \in \mathcal{Y}_{l,0,m}], & \text{if } j < N, k = 2^{N - j} l + m,
\end{cases}
\end{align*}
and $\widetilde{\gamma}_{j,k}(g, r) = \gamma_{j-1,2k}(g, r) - \gamma_{j-1,2k+1}(g, r)$. We refer to $\mathtt{\Pi}_1(\mathcal{C}_{M,N}(\mathbbm{P}, \rho))$ as $\mathtt{\Pi}_{1}$ for simplicity.
The Haar basis representation in Equation~\eqref{sa-eq: projmult} decomposes the function into layers of increasingly localized fluctuations. However, at lower layers ($1 \leq j \leq N$), the local fluctuation is characterized by the \textit{product-factorized projection} $\mathbbm{E}[g(\mathbf{X}) | \mathbf{X} \in \mathcal{X}_{0,l}] \cdot \mathbbm{E}[r(Y) | \mathbf{X} \in \mathcal{X}_{0,l}, Y \in \mathcal{Y}_{l,0,m}]$, rather than $\mathbbm{E}[g(\mathbf{X}) r(Y) | \mathbf{X} \in \mathcal{X}_{0,l} \times \mathcal{Y}_{l,0,m}]$. This distinction makes $\mathtt{\Pi}_1(\mathcal{C}_{M,N}(\mathbbm{P},\rho))[g, r]$ generally different from $\mathtt{\Pi}_{0}(\mathcal{C}_{M,N}(\mathbbm{P},\rho))[g \cdot r]$.
Now, we define the empirical processes indexed by projected functions. For any real valued functions $g$ on $\mathbb{R}^d$ and $r$ on $\mathbb{R}$ such that $\int_{\mathbb{R}^d} \int_{\mathbb{R}} g(\mathbf{x})^2 \mathbbm{P}(d y d \mathbf{x}) < \infty$ and $\int_{\mathbb{R}^d}\int_{\mathbb{R}}r(y)^2\mathbbm{P}(d y d \mathbf{x}) < \infty$, we define
\begin{equation}\label{sa-eq: proj processes}
\begin{aligned}
\mathtt{\Pi}_{1} G_n(g, r) & = X_n \circ \mathtt{\Pi}_{1}[\mathcal{C}_{M,N}(\mathbbm{P}, \rho)](g, r), \\
\mathtt{\Pi}_{0} G_n(g, r) & = X_n \circ \mathtt{\Pi}_{0}[\mathcal{C}_{M,N}(\mathbbm{P}, \rho)](g r),
\end{aligned}
\end{equation}
recalling $(X_n(f) : f \in \mathscr{F})$ is the empirical process based on a random sample $(\mathbf{z}_i = (\mathbf{x}_i, y_i) : 1 \leq i \leq n)$ with
\begin{align*}
X_n(f) = n^{-1/2} \sum_{i = 1}^n (f(\mathbf{x}_i, y_i) - \mathbbm{E}[f(\mathbf{x}_i, y_i)]).
\end{align*}
\subsubsection{Strong Approximation Construction}\label{sa-sec: MULT -- sa construction}
In this section, we construct the Gaussian process $Z_n^G$ (along with some auxiliary Gaussian processes) on a possibly enlarged probability space to couple with the empirical process $G_n$.
\begin{lemma}\label{sa-lem: M- processes pregaussian dyadic}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds, and a cylindered quasi-dyadic expansion $\mathscr{C}_{K}(\mathbbm{P}_Z,\rho)$ is given. Then, $(\mathscr{G} \cdot \mathscr{R}) \cup \mathtt{\Pi}_{0}(\mathscr{G} \times \mathscr{R}) \cup \mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R})$ is $\mathbbm{P}_Z$-pregaussian.
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: M- processes pregaussian dyadic}}
By the entropy integral conditions on $\mathscr{G}$ and $\mathscr{R}$ and Definitions~10 and~\ref{sa-defn: uniform entropy integral for product space},
\begin{align*}
J_{\mathcal{X} \times \mathcal{Y}}(\mathscr{G} \cdot \mathscr{R}, \mathtt{M}_{\mathscr{G},\mathcal{X}} M_{\mathscr{R},\mathcal{Y}}, \delta)
& =
J_{\mathcal{X} \times \mathcal{Y}}(\mathscr{G} \times \mathscr{R}, \mathtt{M}_{\mathscr{G},\mathcal{X}} M_{\mathscr{R},\mathcal{Y}}, \delta) \\
& \leq \sqrt{2} J_{\mathcal{X} \times \mathcal{Y}}(\overline{\mathscr{G}}, \mathtt{M}_{\mathscr{G},\mathcal{X}}, \delta/\sqrt{2}) + \sqrt{2} J_{\mathcal{X} \times \mathcal{Y}}(\overline{\mathscr{R}}, M_{\mathscr{R},\mathcal{Y}}, \delta/\sqrt{2}) \\
& \leq \sqrt{2} J_{\mathcal{X}}(\mathscr{G}, \mathtt{M}_{\mathscr{G},\mathcal{X}}, \delta/\sqrt{2}) + \sqrt{2} J_{\mathcal{Y}}(\mathscr{R}, M_{\mathscr{R},\mathcal{Y}}, \delta/\sqrt{2})
\end{align*}
where $\overline{\mathscr{G}} = \{(\mathbf{x},y) \in \mathcal{X} \times \mathcal{Y} \mapsto g(\mathbf{x}): g \in \mathscr{G}\}$ and $\overline{\mathscr{R}} = \{(\mathbf{x},y) \in \mathcal{X} \times \mathcal{Y} \mapsto r(y): r \in \mathscr{R}\}$. \\
\noindent \underline{Claim 1}: For all $0 < \delta < 1$,
\begin{align*}
J_{\mathcal{X} \times \mathcal{Y}}(\mathtt{\Pi}_{0}(\mathscr{G} \times \mathscr{R}), c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G},\mathcal{X}} N^{\alpha}, \delta) \leq J_{\mathcal{X} \times \mathcal{Y}}(\mathscr{G} \times \mathscr{R}, \mathtt{M}_{\mathscr{G},\mathcal{X}} M_{\mathscr{R},\mathcal{Y}}, \delta),
\end{align*}
where $c_{\mathtt{v},\alpha} = \mathtt{v} \max\{1 + (2 \alpha)^{\frac{\alpha}{2}}, 1 + (4 \alpha)^{\alpha}\}$.\newline
\noindent \underline{Proof of Claim 1}: We consider the two cases of whether $\alpha > 0$ in Assumption~\ref{sa-assump: MULT & REG -- step} (iv) separately.
If $\alpha > 0$, by Step 2 in Definition~\ref{sa-defn: cylindered quasi dyadic expansion}, $\max_{0 \leq l < 2^{M + N}}\mathbbm{E}[\exp(y_i/(N \log 2))|(\mathbf{x}_i, y_i) \in \mathcal{C}_{0,l}] \leq 2$. Hence
\begin{align}\label{sa-eq: proj r bound}
\nonumber \max_{0 \leq l < 2^{M + N}} \sup_{r \in \mathscr{R}} \mathbbm{E}[|r(y_i)||(\mathbf{x}_i, y_i) \in \mathcal{C}_{0,l}] & \leq \mathtt{v}(1 + \max_{0 \leq l < 2^{M + N}} \mathbbm{E}[|y_i|^{\alpha}|(\mathbf{x}_i,y_i) \in \mathcal{C}_{0,l}]) \\
& \leq \mathtt{v}(1 + (2 N \sqrt{\alpha})^{\alpha}).
\end{align}
Definition of $\mathtt{\Pi}_{0}$ then implies
\begin{align}\label{sa-eq: proj_0 bound}
\sup_{g \in \mathscr{G}} \sup_{r \in \mathscr{R}} \sup_{(\mathbf{x},y) \in \mathcal{C}_{M+N,0}} |\mathtt{\Pi}_{0}(gr)(\mathbf{x},y)| \leq c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G},\mathcal{X}} N^{\alpha}.
\end{align}
Moreover, if $\alpha = 0$, Assumption~\ref{sa-assump: MULT & REG -- step} (iv) implies $\mathtt{M}_{\mathscr{R},\mathcal{Y}} \leq 1$, Equations~\eqref{sa-eq: proj r bound}, \eqref{sa-eq: proj_0 bound} hold with $\alpha = 0$.
Let $Q$ be a finite discrete measure on $\mathcal{X} \times \mathcal{Y}$. Definition of $\mathtt{\Pi}_{0}$ and Jensen's inequality implies
\begin{align*}
\lVert \mathtt{\Pi}_{0} f - \mathtt{\Pi}_{0} g \rVert_{Q,2}^2 & \leq \sum_{0 \leq k < 2^{M + N}} Q(C_{0,k}) (2^{M + N}\int_{C_{0,k}} f - g d\mathbbm{P}_Z)^2 \\
& \leq \sum_{0 \leq k < 2^{M + N}} Q(C_{0,k}) 2^{M + N}\int_{C_{0,k}} (f - g)^2 d\mathbbm{P}_Z, \qquad \forall f, g \in \mathscr{G} \cdot \mathscr{R}.
\end{align*}
Define a measure $\widetilde{Q}$ such that for any $A \in \mathcal{B}(\mathbb{R}^d \times \mathbb{R})$, $\widetilde{Q}(A) = \sum_{0 \leq k < 2^{M + N}} Q(C_{0,k}) 2^{M + N} \mathbbm{P}_Z(A \cap C_{0,k})$, then
\begin{align*}
\lVert \mathtt{\Pi}_{0} f - \mathtt{\Pi}_{0} g \rVert^2_{Q,2} \leq \lVert f - g \rVert_{\widetilde{Q},2}^2, \qquad \forall f, g \in \mathscr{G} \cdot \mathscr{R}.
\end{align*}
Lemma~\ref{sa-lem: vc to rho} implies that there exists an $\delta c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G},\mathcal{X}} N^{\alpha}$-net $\mathcal{L}$ of $\mathscr{G} \times \mathscr{R}$ with cardinality no greater than $\mathtt{N}_{\mathscr{G} \times \mathscr{R}, \mathcal{X} \times \mathcal{Y}}(\delta,\lVert \mathtt{M}_{\mathscr{G},\mathcal{X}} M_{\mathscr{R},\mathcal{Y}} \rVert_{\widetilde{Q},2})$ such that for all $f \in \mathtt{\Pi}_{0}(\mathscr{G} \times \mathscr{R})$, there exists $g \in \mathcal{L}$ such that
\begin{align*}
\lVert f - g \rVert_{\widetilde{Q},2}^2 \leq \delta^{2} \lVert \mathtt{M}_{\mathscr{G},\mathcal{X}} M_{\mathscr{R},\mathcal{Y}} \rVert_{\widetilde{Q},2}^2 \leq \delta^2 (c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G},\mathcal{X}} N^{\alpha})^2.
\end{align*}
The claim then follows. \\
\noindent \underline{Claim 2}: For all $0 < \delta < 1$,
\begin{align*}
J_{\mathcal{X}\times\mathcal{Y}}(\mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}), c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G},\mathcal{X}} N^{\alpha}, \delta)
\lesssim J_{\mathcal{X}\times\mathcal{Y}}(\mathscr{G} \times \mathscr{R}, \mathtt{M}_{\mathscr{G},\mathcal{X}} M_{\mathscr{R},\mathcal{Y}}, \delta/3).
\end{align*}
\noindent \underline{Proof of Claim 2}: Definition~\ref{sa-defn: quasi dyadic expansion} and the definition of product factorized projection imply that for the upper layers with $N \leq j \leq M + N$,
\begin{align*}
\gamma_{j,k}(g,r)
= \mathbbm{E}[g(\mathbf{x}_i)r(y_i)|\mathbf{x}_i \in \mathcal{X}_{j-N,k}]
= \mathbbm{E}[g(\mathbf{x}_i)r(y_i)|(\mathbf{x}_i,y_i) \in \mathcal{C}_{j-N,k}],
\end{align*}
that is, the coefficients coincide with those from the mean square projection. Take $\mathscr{C}_{M,0} = \{\mathcal{C}_{j,k}: N \leq j \leq M+ N, 0 \leq k < 2^{M + N - j}\}$ to be the collection of all upper layer cells down to the $N$-th layer, then
\begin{align*}
\mathtt{\Pi}_{1}[\mathscr{C}_{M,0}(\mathbbm{P}_Z, \rho)](g,r) = \mathtt{\Pi}_{0}[\mathscr{C}_{M,0}(\mathbbm{P}_Z, \rho)](gr), \qquad g \in \mathscr{G}, r \in \mathscr{R}.
\end{align*}
For the lower layers $0 \leq j < N$, suppose $\widetilde{\mathbbm{P}}_Z$ is a mapping from $\mathcal{B}(\mathbb{R}^{d+1})$ to $[0,1]$ such that
\begin{align*}
\widetilde{\mathbbm{P}}_Z(E)
& = \inf \bigg\{\sum_{\ell = 1}^{\mathfrak{L}} \sum_{0 \leq l < 2^M} \sum_{0 \leq m < 2^N} \mathbbm{E}[\mathbbm{1}(\mathbf{x}_i \in A_\ell)|\mathbf{x}_i \in \mathcal{X}_{0,l}]\cdot \mathbbm{E}[\mathbbm{1}(y_i \in B_\ell)|\mathbf{x}_i \in \mathcal{X}_{0,l}, y_i \in \mathcal{Y}_{l,0,m}]: \\
& \qquad E \subseteq \sqcup_{\ell=1}^{\mathfrak{L}} A_\ell \times B_\ell \text{ with } A_\ell \times B_\ell, 1 \leq l \leq \mathfrak{L} \in \mathbb{N}, \text{ disjoint rectangles in } \mathcal{B}(\mathbb{R}^d) \otimes \mathcal{B}(\mathbb{R}) \bigg\},
\end{align*}
with $E \in \mathcal{B}(\mathbb{R}^{d+1})$. It follows that $\widetilde{\mathbbm{P}}_Z$ defines a probability measure on $(\mathbb{R}^{d+1}, \mathcal{B}(\mathbb{R}^{d+1}))$, and
\begin{align*}
\gamma_{j,k}(g,r)
& = \mathbbm{E}[g(\mathbf{x}_i) | \mathbf{x}_i \in \mathcal{X}_{0,l}] \cdot \mathbbm{E}[r(y_i) | \mathbf{x}_i \in \mathcal{X}_{0,l}, y_i \in \mathcal{Y}_{l,j,m}] \\
& = \sum_{m^{\prime}: \mathcal{Y}_{l,j,m^{\prime}}\subseteq \mathcal{Y}_{l,j,m}} 2^{-j} \mathbbm{E}[g(\mathbf{x}_i) | \mathbf{x}_i \in \mathcal{X}_{0,l}] \cdot \mathbbm{E}[r(y_i) | \mathbf{x}_i \in \mathcal{X}_{0,l}, y_i \in \mathcal{Y}_{l,0,m^{\prime}}] \\
& = \sum_{m^{\prime}: \mathcal{Y}_{l,j,m^{\prime}}\subseteq \mathcal{Y}_{l,j,m}} 2^{-j} \mathbbm{E}_{\widetilde{\mathbbm{P}}_Z}[g(\mathbf{x}_i)r(y_i)|\mathbf{x}_i \in \mathcal{X}_{0,l}, y_i \in \mathcal{Y}_{l,j,m^{\prime}}] \\
& = \mathbbm{E}_{\widetilde{\mathbbm{P}}_Z}[g(\mathbf{x}_i)r(y_i)|\mathbf{x}_i \in \mathcal{X}_{0,l}, y_i \in \mathcal{Y}_{l,j,m}] \\
& = \mathbbm{E}_{\widetilde{\mathbbm{P}}_Z}[g(\mathbf{x}_i)r(y_i)|(\mathbf{x}_i,y_i) \in \mathcal{C}_{j,k}], \qquad 0 \leq j < N, 0 \leq k < 2^{M+N-j},
\end{align*}
where $\mathbbm{E}_{\widetilde{\mathbbm{P}}_Z}$ means the expectation is taken with $(\mathbf{x}_i,y_i)$ following the law of $\widetilde{\mathbbm{P}}_Z$ instead of $\mathbbm{P}_Z$. This implies
\begin{align*}
\mathtt{\Pi}_{1}[\mathscr{C}_{M, N}(\mathbbm{P}_Z, \rho)](g,r) - \mathtt{\Pi}_{1}[\mathscr{C}_{M,0}(\mathbbm{P}_Z, \rho)](g,r)
= \mathtt{\Pi}_{0}[\mathscr{C}_{M,N}(\widetilde{\mathbbm{P}}_Z, \rho)](gr) - \mathtt{\Pi}_{0}[\mathscr{C}_{M,0}(\widetilde{\mathbbm{P}}_Z, \rho)](gr), \quad g \in \mathscr{G}, r \in \mathscr{R}.
\end{align*}
We can then express the $\mathtt{\Pi}_{1}$ projection of $(g,r)$ as three $L_2$ projections as follows:
\begin{align*}
\mathtt{\Pi}_{1}[\mathscr{C}_{M, N}(\mathbbm{P}_Z, \rho)](g,r) & = \mathtt{\Pi}_{1}[\mathscr{C}_{M,0}(\mathbbm{P}_Z, \rho)](g,r) + \mathtt{\Pi}_{1}[\mathscr{C}_{M, N}(\mathbbm{P}_Z, \rho)](g,r) - \mathtt{\Pi}_{1}[\mathscr{C}_{M,0}(\mathbbm{P}_Z, \rho)](g,r) \\
& = \mathtt{\Pi}_{0}[\mathscr{C}_{M,0}(\mathbbm{P}_Z, \rho)](gr) + \mathtt{\Pi}_{0}[\mathscr{C}_{M,N}(\widetilde{\mathbbm{P}}_Z, \rho)](gr) - \mathtt{\Pi}_{0}[\mathscr{C}_{M,0}(\widetilde{\mathbbm{P}}_Z, \rho)](gr), g \in \mathscr{G}, r \in \mathscr{R}.
\end{align*}
Since $\lVert \mathtt{\Pi}_{0}[\mathscr{C}_{M,N}(\widetilde{\mathbbm{P}}_Z,\rho)]\rVert_{\mathscr{G} \times \mathscr{R}} \leq c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G},\mathcal{X}} N^{\alpha}$, Claim 1 applies to all of the three terms:
\begin{align*}
& J_{\mathcal{X} \times \mathcal{Y}}(\mathtt{\Pi}_{0}[\mathscr{C}_{M,N}(\mathbbm{P}_Z,\rho)](\mathscr{G} \times \mathscr{R}), c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G},\mathcal{X}} N^{\alpha}, \delta) + J_{\mathcal{X} \times \mathcal{Y}}(\mathtt{\Pi}_{0}[\mathscr{C}_{M,N}(\widetilde{\mathbbm{P}}_Z,\rho)](\mathscr{G} \times \mathscr{R}), c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G},\mathcal{X}} N^{\alpha}, \delta) \\
& \qquad + J_{\mathcal{X} \times \mathcal{Y}}(\mathtt{\Pi}_{0}[\mathscr{C}_{M,0}(\widetilde{\mathbbm{P}}_Z,\rho)](\mathscr{G} \times \mathscr{R}), c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G},\mathcal{X}} N^{\alpha}, \delta) \lesssim J_{\mathcal{X} \times \mathcal{Y}}(\mathscr{G} \times \mathscr{R}, \mathtt{M}_{\mathscr{G},\mathcal{X}} M_{\mathscr{R},\mathcal{Y}}, \delta).
\end{align*}
Then Claim 2 follows from Claim 1.
Putting together,
\begin{align*}
& \quad J_{\mathcal{X} \times \mathcal{Y}}((\mathscr{G} \times \mathscr{R}) \cup \mathtt{\Pi}_{0}(\mathscr{G} \times \mathscr{R}) \cup \mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}),\mathtt{M}_{\mathscr{G},\mathcal{X}}M_{\mathscr{R},\mathcal{Y}} + c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G},\mathcal{X}} N^{\alpha},1) \\
& \qquad \lesssim J_{\mathcal{X}}(\mathscr{G},\mathtt{M}_{\mathscr{G},\mathcal{X}},1) + J_{\mathcal{Y}}(\mathscr{R},M_{\mathscr{R},\mathcal{Y}},1) < \infty,
\end{align*}
and the conclusion follows from separability of $\mathscr{G}$ and $\mathscr{R}$, and \citet[Corollary 2.2.9]{wellner2013weak-SA}.
\end{myproof}
The construction of the Gaussian process essentially follows from the arguments in Section~\ref{sa-sec: X-process -- Strong Approximation Constructions} with $\mathbf{z}_i$'s replacing $\mathbf{x}_i$'s. (Recall that $\mathbf{z}_i = (\mathbf{x}_i, y_i)$ in this section.) We start with a Gaussian process indexed by $(\mathscr{G} \cdot \mathscr{R}) \cup \mathtt{\Pi}_{0}(\mathscr{G} \times \mathscr{R}) \cup \mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R})$ with almost sure continuous sample paths, and take conditional quantile transformations of Gaussian process indexed by $\mathbbm{1}_{\mathcal{C}_{j,k}}$ to construct counts of $(\mathbf{x}_i, y_i)$'s on the cells $\mathcal{C}_{j,k}$'s. By a Skorohod embedding argument, this Gaussian process can be taken on a possibly enriched probability space. More precisely, we have the following result.
\begin{lemma}\label{sa-lem: M- processes sa for pcw-const}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds and a cylindered dyadic expansion $\mathscr{C}_{M,N}(\mathbbm{P}_Z, 1)$ is given. Then on a possibly enlarged probability space, there exists a $\mathbbm{P}_Z$-Brownian bridge $B_n$ indexed by $\mathscr{F} = (\mathscr{G} \cdot \mathscr{R}) \cup \mathtt{\Pi}_{0}(\mathscr{G} \times \mathscr{R}) \cup \mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R})$ with almost sure continuous trajectories on $(\mathscr{F},\mathfrak{d}_{\mathbbm{P}_Z})$ such that for any $f \in \mathscr{F}$ and any $x > 0$,
\begin{align*}
\mathbbm{P} \left(\bigg|\sum_{i = 1}^n f(\mathbf{x}_i, y_i) - \sqrt{n} B_n(f)\bigg| \geq 24 \sqrt{\lVert f \rVert^2_{\mathscr{E}_{M+N}}x} + 4 \sqrt{\mathtt{C}_{\{f\},M+N}}x \right)
\leq 2 \exp(-x),
\end{align*}
where for both $\lVert f \rVert_{\mathscr{E}_{M+N}}^2$ and $\mathtt{C}_{\{f\},M+N}$ are defined in Lemma~\ref{sa-lem: X-process -- sa for pcw-const}.
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: M- processes sa for pcw-const}}
The result follows from Lemma~\ref{sa-lem: M- processes pregaussian dyadic} and Lemma~\ref{sa-lem: X-process -- sa for pcw-const} with $(\mathbf{x}_i, y_i)$ replacing $\mathbf{x}_i$.
\end{myproof}
\begin{lemma}\label{sa-lem: M- processes sa for pcw-const quasi}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds and a cylindered quasi-dyadic expansion $\mathscr{C}_{M,N}(\mathbbm{P}_Z, \rho)$ with $\rho > 1$ is given. Then on a possibly enlarged probability space, there exists a $\mathbbm{P}_Z$-Brownian bridge $B_n$ indexed by $\mathscr{F} = (\mathscr{G} \cdot \mathscr{R}) \cup \mathtt{\Pi}_{0}(\mathscr{G} \times \mathscr{R}) \cup \mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R})$ with almost sure continuous trajectories on $(\mathscr{F},\mathfrak{d}_{\mathbbm{P}_Z})$ such that for any $f \in \mathscr{F}$ and $x > 0$,
\begin{align*}
\mathbbm{P} \left(\bigg|\sum_{i = 1}^n f(\mathbf{x}_i, y_i) - \sqrt{n} B_n(f) \bigg| \geq C_{\rho} \sqrt{\lVert f \rVert^2_{\mathscr{E}_{M+N}}x} + C_{\rho}\sqrt{\mathtt{C}_{\{f\},M+N}}x \right)
\leq 2 \exp(-x) + 2^{M + 2}\exp(- C_{\rho} n 2^{-M}),
\end{align*}
where $C_{\rho}$ is a constant that only depends on $\rho$.
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: M- processes sa for pcw-const quasi}}
Replacing $\mathbf{x}_i$ by $\mathbf{z}_i = (\mathbf{x}_i, y_i)$ in Section~\ref{sa-sec: X-process -- Strong Approximation Constructions} (and with the help of the pregaussian lemma \ref{sa-lem: M- processes pregaussian dyadic}), suppose we constructed as therein on a possibly enlarged probability space the i.i.d standard Gaussian random variables $(\widetilde{\xi}_{j,k}:(j,k) \in \mathcal{I}_{M + N})$ and the Binomial counts $(U_{j,k}: (j,k) \in \mathcal{J}_{M+N}) = (\sum_{i=1}^n e_{j,k}(\mathbf{z}_i): (j,k) \in \mathcal{J}_{M+N})$. Again, we take $\widetilde{U}_{j,k} = U_{j-1,2k} - U_{j-1,2k+1}$ for $(j,k) \in \mathcal{I}_{M+N}$. By Definition~\ref{sa-defn: cylindered quasi dyadic expansion}, the upper layer cells $(N \leq j \leq M + N)$ may not be dyadic with respect to $\mathbbm{P}_Z$, but the lower layer cells $(0 \leq j < N)$ are. Tusnády's Lemma \citep[Lemma 4]{Bretagnolle-Massart_1989_AOP-SA} and Lemma~\ref{sa-lem: binom coupling} then imply whenever the event $\mathcal{A}$ holds, with
\begin{align*}
\mathcal{A} = \{|\widetilde{U}_{j,k}| \leq c_{1,\rho} U_{j,k}, \text{ for all } N \leq j \leq M + N, 0 \leq k < 2^{M + N - j}\},
\end{align*}
we know the following relations hold almost surely in $\mathbbm{P}_Z$,
\begin{align*}
& \left|\widetilde{U}_{j,k} - \sqrt{U_{j,k} \frac{\mathbbm{P}_Z(\mathcal{C}_{j-1,2k}) \mathbbm{P}_Z(\mathcal{C}_{j-1,2k+1})}{\mathbbm{P}_Z(\mathcal{C}_{j,k})^2}} \widetilde{\xi}_{j,k} \right| < c_{2,\rho} \widetilde{\xi}_{j,k}^2 + c_{3,\rho}, \\
& \left|\widetilde{U}_{j,k} \right| \leq 1/c_{0,\rho} + 2 \sqrt{\frac{\mathbbm{P}_Z(\mathcal{C}_{j-1,2k}) \mathbbm{P}_Z(\mathcal{C}_{j-1,2k+1})}{\mathbbm{P}_Z(\mathcal{C}_{j,k})^2}U_{j,k}}|\widetilde{\xi}_{j,k}|,
\end{align*}
for all $(j,k)\in\mathcal{I}_{M + N}$, and where $c_{0,\rho},c_{1,\rho}, c_{2,\rho},c_{3,\rho}$ are constants that only depends on $\rho$. By similar argument as in the proof for Lemma~\ref{sa-lem: X-process -- sa for pcw-const non-dyadic}, $\mathbbm{P}(\mathcal{A}^c) \leq 3 \cdot 2^M \exp(- \min\{c_{1,\rho}^2/3,1/8\} \rho^{-1} n 2^{-M})$. The rest of the proof follows from Lemma~\ref{sa-lem: X-process -- sa for pcw-const non-dyadic} by replacing $\mathbf{x}_i$ with $(\mathbf{x}_i, y_i)$.
\end{myproof}
The above two lemmas allow for constructions of Gaussian processes and projected Gaussian processes as counterparts of the empirical processes in Section~\ref{sa-sec: X-process -- Strong Approximation Constructions}. In particular, we take $Z_n^G, \mathtt{\Pi}_{0} Z_n^G, \mathtt{\Pi}_{1} Z_n^G$ to be the empirical processes indexed by $\mathscr{G} \times \mathscr{R}$ such that
\begin{align}\label{sa-eq: M- Process -- Gaussian Process 1}
Z_n^G(g,r) = B_n(g \cdot r), \qquad (g,r) \in \mathscr{G} \times \mathscr{R}.
\end{align}
We also define the following ancillary processes for analysis:
\begin{gather}\label{sa-eq: M- Process -- Gaussian Process 2}
\mathtt{\Pi}_{0} Z_n^G(g,r) = B_n(\mathtt{\Pi}_{0}[g \cdot r]),\qquad \mathtt{\Pi}_{1} Z_n^G(g,r) = B_n(\mathtt{\Pi}_{1}[g,r]), \qquad (g,r) \in \mathscr{G} \times \mathscr{R}.
\end{gather}
In particular, $(Z_n^G(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ has almost sure continuous trajectories in $(\mathscr{G} \times \mathscr{R},\mathfrak{d}_{\mathbbm{P}_Z})$
The following ancillary lemma for uniform covering number and uniform entropy integrals is used in the proof of Lemma~\ref{sa-lem: M- processes pregaussian dyadic}.
\begin{lemma}[Covering Number using Covariance Semi-metric]\label{sa-lem: vc to rho}
Assume $\mathscr{F}$ is a class of functions from a measurable space $(\mathcal{X}, \mathcal{B}(\mathcal{X}))$ to $\mathbb{R}$ with envelope function $M_{\mathscr{F},\mathcal{X}}$. Let $P$ be a probability measure on $(\mathcal{X}, \mathcal{B}(\mathcal{X}))$. Then, for any $0 < \varepsilon < 1$,
\begin{align*}
N(\mathscr{F}, \lVert \cdot \rVert_{P,2}, \varepsilon \lVert M_{\mathscr{F},\mathcal{X}} \rVert_{P,2}) \leq \mathtt{N}_{\mathscr{F},\mathcal{X}}(\varepsilon,M_{\mathscr{F},\mathcal{X}}).
\end{align*}
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: vc to rho}}
The proof essentially follows from the arguments for \cite[Theorem 2.5.2]{wellner2013weak-SA}, but we present here for completeness. Define $\mathscr{H} = \{(f - g)^2: f,g \in \mathscr{F}\} \cup \{M_{\mathscr{F},\mathcal{X}}\}$. Then, for all $0 < \varepsilon < 1$,
\begin{align*}
\sup_{Q} N(\mathscr{H}, \lVert \cdot \rVert_{Q,1}, \varepsilon \lVert M_{\mathscr{F},\mathcal{X}}^2 \rVert_{Q,1}) \leq \sup_{Q} N(\mathscr{H}, \lVert \cdot \rVert_{Q,1}, \varepsilon \lVert M_{\mathscr{F},\mathcal{X}}^2 \rVert_{Q,2}) \leq \sup_{Q} N(\mathscr{F}, \lVert \cdot \rVert_{Q,1}, \varepsilon \lVert M_{\mathscr{F},\mathcal{X}} \rVert_{Q,1})^2,
\end{align*}
where the supremums are all taken over finite discrete measures on $\mathcal{X}$. By Theorem 2.4.3 in \cite{wellner2013weak-SA}, $\mathscr{H}$ is Glivenko-Cantelli. Let $X_1, X_2, \ldots$ be a sequence of i.i.d. random variables with distribution $P$. Define $Q_N = \frac{1}{N}\sum_{j = 1}^N \delta_{X_j}$. Let $0 <\varepsilon <1$ and $\delta > 0$. Then there exists $N \in \mathbb{N}$ and a realization $x_1, \ldots, x_N$ of $X_1, \ldots, X_N$ such that if we denote $P_N = \frac{1}{N}\sum_{i = 1}^N \delta_{x_i}$, then for all $f_1, f_2 \in \mathscr{F}$,
\begin{align*}
& \left|\lVert f_1 - f_2 \rVert_{P,2}^2- \lVert f_1 - f_2 \rVert_{P_N,2}^2 \right| \leq \delta^2 \varepsilon^2 \lVert M_{\mathscr{F},\mathcal{X}} \rVert_{P,2}^2, \\
& \left|\lVert M_{\mathscr{F},\mathcal{X}} \rVert_{P,2} - \lVert M_{\mathscr{F},\mathcal{X}} \rVert_{P_N,2}\right| \leq \delta \lVert M_{\mathscr{F},\mathcal{X}} \rVert_{P,2}.
\end{align*}
Since $P_N$ is a finite discrete measure on $\mathcal{X}$, there exists an $\varepsilon \lVert M_{\mathscr{F},\mathcal{X}} \rVert_{P_N}$-net, $\mathscr{G}$, of $\mathscr{F}$ with minimal cardinality such that for all $f \in \mathscr{F}$, there exists $f_0 \in \mathscr{G}$ such that $\lVert f - f_0 \rVert_{P_N,2} \leq \varepsilon \lVert M_{\mathscr{F},\mathcal{X}} \rVert_{P_N,2} \leq \varepsilon(\lVert M_{\mathscr{F},\mathcal{X}} \rVert_{P,2} + \delta \lVert M_{\mathscr{F},\mathcal{X}} \rVert_{P,2}) \leq (1 + \delta)\varepsilon \lVert M_{\mathscr{F},\mathcal{X}} \rVert_{P,2}$. It follows that for all $f \in \mathscr{F}$, there exists $g \in \mathscr{G}$ such that
\begin{align*}
& \lVert f - g \rVert_{P,2} \leq \lVert f - g \rVert_{P_N,2} + \left| \lVert f-g \rVert_{P,2} - \lVert f-g \rVert_{P_N,2}\right| \leq (1 + 2 \delta) \varepsilon \lVert M_{\mathscr{F},\mathcal{X}} \rVert_{P,2},
\end{align*}
Hence, $N(\mathscr{F}, \lVert \cdot \rVert_{P,2}, \varepsilon \lVert M_{\mathscr{F},\mathcal{X}} \rVert_{P,2}) \leq \mathtt{N}_{\mathscr{F},\mathcal{X}}(\varepsilon/(1 + 2 \delta), \mathtt{M}_{\mathscr{F},\mathcal{X}})$. Take $\delta \to 0$ to obtain the desired result.
\end{myproof}
\subsubsection{Meshing Error}\label{sa-sec: MULT -- meshing error}
To simplify notation, the parameters of \(\mathscr{G}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{X}\), and the index \(\mathcal{X}\) is omitted where there is no ambiguity; the parameters of \(\mathscr{R}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{Y}\), and the index \(\mathcal{Y}\) is omitted where there is no ambiguity; and the parameters of \(\mathscr{G} \times \mathscr{R}\) (Definitions 4 to 12, \ref{sa-defn: uniform covering number for product space}, \ref{sa-defn: uniform entropy integral for product space}) are taken with \(\mathcal{C} = \mathcal{X} \times \mathcal{Y}\), and the index \(\mathcal{X} \times \mathcal{Y}\) is omitted where there is no ambiguity. We also define, for $\delta \in (0,1]$,
\begin{align*}
\mathtt{N}(\delta) = \mathtt{N}_{\mathscr{G}}(\delta/\sqrt{2}, \mathtt{M}_{\mathscr{G}}) \mathtt{N}_{\mathscr{R}}(\delta/\sqrt{2},M_{\mathscr{R}})
\end{align*}
and
\begin{align*}
J(\delta) = \sqrt{2}J(\mathscr{G},\mathtt{M}_{\mathscr{G}},\delta/\sqrt{2}) + \sqrt{2}J(\mathscr{R},M_{\mathscr{R}}, \delta/\sqrt{2}).
\end{align*}
For \(0 < \delta \leq 1\), consider a \(\delta \mathtt{M}_{\mathscr{G}} \|M_{\mathscr{R}}\|_{\mathbbm{P}_Y
,2}\)-net of \((\mathscr{G} \times \mathscr{R}, \|\cdot\|_{\mathbbm{P}_Z,2})\), denoted by \((\mathscr{G} \times \mathscr{R})_{\delta}\), with cardinality at most \(\mathtt{N}_{\mathscr{G} \times \mathscr{R}}(\delta, \mathtt{M}_{\mathscr{G}} \|M_{\mathscr{R}}\|_{\mathbbm{P}_Y,2})\). Define the projection onto the \(\delta\)-net as a mapping \(\pi_{(\mathscr{G} \times \mathscr{R})_{\delta}} : \mathscr{G} \times \mathscr{R} \to \mathscr{G} \times \mathscr{R}\) such that \(\|\pi_{(\mathscr{G} \times \mathscr{R})_{\delta}}(g, r) - g r\|_{\mathbbm{P}_Z,2} \leq \delta \mathtt{M}_{\mathscr{G}} \|M_{\mathscr{R}}\|_{\mathbbm{P}_Y,2}\) for all \(g \in \mathscr{G}\) and \(r \in \mathscr{R}\).
\begin{lemma}\label{sa-lem: M- Process -- Fluctuation off the net}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds, a cylindered quasi-dyadic expansion $\mathscr{C}_{M,N}(\mathbbm{P}_Z, \rho)$ is given, $(Z_n^G(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ is the Gaussian process constructed as in Equation~\eqref{sa-eq: M- Process -- Gaussian Process 1} on a possibly enlarged probability space, and $(\mathscr{G} \times \mathscr{R})_\delta$ is chosen in Section~\ref{sa-sec: MULT -- meshing error}. For all $t > 0$ and $0<\delta<1$,
\begin{align*}
\mathbbm{P}\big[\lVert G_n - G_n \circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta} \rVert_{\mathscr{G} \times \mathscr{R}} + \lVert Z_n^G\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta}-Z_n^G \rVert_{\mathscr{G} \times \mathscr{R}}> C_1 c_{\mathtt{v},\alpha} \mathsf{F}_n^G(t,\delta)\big] &\leq 8 \exp(-t),
\end{align*}
where $c_{\mathtt{v},\alpha} = \mathtt{v}(1 + (2 \alpha)^{\frac{\alpha}{2}})$ and
\begin{align*}
\mathsf{F}_n^G(t,\delta) = J(\delta) \mathtt{M}_{\mathscr{G}} + \frac{(\log n)^{\alpha/2} \mathtt{M}_{\mathscr{G}} J^2(\delta)}{\delta^2 \sqrt{n}} + \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} t + (\log n)^{\alpha} \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} t^{\alpha}.
\end{align*}
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: M- Process -- Fluctuation off the net}}
By standard empirical process arguments, we can show for any $0 < \delta < 1$, $\mathtt{N}_{\mathscr{G} \times \mathscr{R}}(\delta,\mathtt{M}_{\mathscr{G}}M_{\mathscr{R}}) \leq \mathtt{N}(\delta)$ and $J(\delta, \mathscr{G} \times \mathscr{R}, \mathtt{M}_{\mathscr{G}} M_{\mathscr{R}}) \leq J(\delta)$. By definition of $\pi_{(\mathscr{G} \times \mathscr{R})_{\delta}}$, $\lVert \pi_{(\mathscr{G} \times \mathscr{R})_{\delta}}h - h \rVert_{\mathbbm{P}_Z,2} \leq \delta \lVert \mathtt{M}_{\mathscr{G}} M_{\mathscr{R}} \rVert_{\mathbbm{P}_Z,2} = \delta \mathtt{M}_{\mathscr{G}} \lVert M_{\mathscr{R}} \rVert_{\mathbbm{P}_{Y},2}$. Take $\mathscr{L} = \{h - \pi_{(\mathscr{G}\times \mathscr{R})_{\delta}}h: h \in \mathscr{G} \times \mathscr{R}\}$. Then, by Theorem 5.2 in \cite{chernozhukov2014gaussian-SA},
\begin{align*}
\mathbbm{E}[\lVert X_n \rVert_{\mathscr{L}}] & \lesssim J(\delta) \mathtt{M}_{\mathscr{G}}\lVert M_{\mathscr{R}}(y_i) \rVert_{2} + \frac{\mathtt{M}_{\mathscr{G}} \lVert \max_{1 \leq i \leq n}M_{\mathscr{R}}(y_i) \rVert_{2} J^2(\delta)}{\delta^2 \sqrt{n}} \\
& \lesssim c_{\mathtt{v},\alpha} J(\delta) \mathtt{M}_{\mathscr{G}} + c_{\mathtt{v},\alpha} (\log n)^{\alpha/2} \frac{\mathtt{M}_{\mathscr{G}} J^2(\delta)}{\delta^2 \sqrt{n}}.
\end{align*}
Moreover, $\lVert \max_{1 \leq i \leq n}\sup_{g \in \mathscr{G}, r \in \mathscr{R}} |g(\mathbf{x}_i)r(y_i)| \rVert_{\psi_{\alpha^{-1}}} \lesssim \mathtt{v} \mathtt{M}_{\mathscr{G}} (\lVert \max_{1 \leq i \leq n}y_i \rVert_{\psi_1})^{\alpha} \lesssim \mathtt{v} \mathtt{M}_{\mathscr{G}} (\log n)^{\alpha}$. Hence, by Theorem 4 in \cite{Adamczak_2008}, for any $t > 0$, with probability at least $1 - 4 \exp(-t)$,
\begin{equation*}
\lVert X_n \rVert_{\mathscr{L}} \lesssim c_{\mathtt{v},\alpha} J(\delta) \mathtt{M}_{\mathscr{G}} + c_{\mathtt{v},\alpha} \frac{\mathtt{M}_{\mathscr{G}} J^2(\delta)}{\delta^2 \sqrt{n}} + c_{\mathtt{v},\alpha} \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} t + c_{\mathtt{v},\alpha} (\log n)^{\alpha} \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} t^{\alpha}.
\end{equation*}
In particular, $\lVert X_n \rVert_{\mathscr{L}} = \lVert G_n - G_n \circ \pi_{(\mathscr{G} \times \mathscr{R})_{\delta}} \rVert_{\mathscr{G} \times \mathscr{R}}$. The bound for $\lVert Z_n^G - Z_n^G \circ \pi_{(\mathscr{G} \times \mathscr{R})_{\delta}} \rVert$ follows from a standard concentration inequality for Gaussian suprema.
\end{myproof}
\subsubsection{Strong Approximation Errors}\label{sa-sec: MULT -- sa error}
To simplify notation, the parameters of \(\mathscr{G}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{X}\), and the index \(\mathcal{X}\) is omitted where there is no ambiguity; the parameters of \(\mathscr{R}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{Y}\), and the index \(\mathcal{Y}\) is omitted where there is no ambiguity; and the parameters of \(\mathscr{G} \times \mathscr{R}\) (Definitions 4 to 12, \ref{sa-defn: uniform covering number for product space}, \ref{sa-defn: uniform entropy integral for product space}) are taken with \(\mathcal{C} = \mathcal{X} \times \mathcal{Y}\), and the index \(\mathcal{X} \times \mathcal{Y}\) is omitted where there is no ambiguity. Recall we also define, for $\delta \in (0,1]$,
\begin{align*}
\mathtt{N}(\delta) = \mathtt{N}_{\mathscr{G}}(\delta/\sqrt{2}, \mathtt{M}_{\mathscr{G}}) \mathtt{N}_{\mathscr{R}}(\delta/\sqrt{2},M_{\mathscr{R}})
\end{align*}
and
\begin{align*}
J(\delta) = \sqrt{2}J(\mathscr{G},\mathtt{M}_{\mathscr{G}},\delta/\sqrt{2}) + \sqrt{2}J(\mathscr{R},M_{\mathscr{R}}, \delta/\sqrt{2}).
\end{align*}
\begin{lemma}\label{sa-lem: M- process -- SA error}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds, a cylindered dyadic expansion $\mathscr{C}_{M,N}(\mathbbm{P}_Z, 1)$ is given, $(\mathtt{\Pi}_{1} Z_n^G(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ is the Gaussian process constructed as in Equation~\eqref{sa-eq: M- Process -- Gaussian Process 2} on a possibly enlarged probability space, and $(\mathscr{G} \times \mathscr{R})_\delta$ is chosen in Section~\ref{sa-sec: MULT -- meshing error}. Then for all $t > 0$,
\begin{align*}
& \mathbbm{P}\Big[\lVert \mathtt{\Pi}_{1} G_n - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
> C_1 c_{\mathtt{v},\alpha} \sqrt{\frac{N^{2\alpha + 1} 2^{M} \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}}{n} t}
+ C_1 c_{\mathtt{v},\alpha}\sqrt{\frac{\mathtt{C}_{\mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R})_{\delta},M+N}}{n}} t \Big]
\leq 2 \mathtt{N}(\delta) e^{-t},
\end{align*}
where $C_1 > 0$ is a universal constant.
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: M- process -- SA error}}
To simplify notation, we will use $\mathbbm{E} [\cdot |\mathcal{X}_{0,l}]$ in short for $\mathbbm{E} [\cdot | \mathbf{x}_i \in \mathcal{X}_{0,l}]$, and $\mathbbm{E}[\cdot|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}]$ in short for $\mathbbm{E}[\cdot|(\mathbf{x}_i, y_i) \in \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}]$. \\
\noindent \textbf{Layers $N + 1 \leq j \leq M + N$:}
For these layers, $\mathcal{C}_{j,k} = \mathcal{X}_{j- N, k} \times \mathcal{Y}_{*,N,0}$. By definition of $\widetilde{\gamma}_{j,k}$,
\begin{align*}
\sum_{N < j \leq M + N} \sum_{0 \leq k < 2^{M + N-j}} \left|\widetilde{\gamma}_{j,k}(g,r)\right|
& \leq \sum_{N \leq j < M + N} \sum_{0 \leq k < 2^{M + N -j}} \mathbbm{E}\big[|g(\mathbf{x}_i) r(y_i)| \big| \mathbf{x}_i \in \mathcal{X}_{j-N,k}\big]\\
& \leq \sum_{N \leq j < M + N} \sum_{0 \leq k < 2^{M + N -j}} \mathbbm{E}\big[|g(\mathbf{x}_i) \mathbbm{E}[r(y_i)|\mathbf{x}_i]| \big| \mathbf{x}_i \in \mathcal{X}_{j-N,k}\big] \\
& \leq c_{\mathtt{v},\alpha} \sum_{N \leq j < M + N} \sum_{0 \leq k < 2^{M + N-j}} \mathbbm{E}[|g(\mathbf{x}_i)\mathbbm{1}(\mathbf{x}_i \in \mathcal{X}_{j-N,k})|] \mathbbm{P}( \mathbf{x}_i \in \mathcal{X}_{j-N,k} )^{-1}\\
& \leq c_{\mathtt{v},\alpha} \sum_{N \leq j < M + N} \mathtt{E}_{\mathscr{G}} 2^{M + N -j} \\
& \leq c_{\mathtt{v},\alpha} 2^{M} \mathtt{E}_{\mathscr{G}},
\end{align*}
where in the third line we have used $\mathbbm{E} [|r(y_i)| | \mathbf{x}_i = \mathbf{x}] \leq c_{\mathtt{v},\alpha} = \mathtt{v}(1 + (2 \alpha)^{\alpha/2})$ for all $\mathbf{x} \in \mathcal{X}$. Moreover, $|\widetilde{\gamma}_{j,k}(g,r)| \leq 2 c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}}$ for all $j \in (N, M + N]$, hence
\begin{align*}
\sum_{N \leq j \leq M + N} \sum_{0 \leq k < 2^{M + N-j}}|\widetilde{\gamma}_{j,k}(g,r)|^2 \leq 2 c_{\mathtt{v},\alpha}^2 2^{M} \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}.
\end{align*}
\textbf{Layers $1 \leq j \leq N$:}
By definition, $\mathcal{C}_{j,k} = \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}$, where $k = 2^{N -j}l + m$, for some unique $l \in [0,2^{M})$ and $m \in [0, 2^{N - j})$. Denote $k = (l,m)$. Fix $j$ and $l$, sum across $m$,
\begin{align*}
& \sum_{m = 0}^{ 2^{N - j}-1} \left|\widetilde{\gamma}_{j,(l,m)}(g, r) \right| = \sum_{m = 0}^{ 2^{N - j}-1} \left|\mathbbm{E} \left[ g(\mathbf{x}_i) \middle | \mathcal{X}_{0,l}\right] \left(\mathbbm{E} \left[r(y_i)\middle| \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,2m}\right] - \mathbbm{E} \left[r(y_i)\middle| \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,2m+1}\right] \right)\right|.
\end{align*}
\underline{\emph{Case 1}}: $\alpha > 0$ in (iv) from Assumption~\ref{sa-assump: MULT & REG -- step}. Then $\sup_{\mathbf{x} \in \mathcal{X}}\mathbbm{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] \leq 2$, and Markov's inequality implies $\min\{|y|: y \in \mathcal{Y}_{l,0,0})\} \leq \log (\mathbbm{E}[\exp(|y_i|)| \mathcal{X}_{0,l} \times \mathcal{Y}_{l,0,0}]) \leq \log(2 \cdot 2^{N}) \leq 2 N$, and similarly $\min \{|y|: y \in \mathcal{Y}_{l, 0, 2^{N}-1}\} \leq 2 N$. Hence the middle cells satisfy $\mathcal{Y}_{l,j,m} \subseteq [-2N,2N]$ for all $0 \leq j < N$, $1 \leq m \leq 2^{N-j}-2$, and
\begin{align*}
\sum_{m = 1}^{ 2^{N - j}-2} \big|\mathbbm{E}[r(y_i) | \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,2m}] - \mathbbm{E}[r(y_i) | \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,2m+1}] \big|
\leq \mathtt{pTV}_{r|_{[-2N, 2 N]}}
\leq c_{\mathtt{v},\alpha} N^{\alpha},
\end{align*}
and for the left-most cells,
\begin{align*}
&\big|\mathbbm{E} \left[r(y_i)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,1} \right] - \mathbbm{E} \left[r(y_i)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,0} \right]\big|\\
&\qquad \leq \max_{0 \leq m < 2^{N-j+1}} \mathbbm{E}[r(y_i)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,m}] - \min_{0 \leq m < 2^{N - j + 1}} \mathbbm{E}[r(y_i)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,m}]
\leq 2 c_{\mathtt{v},\alpha} N^{\alpha},
\end{align*}
and similarly for the right-most cells,
\begin{align*}
\big|\mathbbm{E}[r(y_i)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,2^{N-j}-1}] - \mathbbm{E}[r(y_i)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,2^{N - j}-2}] \big|
\leq 2 c_{\mathtt{v},\alpha} N^{\alpha}.
\end{align*}
\underline{\emph{Case 2}}: $\alpha = 0$ in (iv) from Assumption~\ref{sa-assump: MULT & REG -- step}, since $\mathtt{pTV}_{\{r\}} \leq 2 \mathtt{v}$ and $\mathtt{M}_{\{r\}} \leq 2 \mathtt{v} $ for all $r \in \mathscr{R}$, the above three inequality still hold. It follows that for all $g \in \mathscr{G}, r \in \mathscr{R}$, fix $j,l$ and sum across $m$,
\begin{align*}
\sum_{m = 0}^{ 2^{N - j}-1} \left|\widetilde{\gamma}_{j,(l,m)}(g, r) \right| \leq 2 c_{\mathtt{v},\alpha} N^{\alpha} |\mathbbm{E}[ g(\mathbf{x}_i)| \mathcal{X}_{0,l}]|.
\end{align*}
Fix $j$ and sum the above across $l$,
\begin{align*}
\sum_{0 \leq k < 2^{M + N - j}}\left|\widetilde{\gamma}_{j,(l,m)}(g,r) \right| & = \sum_{l = 0}^{2^{M}-1}\sum_{m = 0}^{ 2^{N - j}-1} \left|\widetilde{\gamma}_{j,(l,m)}(g,r) \right| \\
&\leq 2 c_{\mathtt{v},\alpha} N^{\alpha} \sum_{l = 0}^{2^{M}-1} \mathbbm{E}[|g(\mathbf{x}_i) \mathbbm{1}(\mathbf{x}_i \in\mathcal{X}_{0,l})|] \mathbbm{P}(\mathbf{x}_i \in \mathcal{X}_{0,l})^{-1} \\
& \leq 2 c_{\mathtt{v},\alpha} N^{\alpha} 2^{M} \mathtt{E}_{\mathscr{G}}.
\end{align*}
We can now sum across $j$ to get
\begin{align*}
\sum_{j = 1}^{N}\sum_{0 \leq k < 2^{M + N - j}} \left|\widetilde{\gamma}_{j,k}(g, r) \right| \leq 2 c_{\mathtt{v},\alpha} N^{\alpha + 1} 2^{M} \mathtt{E}_{\mathscr{G}}.
\end{align*}
By Equation~\eqref{sa-eq: proj r bound}, $\sup_{g \in \mathscr{G}, r \in \mathscr{R}} \max_{(j,k) \in \mathcal{I}_{M + N}}\left|\widetilde{\gamma}_{j,k}(g,r)\right| \leq 2 c_{\mathtt{v},\alpha} N^{\alpha} \mathtt{M}_{\mathscr{G}}$, and hence
\begin{align*}
\sum_{1 \leq j \leq N} \sum_{0 \leq k < 2^{M + N -j}}|\widetilde{\gamma}_{j,k}(g,r)|^2 \leq 4 c_{\mathtt{v},\alpha}^2 N^{2 \alpha + 1} 2^{M} \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}, \qquad g \in \mathscr{G}, r \in \mathscr{R}.
\end{align*}
\textbf{Putting Together:}
Putting together the previous two parts,
\begin{align*}
\sum_{j = 1}^{M + N} \sum_{k = 0}^{2^{M + N - j}} \widetilde{\gamma}_{j,k}^2 (g,r)
\leq 6 c_{\mathtt{v},\alpha}^2 N^{2\alpha + 1} 2^{M}\mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}, \qquad g \in \mathscr{G}, r \in \mathscr{R}.
\end{align*}
By Lemma~\ref{sa-lem: M- processes sa for pcw-const}, we know for any $(g,r) \in \mathscr{G} \times \mathscr{R}$, for any $x > 0$, with probability at least $1 - 2 \exp(-x)$,
\begin{align*}
& |G_n \circ \mathtt{\Pi}_{1}(g, r) - \mathtt{\Pi}_{1} Z_n^G (g,r)| \lesssim c_{\mathtt{v},\alpha} \sqrt{\frac{N^{2\alpha + 1} 2^{M} \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}}{n}x} + c_{\mathtt{v},\alpha}\sqrt{\frac{\mathtt{C}_{\mathtt{\Pi}_{1}\{(g,r)\},M+N}}{n}}x,
\end{align*}
and the proof is complete.
\end{myproof}
\begin{lemma}\label{sa-lem: M- process -- SA error quasi-dyadic}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds, a cylindered dyadic expansion $\mathscr{C}_{M,N}(\mathbbm{P}_Z, \rho)$ is given with $\rho > 1$, $(\mathtt{\Pi}_{1} Z_n^G(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ is the Gaussian process constructed as in Equation~\eqref{sa-eq: M- Process -- Gaussian Process 2} on a possibly enlarged probability space, and $(\mathscr{G} \times \mathscr{R})_\delta$ is chosen in Section~\ref{sa-sec: MULT -- meshing error}. Then for all $t > 0$,
\begin{align*}
& \mathbbm{P}\Big[\lVert \mathtt{\Pi}_{1} G_n - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
> C_1 C_{\rho} c_{\mathtt{v},\alpha} \sqrt{\frac{N^{2\alpha + 1} 2^{M} \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}}{n} t}
+ C_1 C_{\rho} c_{\mathtt{v},\alpha}\sqrt{\frac{\mathtt{C}_{\mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R})_{\delta},M+N}}{n}} t \Big] \\
& \quad \leq 2 \mathtt{N}(\delta) e^{-t} + 2^{M}\exp \left(-C_{\rho} n 2^{-M}\right),
\end{align*}
where $C_1 > 0$ is a universal constant, $c_{\mathtt{v},\alpha} = \mathtt{v}(1 + (2 \alpha)^{\alpha/2})$ and $C_{\rho}$ is a constant that only depends on $\rho$.
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: M- process -- SA error quasi-dyadic}}
Since $\mathscr{C}_{M,N}$ is a cylindered quasi-dyadic expansion, $\rho^{-1} 2^{-M-N+j} \leq \mathbbm{P}_Z(\mathcal{C}_{j,k}) \leq \rho 2^{-M-N+j}$, for all $0 \leq j \leq M + N$, $0 \leq k < 2^{M + N - j}$. The same argument for Lemma~\ref{sa-lem: M- process -- SA error} implies
\begin{align*}
\sum_{j = 1}^{M + N} \sum_{k = 0}^{2^{M + N - j}} \widetilde{\gamma}_{j,k}^2 (g,r)
\leq c_{\rho} c_{\mathtt{v},\alpha}^2 N^{2\alpha + 1} 2^{M}\mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}, \qquad g \in \mathscr{G}, r \in \mathscr{R},
\end{align*}
where $c_{\rho}$ is a constant that only depends on $\rho$. The result then follows from Lemma~\ref{sa-lem: M- processes sa for pcw-const quasi}.
\end{myproof}
\subsubsection{Projection Error}\label{sa-sec: MULT -- proj error}
To simplify notation, the parameters of \(\mathscr{G}\) (Definitions 4 to 12, \ref{sa-defn: smooth tv}, \ref{sa-defn: smooth ktv}) are taken with \(\mathcal{C} = \mathcal{X}\), and the index \(\mathcal{X}\) is omitted where there is no ambiguity; the parameters of \(\mathscr{R}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{Y}\), and the index \(\mathcal{Y}\) is omitted where there is no ambiguity; and the parameters of \(\mathscr{G} \times \mathscr{R}\) (Definitions 4 to 12, \ref{sa-defn: uniform covering number for product space}, \ref{sa-defn: uniform entropy integral for product space}) are taken with \(\mathcal{C} = \mathcal{X} \times \mathcal{Y}\), and the index \(\mathcal{X} \times \mathcal{Y}\) is omitted where there is no ambiguity. Recall we also define, for $\delta \in (0,1]$,
\begin{align*}
\mathtt{N}(\delta) = \mathtt{N}_{\mathscr{G}}(\delta/\sqrt{2}, \mathtt{M}_{\mathscr{G}}) \mathtt{N}_{\mathscr{R}}(\delta/\sqrt{2},M_{\mathscr{R}})
\end{align*}
and
\begin{align*}
J(\delta) = \sqrt{2}J(\mathscr{G},\mathtt{M}_{\mathscr{G}},\delta/\sqrt{2}) + \sqrt{2}J(\mathscr{R},M_{\mathscr{R}}, \delta/\sqrt{2}).
\end{align*}
To analyze the projection error, we employ the decomposition
\begin{align*}
\mathtt{\Pi}_{1} G_n(g,r) - G_n(g,r) &= \big(\mathtt{\Pi}_{0} G_n(g,r) - G_n(g,r) \big) + \big(\mathtt{\Pi}_{1} G_n(g,r) - \mathtt{\Pi}_{0} G_n(g,r) \big),
\end{align*}
where $\mathtt{\Pi}_{0} G_n(g,r) - G_n(g,r)$ represents the $L_2$ projection error, and $\mathtt{\Pi}_{1} G_n(g,r) - \mathtt{\Pi}_{0} G_n(g,r)$ denotes the mis-specification error. Specifically, the $L_2$ projection error captures the minimum loss incurred by projecting onto the class of piecewise constant functions over the cells $\mathscr{E}_{M + N}$. In contrast, the mis-specification error reflects the additional loss introduced when shifting from the $L_2$ projection to the product-factorized projection.
First we bound the mis-specification error.
\begin{lemma}\label{sa-lem: M- process -- Misspecification Error Moment}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds, a cylindered dyadic expansion $\mathscr{C}_{M,N}(\mathbbm{P}_Z, 1)$ is given, $(\mathtt{\Pi}_{0} Z_n^G(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ and $(\mathtt{\Pi}_{1} Z_n^G(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ are the Gaussian processes constructed as in Equation~\eqref{sa-eq: M- Process -- Gaussian Process 2} on a possibly enlarged probability space, and $(\mathscr{G} \times \mathscr{R})_\delta$ is chosen in Section~\ref{sa-sec: MULT -- meshing error}. Suppose $\mathbbm{P}_X$ admits a Lebesgue density $f_X$ supported on $\mathcal{X} \subseteq \mathbb{R}^d$. Let $\tau > 0$. Define $r_{\tau} = r \mathbbm{1}([-\tau^{\frac{1}{\alpha}}, \tau^{\frac{1}{\alpha}}])$. Then, for any $g \in \mathscr{G}, r \in \mathscr{R}$,
\begin{align*}
\mathbbm{E} \left[\left(\mathtt{\Pi}_{1} G_n(g, r_{\tau}) - \mathtt{\Pi}_{0} G_n(g, r_{\tau}) \right)^2\right] = \mathbbm{E} \left[\left(\mathtt{\Pi}_{1} Z_n^G(g, r_{\tau}) - \mathtt{\Pi}_{0} Z_n^G(g, r_{\tau}) \right)^2\right] \leq 4 \mathtt{v}^2 (1 + \tau)^2 N^2 \mathtt{V}_{\mathscr{G}},
\end{align*}
where
\begin{align*}
\mathtt{V}_{\mathscr{G}} = \min\{2 \mathtt{M}_{\mathscr{G}}, \mathtt{L}_{\mathscr{G}}\lVert \mathcal{V}_M \rVert_{\infty}\} \Big(\sup_{\mathbf{x} \in \mathcal{X}} f_X(\mathbf{x})\Big)^2 2^M \mathfrak{m}(\mathcal{V}_M) \lVert \mathcal{V}_M \rVert_{\infty} \mathtt{TV}_{\mathscr{G}}^*,
\end{align*}
and, as in Section~\ref{sa-sec: X-process -- proj error}, $\mathcal{V}_M = \cup_{0 \leq l < 2^{M}} (\mathcal{X}_{0,l} - \mathcal{X}_{0,l})$ is the upper level quasi-dyadic variation set.
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: M- process -- Misspecification Error Moment}}
To simplify notation, we will use $\mathbbm{E} [\cdot |\mathcal{X}_{0,l}]$ in short for $\mathbbm{E} [\cdot | \mathbf{x}_i \in \mathcal{X}_{0,l}]$, and $\mathbbm{E}[\cdot|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}]$ in short for $\mathbbm{E}[\cdot|(\mathbf{x}_i, y_i) \in \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}]$ in this proof.
Expanding $\mathtt{\Pi}_{1} G_n(g,r_{\tau}) - \mathtt{\Pi}_{0} G_n(g, r_{\tau})$ by Haar basis representation,
\begin{align*}
& \mathtt{\Pi}_{1} G_n(g,r_{\tau}) - \mathtt{\Pi}_{0} G_n(g, r_{\tau}) = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \Delta_i(g, r_{\tau}), \\
& \Delta_i(g,r_{\tau}) = \sum_{1 \leq j \leq N} \sum_{0 \leq k < 2^{M + N -j}} \left(\widetilde{\gamma}_{j,k}(g, r_{\tau}) - \widetilde{\beta}_{j,k}(g, r_{\tau})\right) \widetilde{e}_{j,k}(\mathbf{x}_i, y_i),
\end{align*}
where we have used $\widetilde{\gamma}_{j,k}(g,r_{\tau}) = \widetilde{\beta}_{j,k}(g,r_{\tau})$ for $j > N$. Moreover,
\begin{align*}
& \mathbbm{E}[|\Delta_i(g,r_{\tau})|]
\leq 2 \sum_{0 \leq j < N} \sum_{0 \leq k < 2^{M + N -j}} \left|\gamma_{j,k}(g,r) - \beta_{j,k}(g,r)\right| \mathbbm{P}((\mathbf{x}_i, y_i) \in \mathcal{C}_{j,k}).
\end{align*}
Recall in Definition~\ref{sa-defn: cylindered quasi dyadic expansion}, $\mathcal{C}_{j,k} = \mathcal{X}_{j - N, l} \times \mathcal{Y}_{l,j,m}$, where $k = 2^{N-j} l + m$, $0 \leq l < 2^M$ and $0 \leq m < 2^{N-j}$. Definitions of $\gamma_{j,k}$ and $\beta_{j,k}$ from Section~\ref{sa-sec: MULT -- proj} imply
\begin{align*}
\left|\gamma_{j,k}(g, r_{\tau}) - \beta_{j,k}(g, r_{\tau})\right|
& = \left|\mathbbm{E} \left[ g(\mathbf{x}_i) \middle | \mathcal{X}_{0,l}\right] \cdot \mathbbm{E} \left[r_{\tau}(y_i)\middle| \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}\right] - \mathbbm{E} \left[g(\mathbf{x}_i) r_{\tau}(y_i) \middle | \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}\right] \right| \\
& = \left|\mathbbm{E} \left[\left(g(\mathbf{x}_i) - \mathbbm{E} \left[ g(\mathbf{x}_i) \middle | \mathcal{X}_{0,l}\right] \right) r_{\tau}(y_i) \middle | \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}\right] \right|\\
& \leq \mathtt{v} (1 + \tau) \mathbbm{E} \left[\left| g(\mathbf{x}_i) - \mathbbm{E} \left[ g(\mathbf{x}_i) \middle | \mathcal{X}_{0,l}\right] \right| \middle |\mathcal{C}_{j,k}\right],
\end{align*}
where the first line is simply the definitions of $\gamma_{j,k}$ and $\beta_{j,k}$; the second line is because $\boldsymbol{\sigma}
(\mathbbm{1}(\mathbf{x}_i \in \mathcal{X}_{0,l})) \subseteq \boldsymbol{\sigma}(\mathbbm{1}((\mathbf{x}_i, y_i) \in \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}))$; and the third line is because Assumption~\ref{sa-assump: MULT & REG -- step} (iv) implies $\sup_{y \in \mathbb{R}}|r_{\tau}(y)| \leq \mathtt{v}(1 + \tau)$ for all $r \in \mathscr{R}$.
Summing across $j$ and $k$, then by similar argument as in the proof of Lemma~\ref{sa-lem: X-process -- projection error},
\begin{align*}
\mathbbm{E} \left[\left|\Delta_i(g, r_{\tau})\right|\right]
& \leq 2 \mathtt{v}(1 + \tau) N \mathbbm{E} \left[\left|g(\mathbf{x}_i) - \mathtt{\Pi}_{0}(\mathtt{p}_X[\mathscr{C}_{M,N}(\mathbbm{P}, \rho)])g(\mathbf{x}_i) \right|\right] \\
& \leq 2 \mathtt{v}(1 + \tau) N \Big(\sup_{\mathbf{x} \in \mathcal{X}} f_X(\mathbf{x})\Big)^2 2^M \mathfrak{m}(\mathcal{V}_M) \lVert \mathcal{V}_M \rVert_{\infty} \mathtt{TV}_{\{g\}}^*.
\end{align*}
For each fixed $j$, $\widetilde{e}_{j,k}(\mathbf{x},y)$ can be non-zero for only one $k$. Hence, almost surely,
\begin{align*}
|\Delta_i(g, r_{\tau})|
& = \big|\sum_{j = 1}^{N} \sum_{0 \leq k < 2^{M + N -j}} (\widetilde{\gamma}_{j,k}(g,r_{\tau}) - \widetilde{\beta}_{j,k}(g, r_{\tau})) \widetilde{e}_{j,k}(\mathbf{x}_i, y_i)\big| \\
& \leq \sum_{j = 1}^{N} \max_{0 \leq k < 2^{M + N - j}} \left|\widetilde{\gamma}_{j,k}(g,r_{\tau}) - \widetilde{\beta}_{j,k}(g, r_{\tau}) \right| \\
& \leq 2 \sum_{j = 0}^{N-1} \max_{0 \leq k < 2^{M + N - j}} \left|\gamma_{j,k}(g, r_{\tau}) - \beta_{j,k}(g, r_{\tau}) \right| \\
& \leq 2 \mathtt{v}(1 + \tau) \sum_{j = 0}^{N - 1} \max_{0 \leq k < 2^{M + N -j}} \left|\mathbbm{E} \left[\left| g(\mathbf{x}_i) - \mathbbm{E} \left[ g(\mathbf{x}_i) \middle | \mathcal{X}_{0,l}\right] \right| \middle | \mathcal{C}_{j,k}\right] \right| \\
& \leq 2 N \mathtt{v}(1 + \tau) \min\{2 \mathtt{M}_{\mathscr{G}}, \mathtt{L}_{\mathscr{G}}\lVert \mathcal{V}_M \rVert_{\infty}\}.
\end{align*}
This shows the results.
\end{myproof}
Next we bound the $L_2$ projection error.
\begin{lemma}\label{sa-lem: M-Process -- L2 Error Moment}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds, a cylindered dyadic expansion $\mathscr{C}_{M,N}(\mathbbm{P}_Z, 1)$ is given, $(Z_n^G(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ and $(\mathtt{\Pi}_{0} Z_n^G(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ are the Gaussian processes constructed as in Equations~\eqref{sa-eq: M- Process -- Gaussian Process 1} and \eqref{sa-eq: M- Process -- Gaussian Process 2} on a possibly enlarged probability space, and $(\mathscr{G} \times \mathscr{R})_\delta$ is chosen in Section~\ref{sa-sec: MULT -- meshing error}. Suppose $\mathbbm{P}_X$ admits a Lebesgue density $f_X$ supported on $\mathcal{X} \subseteq \mathbb{R}^d$. Let $\tau > 0$. Define $r_{\tau} = r \mathbbm{1}([-\tau^{\frac{1}{\alpha}}, \tau^{\frac{1}{\alpha}}])$. Then for any $g \in \mathscr{G}, r \in \mathscr{R}$,
\begin{align*}
& \mathbbm{E} \left[\left(\mathtt{\Pi}_{0} Z_n^G(g, r_{\tau}) - Z_n^G(g, r_{\tau}) \right)^2\right] = \mathbbm{E} \left[\left(\mathtt{\Pi}_{0} G_n(g, r_{\tau}) - G_n(g, r_{\tau}) \right)^2\right] \leq 4 \mathtt{v}^2 (1 + \tau)^2 \left(2^{-N} \mathtt{M}_{\mathscr{G}}^2 + \mathtt{V}_{\mathscr{G}}\right),
\end{align*}
where $\mathtt{V}_{\mathscr{G}}$ is defined in Lemma~\ref{sa-lem: M- process -- Misspecification Error Moment}.
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: M-Process -- L2 Error Moment}}
To simplify notation, we will use $\mathbbm{E} [\cdot |\mathcal{X}_{0,l}]$ in short for $\mathbbm{E} [\cdot | \mathbf{x}_i \in \mathcal{X}_{0,l}]$, and $\mathbbm{E}[\cdot|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}]$ in short for $\mathbbm{E}[\cdot|(\mathbf{x}_i, y_i) \in \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}]$ in this proof.
Let $\mathcal{B} = \boldsymbol{\sigma}(\{\mathbbm{1}((\mathbf{x}_i,y_i) \in \mathcal{C}_{0,k}): 0 \leq k < 2^{M + N}\})$ be the $\sigma$-algebra generated by $\{\mathbbm{1}((\mathbf{x}_i, y_i) \in \mathcal{C}_{0,k}): 0 \leq k < 2^{M + N}\}$. Then the difference between the $L_2$ projection and the original can be expressed as
\begin{align*}
\mathtt{\Pi}_{0}(g \cdot r_{\tau})(\mathbf{x}_i, y_i) - g(\mathbf{x}_i) r_{\tau}(y_i)
& =
\mathbbm{E}[g(\mathbf{x}_i)r_{\tau}(y_i)|\mathcal{B}] - g(\mathbf{x}_i)r_{\tau}(y_i) \\
& = \mathbbm{E}[g(\mathbf{x}_i)r_{\tau}(y_i)|\mathcal{B}] - \mathbbm{E}[g(\mathbf{x}_i)|\mathcal{B}]r_{\tau}(y_i)
+ \mathbbm{E}[g(\mathbf{x}_i)|\mathcal{B}]r_{\tau}(y_i) - g(\mathbf{x}_i)r_{\tau}(y_i).
\end{align*}
By Definition~\ref{sa-defn: cylindered quasi dyadic expansion}, each cell $\mathcal{C}_{0,k}$ is of the form of a product, that is,
\begin{align*}
\mathcal{C}_{0,k} = \mathcal{X}_{0,l} \times \mathcal{Y}_{l,0,m} \text{ with } k = 2^N l + m,
\end{align*}
where $0 \leq k < 2^{M + N}$, $0 \leq l < 2^M$ and $0 \leq m < 2^N$.
The first two terms $\mathbbm{E}[g(\mathbf{x}_i)r_{\tau}(y_i)|\mathcal{B}] - \mathbbm{E}[g(\mathbf{x}_i)|\mathcal{B}]r_{\tau}(y_i)$ in the decomposition are driven by projection of $r_{\tau}$ on grids $\mathcal{Y}_{l,0,m}$'s, and can be upper bounded through probability measure assigned to each grid ($2^{-N}$) and total variation of $r_{\tau}$. We consider the positive and negative parts separately: Consider the function
\begin{align*}
q_{l,m}^+(y) = \mathbbm{E}[g(\mathbf{x}_i)\mathbbm{1}(g(\mathbf{x}_i) \geq 0)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,0,m}]r_{\tau}(y) - \mathbbm{E}[g(\mathbf{x}_i)r_{\tau}(y_i)\mathbbm{1}(g(\mathbf{x}_i) \geq 0)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,0,m}], \quad y \in \mathcal{Y}_{l,0,m}.
\end{align*}
Either $q_{l,m}^+$ is constantly zero on $\mathcal{Y}_{l,0,m}$ or $q_{l,m}^+$ takes both positive and negative values on $\mathcal{Y}_{l,0,m}$. Under either case, we have $|q_{l,m}^+(y)| \leq \mathtt{pTV}_{\{q_{l,m}^+\},\mathcal{Y}_{l,0,m}}$ for all $y \in \mathcal{Y}_{l,0,m}$. Hence
\begin{align*}
\mathbbm{E}[|q_{l,m}^+(y_i)| \mathbbm{1}(y_i \in \mathcal{Y}_{l,0,m})|\mathbf{x}_i = \mathbf{x}]
& = \int_{\mathcal{Y}_{l,0,m}} |q_{l,m}^+(y)| d \mathbbm{P}(y_i \leq y|\mathbf{x}_i = \mathbf{x}) \\
& \leq \mathbbm{P}(y_i \in \mathcal{Y}_{l,0,m} | \mathbf{x}_i = \mathbf{x}) \mathtt{pTV}_{\{q_{l,m}^+\},\mathcal{Y}_{l,0,m}} \\
& \leq \mathbbm{P}(y_i \in \mathcal{Y}_{l,0,m} | \mathbf{x}_i = \mathbf{x})\mathtt{M}_{\{g\}}\mathtt{pTV}_{\{r_{\tau}\},\mathcal{Y}_{l,0,m}}, \qquad \mathbf{x} \in \mathcal{X}_{0,l}.
\end{align*}
Similarly,
\begin{align*}
q_{l,m}^-(y) = \mathbbm{E}[g(\mathbf{x}_i)\mathbbm{1}(g(\mathbf{x}_i) < 0)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,0,m}]r_{\tau}(y) - \mathbbm{E}[g(\mathbf{x}_i)r_{\tau}(y_i)\mathbbm{1}(g(\mathbf{x}_i) < 0)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,0,m}], \quad y \in \mathcal{Y}_{l,0,m},
\end{align*}
and we have
\begin{align*}
\mathbbm{E}[|q_{l,m}^-(y_i)| \mathbbm{1}(y_i \in \mathcal{Y}_{l,0,m})|\mathbf{x}_i = \mathbf{x}] \leq \mathbbm{P}(y_i \in \mathcal{Y}_{l,0,m} | \mathbf{x}_i = \mathbf{x})\mathtt{M}_{\{g\}}\mathtt{pTV}_{\{r_{\tau}\},\mathcal{Y}_{l,0,m}}, \qquad \mathbf{x} \in \mathcal{X}_{0,l}.
\end{align*}
Combining the two parts, and integrate over the event $\mathbf{x}_i \in \mathcal{X}_{0,l}$,
\begin{align*}
& \mathbbm{E} \Big[\big|\mathbbm{E}[g(\mathbf{x}_i)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,0,m}]r_{\tau}(y_i) - \mathbbm{E}[g(\mathbf{x}_i)r_{\tau}(y_i)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,0,m}]\big| \mathbbm{1}(y_i \in \mathcal{Y}_{l,0,m}) \Big| \mathbf{x}_i \in \mathcal{X}_{0,l}\Big] \\
& \qquad \leq 2 \mathbbm{P}(y_i \in \mathcal{Y}_{l,0,m} | \mathbf{x}_i \in \mathcal{X}_{0,l})\mathtt{M}_{\{g\}}\mathtt{pTV}_{\{r_{\tau}\},\mathcal{Y}_{l,0,m}}
\leq 2 \cdot 2^{-N} \mathtt{M}_{\{g\}}\mathtt{pTV}_{\{r_{\tau}\},\mathcal{Y}_{l,0,m}}.
\end{align*}
Summing over $m$, we get for each $0 \leq l < 2^M$,
\begin{align*}
\mathbbm{E}\big[|\mathbbm{E}[g(\mathbf{x}_i)|\mathcal{B}]r_{\tau}(y_i) - \mathbbm{E}[g(\mathbf{x}_i)r_{\tau}(y_i)|\mathcal{B}]| \big| \mathbf{x}_i \in \mathcal{X}_{0,l}\big]
\leq 2 \cdot 2^{-N} \mathtt{M}_{\{g\}} \mathtt{pTV}_{\{r_{\tau}\},\mathcal{Y}_{*,N,0}}.
\end{align*}
Hence, using the polynomial growth of total variation,
\begin{align*}
\mathbbm{E} \big[|\mathbbm{E}[g(\mathbf{x}_i)|\mathcal{B}]r_{\tau}(y_i) - \mathbbm{E}[g(\mathbf{x}_i)r_{\tau}(y_i)|\mathcal{B}] | \big] \leq 2^{-N} \mathtt{M}_{\{g\}} \mathtt{pTV}_{\{r_{\tau}\},\mathcal{Y}_{*,N,0}} \leq 2 \cdot 2^{-N} \mathtt{M}_{\mathscr{G}}\mathtt{v}(1 + \tau).
\end{align*}
Since $\big|\mathbbm{E}[g(\mathbf{x}_i)r_{\tau}(y_i)|\mathcal{B}] - \mathbbm{E}[g(\mathbf{x}_i)|\mathcal{B}]r_{\tau}(y_i)\big| \leq 2 \mathtt{M}_{\mathscr{G}} \mathtt{v} (1 + \tau)$ almost surely,
\begin{align*}
\mathbbm{E} \big[(\mathbbm{E}[g(\mathbf{x}_i)|\mathcal{B}]r(y_i) - \mathbbm{E}[g(\mathbf{x}_i)r(y_i)|\mathcal{B}] )^2 \big] \leq 4 \cdot 2^{-N} \mathtt{v}^2 (1 + \tau)^2 \mathtt{M}_{\mathscr{G}}^2.
\end{align*}
Now we look at the last two terms $\mathbbm{E}[g(\mathbf{x}_i)|\mathcal{B}]r_{\tau}(y_i) - g(\mathbf{x}_i)r_{\tau}(y_i)$, which are essentially driven by the $L_2$-projection error of $g$. Denote by $\mathcal{A} = \boldsymbol{\sigma}(\{\mathbbm{1}(\mathbf{x}_i \in \mathcal{X}_{0,l}): 0 \leq l < 2^M\})$ the $\sigma$-algebra generated by $\{\mathbbm{1}(\mathbf{x}_i \in \mathcal{X}_{0,l}): 0 \leq l < 2^M\}$. Then $\mathcal{A} \subseteq \mathcal{B}$. By Jensen's inequality and a similar argument as in the proof of Lemma~\ref{sa-lem: X-process -- projection error},
\begin{align*}
\mathbbm{E} \big[(\mathbbm{E}[g(\mathbf{x}_i)|\mathcal{B}] r_{\tau}(y_i) - g(\mathbf{x}_i)r_{\tau}(y_i) )^2\big]
\leq 4 \mathtt{v}^2(1+\tau)^2 \mathbbm{E} \big[(g(\mathbf{x}_i) - \mathbbm{E}[g(\mathbf{x}_i)|\mathcal{A}])^2\big] \leq 4 \mathtt{v}^2 (1 + \tau)^2\mathtt{V}_{\mathscr{G}}.
\end{align*}
It then follows that $\mathbbm{E}[\left(\mathtt{\Pi}_{0} G_n(g, r_{\tau}) - G_n(g, r_{\tau}) \right)^2] \leq 4 \mathtt{v}^2 (1 + \tau)^2 \left(2^{-N} \mathtt{M}_{\mathscr{G}}^2 + \mathtt{V}_{\mathscr{G}}\right)$.
\end{myproof}
Using a truncation argument and the previous two lemmas, we get the bound on $\mathtt{\Pi}_{1}$-projection error with tail control.
\begin{lemma}\label{sa-lem: M- process -- projection error}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds, a cylindered dyadic expansion $\mathscr{C}_{M,N}(\mathbbm{P}_Z, 1)$ is given, $(Z_n^G(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ and $(\mathtt{\Pi}_{1} Z_n^G(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ are the Gaussian processes constructed as in Equations~\eqref{sa-eq: M- Process -- Gaussian Process 1} and \eqref{sa-eq: M- Process -- Gaussian Process 2} on a possibly enlarged probability space, and $(\mathscr{G} \times \mathscr{R})_\delta$ is chosen in Section~\ref{sa-sec: MULT -- meshing error}. Suppose $\mathbbm{P}_X$ admits a Lebesgue density $f_X$ supported on $\mathcal{X} \subseteq \mathbb{R}^d$. Then for all $t > N$,
\begin{gather*}
\mathbbm{P}\Big[\lVert G_n - \mathtt{\Pi}_{1} G_n \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
> C_1 \sqrt{c_{\mathtt{v}, 2 \alpha}}\sqrt{N^2 \mathtt{V}_{\mathscr{G}} + 2^{-N} \mathtt{M}_{\mathscr{G}}^2} t^{\alpha + \frac{1}{2}} + C_1 c_{\mathtt{v},\alpha} \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}}t^{\alpha + 1}\Big]
\leq 4 \mathtt{N}(\delta) n e^{-t},\\
\mathbbm{P}\Big[\lVert Z_n^G - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
> C_1 \sqrt{c_{\mathtt{v}, 2 \alpha}} \sqrt{N^2 \mathtt{V}_{\mathscr{G}} + 2^{-N} \mathtt{M}_{\mathscr{G}}^2} t^{\frac{1}{2}} + C_1 c_{\mathtt{v},\alpha} \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}}t \Big]
\leq 4 \mathtt{N}(\delta) n e^{-t},
\end{gather*}
where $c_{\mathtt{v},\alpha} = \mathtt{v}(1 + (2 \alpha)^{\frac{\alpha}{2}})$, $c_{\mathtt{v}, 2 \alpha} = \mathtt{v}^2(1 + (4 \alpha)^{\alpha})$, and $C_1$ is a universal constant.
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: M- process -- projection error}}
To simplify notation, we will use $\mathbbm{E} [\cdot |\mathcal{X}_{0,l}]$ in short for $\mathbbm{E} [\cdot | \mathbf{x}_i \in \mathcal{X}_{0,l}]$, and $\mathbbm{E}[\cdot|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}]$ in short for $\mathbbm{E}[\cdot|(\mathbf{x}_i, y_i) \in \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}]$ in this proof. We will use a truncation argument and consider the cases of whether $\alpha > 0$ in (iv) of Assumption~\ref{sa-assump: MULT & REG -- step} separately.
First, suppose $\alpha > 0$ in (iv) of Assumption~\ref{sa-assump: MULT & REG -- step}. Let $\tau > 0$ such that $\tau^{\frac{1}{\alpha}} > \log(2^{N+1})$.
\underline{Projection error for truncated processes}: By Lemmas~\ref{sa-lem: M- process -- Misspecification Error Moment} and \ref{sa-lem: M-Process -- L2 Error Moment}, and using Bernstein inequality, for all $t > 0$, for each $g \in \mathscr{G}$, $r \in \mathscr{R}$,
\begin{align*}
\mathbbm{P} \left[ \left|G_n(g, r_{\tau}) - \mathtt{\Pi}_{1} G_n(g, r_{\tau})\right| \geq 4\mathtt{v} (1 + \tau)\sqrt{N^2 \mathtt{V}_{\mathscr{G}} + 2^{-N} \mathtt{M}_{\mathscr{G}}^2}\sqrt{t} + \frac{4}{3}\mathtt{v} (1 + \tau)\frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}}t\right] \leq 2 e^{-t}.
\end{align*}
\underline{Truncation Error}: Recall Equation~\eqref{sa-eq: proj r bound} implies $\max_{0 \leq k < 2^{M + N}}\mathbbm{E}[|r(y_i)| | (\mathbf{x}_i, \mathbf{y}_i) \in \mathcal{C}_{0,k}] \leq c_{\mathtt{v},\alpha} N^{\alpha}$. The same argument implies $\max_{0 \leq k < 2^{M + N}}\mathbbm{E}[r(y_i)^2|(\mathbf{x}_i,y_i) \in \mathcal{C}_{0,k}] \leq \mathtt{v}^2(1 + (N \log(2) \sqrt{2 \alpha})^{2 \alpha}) \leq c_{\mathtt{v}, 2 \alpha} N^{2 \alpha}$. Hence the following holds almost surely,
\begin{align*}
\left|\mathtt{\Pi}_{1} G_n(g,r) - \mathtt{\Pi}_{1} G_n(g, r_{\tau})\right|
& \leq \max_{0 \leq l < 2^M} \max_{0 \leq m < 2^N} \left|\mathbbm{E}[ g(\mathbf{x}_i) | \mathcal{X}_{0,l}] \mathbbm{E} \big[|r(y_i)| \mathbbm{1}(|y_i| \geq \tau^{1/\alpha}) \big| \mathcal{X}_{0,l} \times \mathcal{Y}_{l,0,m}\big]\right| \\
& \leq c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}} N^{\alpha}.
\end{align*}
Since $\tau^{\frac{1}{\alpha}} > \log(2^{N+1}) > 0.5 N$, $\gamma_{0,k} = \beta_{0,k}$ for all $k$ corresponding to $\mathcal{X}_{0,l} \times \mathcal{Y}_{l,0,m}$ for $0<m<2^{N}-1$, that is, the mismatch only happens at edge cells of $y_i$, we have
\begin{align*}
\mathbbm{E} \Big[ \big|\mathtt{\Pi}_{1} G_n(g, r) - \mathtt{\Pi}_{1} G_n(g, r_{\tau})\big|^2 \Big]
& \leq \mathbbm{P} \big(\mathtt{\Pi}_{1} G_n(g, r) - \mathtt{\Pi}_{1} G_n(g, r_{\tau}) \neq 0\big) c_{\mathtt{v}, 2 \alpha}\mathtt{M}_{\mathscr{G}}^2 N^{2 \alpha} \\
& \leq c_{\mathtt{v}, 2 \alpha} 2^{-N+1} \mathtt{M}_{\mathscr{G}}^2 N^{2\alpha}.
\end{align*}
Using Bernstein's inequality, for all $t > 0$, with probability at least $1 - 2 \exp(-t)$,
\begin{align*}
\nonumber |\mathtt{\Pi}_{1} G_n(g,r) - \mathtt{\Pi}_{1} G_n(g, r_{\tau})| & \lesssim \sqrt{c_{\mathtt{v}, 2 \alpha}} 2^{-N/2} \mathtt{M}_{\mathscr{G}} N^{\alpha}\sqrt{t} + c_{\mathtt{v},\alpha} \frac{\mathtt{M}_{\mathscr{G}} N^{\alpha}}{\sqrt{n}}t \\
& \lesssim \sqrt{c_{\mathtt{v}, 2 \alpha}} 2^{-N/2} \mathtt{M}_{\mathscr{G}} N^{\alpha}\sqrt{t} + c_{\mathtt{v},\alpha} \frac{\mathtt{M}_{\mathscr{G}} \tau}{\sqrt{n}}t.
\end{align*}
Moreover, using $\mathbbm{P}(|y_i| \geq \tau) \leq 2 \cdot 2^{-N}$, we have
\begin{align*}
\mathbbm{E}[(G_n(g,r) - G_n(g, r_{\tau}))^2] & \leq \mathtt{M}_{\mathscr{G}}^2 \mathbbm{E}[(r(y_i) - r_{\tau}(y_i))^2] \leq \mathtt{M}_{\mathscr{G}}^2 \mathbbm{E}[r(y_i)^2 \mathbbm{1}(|y_i| \geq \tau)] \\
& \leq 2 \cdot 2^{-N} \mathtt{M}_{\mathscr{G}}^2 \max_{0 \leq k < 2^{M + N}} \mathbbm{E}[r(y_i)^2|(\mathbf{x}_i, y_i) \in \mathcal{C}_{0,k}]
\leq 2 c_{\mathtt{v}, 2 \alpha}\mathtt{M}_{\mathscr{G}}^2 N^{2\alpha}2^{-N}.
\end{align*}
By Bernstein inequality and a truncation argument, for all $t > 0$,
\begin{align*}
& \mathbbm{P}(\sqrt{n}|G_n(g,r) - G_n(g, r_{\tau})| \geq t) \\ &\leq \min_{y > 0} \bigg\{ 2 \exp \left(-\frac{t^2}{2 n \mathbbm{V}[G_n(g,r) - G_n(g, r_{\tau})] + \frac{2}{3}xy}\right) + 2 \mathbbm{P}\left(\max_{1 \leq i \leq n}|g(\mathbf{x}_i)(r(y_i) - r_{\tau}(y_i))| \geq y\right)\bigg\}.
\end{align*}
Taking $y = \mathtt{M}_{\mathscr{G}} t^{\alpha}$, we get for all $t > 0$, with probability at least $1 - 4 \exp(-t)$,
\begin{equation*}
|G_n(g,r) - G_n(g,r_{\tau})| \lesssim \sqrt{c_{\mathtt{v}, 2 \alpha}} 2^{-N/2} \mathtt{M}_{\mathscr{G}} N^{\alpha} \sqrt{t} + C_{\mathtt{v}, \alpha} \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} t^{\alpha + 1}.
\end{equation*}
\underline{Putting Together}:
Taking $\tau = t^{\alpha} > 0.5^{\alpha}N^{\alpha}$, we get from the previous bounds on $G_n(g,r_{\tau}) - \mathtt{\Pi}_{1} G_n(g,r_{\tau})$, $\mathtt{\Pi}_{1} G_n(g,r) - \mathtt{\Pi}_{1} G_n(g,r_{\tau})$, and $G_n(g,r) - G_n(g,r_{\tau})$ that for all $g \in \mathscr{G}$, $r \in \mathscr{G}$, for all $t > N$, with probability at least $1 - 4 n\exp(-t)$,
\begin{align}\label{sa-eq: M- and R- processes -- proj error unbdd part4}
|{\mathtt{\Pi}_{1} G_n(g,r) - G_n(g,r)}|
\lesssim \sqrt{c_{\mathtt{v}, 2 \alpha}}\sqrt{N^2 \mathtt{V}_{\mathscr{G}} + 2^{-N} \mathtt{M}_{\mathscr{G}}^2} t^{\alpha + \frac{1}{2}} + c_{\mathtt{v},\alpha} \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}}t^{\alpha + 1}.
\end{align}
The bound for $|\mathtt{\Pi}_{1} Z_n^G(g, r) - Z_n^G(g, r)|$ follows from the fact that it is a mean-zero Gaussian random variable with variance equal to $\mathbbm{V}[\mathtt{\Pi}_{1} G_n(g,r) - G_n(g,r)]$. The result follows then follows from a union bound over $(g,r) \in (\mathscr{G} \times \mathscr{R})_{\delta}$. \\
Next, suppose $\alpha = 0$ in (iv) of Assumption~\ref{sa-assump: MULT & REG -- step}. This implies $\mathtt{M}_{\mathscr{R}} \leq 2 \mathtt{v}$. Hence choosing $\tau = 2 \mathtt{v}$, then $G_n(g,r) = G_n(g,r_{\tau})$ almost surely for all $g \in \mathscr{G}$, $r \in \mathscr{R}$, that is, there is no truncation error. Hence the bound on $G_n(g,r_{\tau}) - \mathtt{\Pi}_{1} G_n(g,r_{\tau})$ implies Equation~\eqref{sa-eq: M- and R- processes -- proj error unbdd part4} holds with $\alpha = 0$ and similarly for the $Z_n^G$ counterpart.
\end{myproof}
\subsection{General Result}\label{sa-sec: MULT -- general result}
This section presents the main result for the $G_n$-process. To simplify notation, the parameters of \(\mathscr{G}\) and \(\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}\) (Definitions 4 to 12, \ref{sa-defn: smooth tv}, \ref{sa-defn: smooth ktv}) are taken with \(\mathcal{C} = \mathcal{Q}_{\mathscr{G}}\), and the index \(\mathcal{Q}_{\mathscr{G}}\) is omitted where there is no ambiguity; the parameters of \(\mathscr{R}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{Y}\), and the index \(\mathcal{Y}\) is omitted where there is no ambiguity; and the parameters of \(\mathscr{G} \times \mathscr{R}\) (Definitions 4 to 12, \ref{sa-defn: uniform covering number for product space}, \ref{sa-defn: uniform entropy integral for product space}) are taken with \(\mathcal{C} = \mathcal{Q}_{\mathscr{G}} \times \mathcal{Y}\), and the index \(\mathcal{Q}_{\mathscr{G}} \times \mathcal{Y}\) is omitted where there is no ambiguity.
\begin{thm}\label{sa-thm: M-process -- main theorem}
Suppose $(\mathbf{z}_i=(\mathbf{x}_i, y_i): 1 \leq i \leq n)$ are i.i.d. random vectors taking values in $(\mathbb{R}^{d+1}, \mathcal{B}(\mathbb{R}^{d+1}))$ with common law $\mathbbm{P}_Z$, where $\mathbf{x}_i$ has distribution $\mathbbm{P}_X$ supported on $\mathcal{X}\subseteq\mathbb{R}^d$, $y_i$ has distribution $\mathbbm{P}_Y$ supported on $\mathcal{Y}\subseteq\mathbb{R}$, and the following conditions hold.
\begin{enumerate}[label=(\roman*)]
\item $\mathscr{G}$ is a real-valued pointwise measurable class of functions on $(\mathbb{R}^d, \mathcal{B}(\mathbb{R}^d), \mathbbm{P}_X)$.
\item There exists a surrogate measure $\mathbb{Q}_\mathscr{G}$ for $\mathbbm{P}_X$ with respect to $\mathscr{G}$ such that $\mathbb{Q}_\mathscr{G} = \operatorname*{\mathfrak{m}} \circ \phi_\mathscr{G}$, where the \textit{normalizing transformation} $\phi_{\mathscr{G}}: \mathcal{Q}_\mathscr{G} \mapsto [0,1]^d$ is a diffeomorphism.
\item $\mathtt{M}_{\mathscr{G}} < \infty$ and $J(\mathscr{G}, \mathtt{M}_{\mathscr{G}},1) < \infty$.
\item $\mathscr{R}$ is a real-valued pointwise measurable class of functions on $(\mathbb{R}, \mathcal{B}(\mathbb{R}),\mathbbm{P}_Y)$.
\item $J(\mathscr{R},M_{\mathscr{R}},1) < \infty$, where $M_{\mathscr{R}}(y) + \mathtt{pTV}_{\mathscr{R},(-|y|,|y|)} \leq \mathtt{v} (1 + |y|^{\alpha})$ for all $y \in \mathcal{Y}$, for some $\mathtt{v}>0$, and for some $\alpha\geq0$. Furthermore, if $\alpha>0$, then $\sup_{\mathbf{x} \in \mathcal{X}}\mathbbm{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] \leq 2$.
\end{enumerate}
Then, on a possibly enlarged probability space, there exists a sequence of mean-zero Gaussian processes $(Z_n^G(g,r): (g,r)\in \mathscr{G}\times \mathscr{R})$ with almost sure continuous trajectories such that:
\begin{itemize}
\item $\mathbbm{E}[G_n(g_1, r_1) G_n(g_2, r_2)] = \mathbbm{E}[Z^G_n(g_1, r_1) Z^G_n(g_2, r_2)]$ for all $(g_1, r_1), (g_2, r_2) \in \mathscr{G} \times \mathscr{R}$, and
\item $\mathbbm{P}\big[\lVert G_n - Z_n^G \rVert_{\mathscr{G} \times \mathscr{R}} > C_1c_{\mathtt{v},\alpha} \mathsf{T}_n^G(t)\big] \leq C_2 e^{-t}$ for all $t > 0$,
\end{itemize}
where $C_1$ and $C_2$ are universal constants, $C_{\mathtt{v},\alpha} = \mathtt{v} \max\{1 + (2 \alpha)^{\frac{\alpha}{2}}, 1 + (4 \alpha)^{\alpha}\}$, and
\begin{align*}
\mathsf{T}^G_n(t) = \min_{\delta \in (0,1)}\{\mathsf{A}^G_n(t,\delta) + \mathsf{F}^G_n(t,\delta)\},
\end{align*}
with
\begin{align*}
\mathsf{A}^G_n(t,\delta)
&= \sqrt{d} \min \Big\{ \Big( \frac{\mathtt{c}_1^{d} \mathtt{E}_{\mathscr{G}} \mathtt{TV}^{d}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}^{d+1}}{n}\Big)^{\frac{1}{2(d+1)}}, \Big(\frac{\mathtt{c}_1^{d} \mathtt{c}_2^{d}\mathtt{E}_{\mathscr{G}}^2 \mathtt{M}_{\mathscr{G}}^2 \mathtt{TV}_{\mathscr{G}}^{d} \mathtt{L}_{\mathscr{G}}^{d}}{n^2} \Big)^{\frac{1}{2(d+2)}} \Big\} (t + \log(n \mathtt{N}(\delta) N^{\ast}))^{\alpha + 1} \\
&\qquad + \sqrt{\frac{\min\{\mathtt{M}_{\mathscr{G}}^2(M^{\ast} + N^{\ast}), \mathtt{M}_{\mathscr{G}}(\mathtt{c}_3 \mathtt{K}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}} + \mathtt{M}_{\mathscr{G}})\}}{n}} (\log n)^{\alpha} (t + \log(n \mathtt{N}(\delta) N^{\ast}))^{\alpha + 1}, \\
\mathsf{F}^G_n(t,\delta)
&= J(\delta) \mathtt{M}_{\mathscr{G}} + \frac{(\log n)^{\alpha/2} \mathtt{M}_{\mathscr{G}} J^2(\delta)}{\delta^2 \sqrt{n}} + \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} \sqrt{t} + (\log n)^{\alpha}\frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} t^{\alpha},
\end{align*}
where
\begin{align*}
\mathtt{c}_1 = d \sup_{\mathbf{x} \in \mathcal{Q}_{\mathscr{H}}} \prod_{j = 1}^{d-1} \sigma_j(\nabla \phi_{\mathscr{H}}(\mathbf{x})),
\qquad
\mathtt{c}_2 = \sup_{\mathbf{x} \in \mathcal{Q}_{\mathscr{H}}} \frac{1}{\sigma_{d}(\nabla \phi_{\mathscr{H}}(\mathbf{x}))},
\qquad
\mathtt{c}_3 = d^{-1/2} (2 \sqrt{d})^{d-1} \mathtt{c}_1 \mathtt{c}_2^{d-1},
\end{align*}
and
\begin{align*}
\mathscr{V}_{\mathscr{R}} &= \{\theta(\cdot,r): r \in \mathscr{R}\}, \\
\mathtt{N}(\delta) & = \mathtt{N}_{\mathscr{G}}(\delta/\sqrt{2}, \mathtt{M}_{\mathscr{G}}) \mathtt{N}_{\mathscr{R}}(\delta/\sqrt{2}, M_{\mathscr{R}}), \qquad \delta \in (0,1], \\
J(\delta) & = \sqrt{2}J(\mathscr{G},\mathtt{M}_{\mathscr{G}},\delta/\sqrt{2}) + \sqrt{2}J(\mathscr{R},M_{\mathscr{R}}, \delta/\sqrt{2}), \qquad \delta \in (0,1],\\
M^{\ast} &= \Big\lfloor\log_2\min \Big\{\Big(\frac{\mathtt{c}_1 n \mathtt{TV}_{\mathscr{G}}}{\mathtt{E}_{\mathscr{G}}}\Big)^{\frac{d}{d+1}}, \Big( \frac{\mathtt{c}_1 \mathtt{c}_2 n \mathtt{L}_{\mathscr{G}} \mathtt{TV}_{\mathscr{G}} }{\mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}}\Big)^{\frac{d}{d+2}} \Big\} \Big\rfloor, \\
N^{\ast} & = \Big\lceil \log_2 \max \Big\{\Big(\frac{n \mathtt{M}_{\mathscr{G}}^{d+1}}{\mathtt{c}_1^d \mathtt{E}_{\mathscr{G}} \mathtt{TV}_{\mathscr{G}}^d}\Big)^{\frac{1}{d+1}}, \Big( \frac{n^2 \mathtt{M}_{\mathscr{G}}^{2d+2}}{\mathtt{c}_1^d \mathtt{c}_2^d \mathtt{TV}_{\mathscr{G}}^d \mathtt{L}_{\mathscr{G}}^d \mathtt{E}_{\mathscr{G}}^2}\Big)^{\frac{1}{d+2}}\Big\}\Big\rceil.
\end{align*}
\end{thm}
\begin{myproof}{Theorem~\ref{sa-thm: M-process -- main theorem}}
To simplify notation, we will use $\mathbbm{E} [\cdot |\mathcal{X}_{0,l}]$ in short for $\mathbbm{E} [\cdot | \mathbf{x}_i \in \mathcal{X}_{0,l}]$, and $\mathbbm{E}[\cdot|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}]$ in short for $\mathbbm{E}[\cdot|(\mathbf{x}_i, y_i) \in \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}]$ in this proof.
First, we make a reduction via the surrogate measure and normalizing transformation. Since $\operatorname{Supp}(\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}) \subseteq \operatorname{Supp}(\mathscr{G})$, we know $\mathcal{Q}_{\mathscr{G}}$ is also a surrogate measure for $\mathbbm{P}_X$ with respect to $\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}$, and $\phi_{\mathscr{G}}$ remains a valid normalizing transformation. Let $\mathcal{Z}_{\mathscr{G}} = \mathcal{X} \cap \operatorname{Supp}(\mathscr{G})$. Since $\mathbb{Q}_\mathscr{G} = \operatorname*{\mathfrak{m}} \circ \phi_\mathscr{G}$ by assumption (ii) in Theorem 1, and $\mathbb{Q}_\mathscr{G}|_{\mathcal{Z}_\mathscr{G}} = \mathbbm{P}_X|_{\mathcal{Z}_\mathscr{G}}$,
\begin{align*}
\mathbbm{P}_X|_{\mathcal{Z}_\mathscr{G}} = \operatorname*{\mathfrak{m}} \circ \phi_{\mathscr{G}}|_{\mathcal{Z}_\mathscr{G}}.
\end{align*}
To define the $\mathsf{Uniform}([0,1]^d)$ random variables on the probability space that $(\mathbf{x}_i,y_i)$'s live in, we define a joint probability measure $\mathbb{O}$ on $(\mathbb{R}^d \times \mathbb{R} \times \mathbb{R}^d, \mathcal{B}(\mathbb{R}^{2d+1}))$ such that for all $A \in \mathcal{B}(\mathbb{R}^{2d+1})$:
\begin{align*}
\mathbb{O}(A \cap (\mathcal{Z}_{\mathscr{H}} \times \mathbb{R} \times \mathcal{Z}_{\mathscr{H}}))
& = \mathbbm{P}_Z(\Pi_{1:d+1}(A \cap \{(\mathbf{x},y,\mathbf{x}): \mathbf{x} \in \mathcal{Z}_{\mathscr{H}}, y \in \mathbb{R}\})), \\
\mathbb{O}(A \cap (\mathcal{Z}_{\mathscr{H}} \times \mathbb{R} \times \mathcal{Z}_{\mathscr{H}}^c))
& = \mathbb{O}(A \cap (\mathcal{Z}_{\mathscr{H}}^c \times \mathbb{R} \times \mathcal{Z}_{\mathscr{H}})) = 0, \\
\mathbb{O}(A \cap (\mathcal{Z}_{\mathscr{H}}^c \times \mathbb{R} \times \mathcal{Z}_{\mathscr{H}}^c))
& = \int_{\mathcal{Z}_{\mathscr{H}}^c \cap \Pi_{d+2:2d+1}(A)} \frac{\mathbbm{P}_Z(A^{\mathbf{u}} \cap (\mathcal{Z}_{\mathscr{H}}^c \times \mathbb{R}))}{\mathbbm{P}_Z(\mathcal{Z}_{\mathscr{H}}^c \times \mathbb{R})} d (\mathfrak{m} \circ \phi_{\mathscr{H}})(\mathbf{u}),
\end{align*}
where $\Pi_{1:d+1}(A) = \{\mathbf{z} \in \mathbb{R}^{d+1}: (\mathbf{z},\mathbf{u}) \in A \text{ for some } \mathbf{u} \in \mathbb{R}^{d}\}$, $\Pi_{d+2:2d+1}(A) = \{\mathbf{u} \in \mathbb{R}^{d}: (\mathbf{z},\mathbf{u}) \in A \text{ for some } \mathbf{z} \in \mathbb{R}^{d+1}\}$, and $A^{\mathbf{u}} = \{\mathbf{z} \in \mathbb{R}^{d+1}: (\mathbf{z},\mathbf{u}) \in A\}$.
Then we can check that (i) the marginals of $\mathbb{O}$ are $\mathbbm{P}_Z$ and $\mathfrak{m} \circ \phi_{\mathscr{H}}$, respectively; (ii) $\mathbb{O}|_{\mathcal{Z}_{\mathscr{H}} \times \mathbb{R} \times \mathbb{R}^d \cup \mathbb{R}^d \times \mathbb{R} \times \mathcal{Z}_{\mathscr{H}}}$ is supported on $\{(\mathbf{x}, y,\mathbf{x}): \mathbf{x} \in \mathcal{Z}_{\mathscr{H}}, y \in \mathbb{R}\}$. By Skorohod embedding \citep[Lemma 3.35]{dudley2014uniform}, on a possibly enlarged probability space, there exists a $\mathbf{u}_i, 1 \leq i \leq n$ i.i.d. $\operatorname{Uniform}([0,1]^d)$ such that $(\mathbf{z}_i = (\mathbf{x}_i,y_i),\phi_\mathscr{H}^{-1}(\mathbf{u}_i))$ has joint law $\mathbb{O}$. In particular, if $\mathbf{x}_i \in \mathcal{Z}_{\mathscr{H}}$, then $\mathbf{x}_i = \phi_{\mathscr{H}}^{-1}(\mathbf{u}_i)$; if $\mathbf{x}_i \in \mathcal{Z}_{\mathscr{H}}^c$, then $\phi_{\mathscr{H}}^{-1}(\mathbf{u}_i) \in \mathcal{Z}_{\mathscr{H}}^c$, and since $\mathcal{Q}_\mathscr{H} \subseteq \mathcal{X} \cup (\cap_{h \in \mathscr{H}} \operatorname{Supp}(h)^c)$, $\phi_{\mathscr{H}}^{-1}(\mathbf{u}_i) \in \cap_{h \in \mathscr{H}} \operatorname{Supp}(h)^c$. In particular, $\sup_{\mathbf{u} \in [0,1]^d}\mathbbm{E}[\exp(|y_i|)|\mathbf{u}_i = \mathbf{u}] \leq 2$.
By the same argument as in the proof for Theorem 1, assumption (ii) implies that on a possibly enriched probability space, there exists $(\mathbf{u}_i: 1 \leq i \leq n)$ i.i.d distributed with law $\mathbbm{P}_U = \mathsf{Uniform}([0,1]^d)$, and
\begin{align*}
g(\mathbf{x}_i) = g(\phi_{\mathscr{H}}^{-1}(\mathbf{u}_i)), \qquad \forall g \in \mathscr{G}, 1 \leq i \leq n.
\end{align*}
Define $\widetilde{G}_n$ to be the empirical process based on $((\mathbf{u}_i,y_i): 1 \leq i \leq n)$, and
\begin{align*}
\widetilde{G}_n(f,s) = \frac{1}{\sqrt{n}}\sum_{i = 1}^n \Big[f(\mathbf{u}_i)s(y_i) - \mathbbm{E}[f(\mathbf{u}_i)s(y_i)] \Big],
\end{align*}
and take $\widetilde{\mathscr{G}} = \{g \circ \phi_{\mathscr{H}}^{-1}: g \in \mathscr{G}\}$, then
\begin{align*}
G_n(g,r) = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \Big[g(\mathbf{x}_i)r(y_i) - \mathbbm{E}[g(\mathbf{x}_i)r(y_i)]\Big] = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \Big[\widetilde{g}(\mathbf{u}_i) r(y_i) - \mathbbm{E}[\widetilde{g}(\mathbf{u}_i) r(y_i)]\Big] = \widetilde{G}_n(\widetilde{g},r).
\end{align*}
The relation between constants for $\widetilde{\mathscr{G}}$ and constants for $\mathscr{G}$ can be deduced from Lemma~\ref{sa-lem: normalizing transformation}. Hence, without loss of generality, we assume $(\mathbf{x}_i: 1 \leq i \leq n)$ are i.i.d under common law $\mathbbm{P}_X = \mathsf{Uniform}([0,1]^d)$ distributed and $\mathcal{X} = [0,1]^d$.
Take $\mathscr{A}_{M,N}(\mathbbm{P}_Z,1)$ to be an axis-aligned cylindered quasi-dyadic expansion of $\mathbb{R}^{d+1}$, of depth $M$ for the main subspace $\mathbb{R}^d$ and depth $N$ for the multiplier subspace $\mathbb{R}$ with respect to $\mathbbm{P}_Z$. Take $(Z_n^G(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ and $(\mathtt{\Pi}_{1} Z_n^G(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ to be the mean-zero Gaussian processes constructed as in Equations~\eqref{sa-eq: M- Process -- Gaussian Process 1} and \eqref{sa-eq: M- Process -- Gaussian Process 2}. Let $(\mathscr{G} \times \mathscr{R})_{\delta}$ be a $\delta \lVert \mathtt{M}_{\mathscr{G}} M_{\mathscr{R}} \rVert_{\mathbbm{P}_Z}$-net of $\mathscr{G} \times \mathscr{R}$ with cardinality no greater than $\mathtt{N}_{\mathscr{G} \times \mathscr{R}}(\delta, \mathtt{M}_{\mathscr{G}} M_{\mathscr{R}})$. By standard empirical process argument, $\mathtt{N}_{\mathscr{G} \times \mathscr{R}}(\delta, \mathtt{M}_{\mathscr{G}} M_{\mathscr{R}}) \leq \mathtt{N}(\delta)$. By Lemma~\ref{sa-lem: M- Process -- Fluctuation off the net}, the meshing error can be bounded by: For all $t > 0$,
\begin{align*}
\mathbbm{P}\big[\lVert G_n - G_n\circ\pi_{(\mathscr{G}\times \mathscr{R})_\delta} \rVert_{\mathscr{G} \times \mathscr{R}} + \lVert Z_n^G - Z_n^G\circ\pi_{(\mathscr{G}\times \mathscr{R})_\delta} \rVert_{\mathscr{G} \times \mathscr{R}} > C_1 c_{\mathtt{v},\alpha} \mathsf{F}_n^G(t,\delta)\big] & \leq 8 \exp(-t),
\end{align*}
where $C_1$ is a universal constant and $c_{\mathtt{v},\alpha} = \mathtt{v}(1 + (2 \alpha)^{\frac{\alpha}{2}})$. Lemma~\ref{sa-lem: M- process -- SA error} implies that the strong approximation error for the projected process on $\delta$-net is bounded by: For all $t > 0$,
\begin{align*}
\mathbbm{P}\Big[\lVert \mathtt{\Pi}_{1} G_n - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
> C_1 c_{\mathtt{v},\alpha} \sqrt{\frac{N^{2\alpha + 1} 2^{M} \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}}{n} t} +C_1 c_{\mathtt{v},\alpha}\sqrt{\frac{\mathtt{C}_{\mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}),M + N}}{n}} t \Big]
\leq 2 \mathtt{N}(\delta) e^{-t}.
\end{align*}
where
\begin{align*}
\mathtt{C}_{\mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}), M+N}
& = \sup_{f \in \mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R})}\min\bigg\{\sup_{(j,k) \in \mathcal{I}_{M+N}} \left[\sum_{j^{\prime} < j} (j-j^{\prime})(j-j^{\prime}+1) 2^{j^{\prime}-j} \sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} \widetilde{\beta}_{j^{\prime},k^{\prime}}^2(f) \right], \\
& \hspace{1in} \sup_{\mathbf{z} \in \mathcal{C}_{M + N,0}}f(\mathbf{z})^2(M + N)\bigg\}.
\end{align*}
Now we upper bound the left hand side of the minimum. Let $f \in \mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R})$. Then there exists $g \in \mathscr{G}$ and $r \in \mathscr{R}$ such that $f = \mathtt{\Pi}_{1}[g,r]$. Since $f$ is already piecewise-constant, by definition of $\beta_{j,k}$'s and $\gamma_{j,k}$'s, we know $\widetilde{\beta}_{l,m}(f) = \widetilde{\gamma}_{l,m}(g,r)$. Fix $(j,k) \in \mathcal{I}_{M+N}$. We consider two cases. \\
\noindent \textbf{Case 1: $j > N$.} By Definition~\ref{sa-defn: cylindered quasi dyadic expansion}, $\mathcal{C}_{j,k} = \mathcal{X}_{j-N,k} \times \mathcal{Y}_{*,N,0}$. By definition of $\mathcal{A}_{M,N}(\mathbbm{P}_Z,1)$ and the assumption that $\mathbf{x}_i$'s are $\mathsf{Uniform}([0,1]^d)$ distributed, $\lVert \mathcal{X}_{j-N,k} \rVert_{\infty} \leq 2^{-\frac{M + N - j}{d}+1}$.
Consider $j^{\prime}$ such that $N \leq j^{\prime} \leq j$. By definition of $\mathcal{A}_{M,N}(\mathbbm{P}_Z,1)$ and the assumption that $\mathbf{x}_i$'s are $\mathsf{Uniform}([0,1]^d)$ distributed, the $j^{\prime}$-th level difference set $\mathcal{U}_{j^{\prime}} = \cup_{0 \leq k < 2^{M + N - j^{\prime}}}(\mathcal{C}_{j^{\prime}-1,2k+1} - \mathcal{C}_{j^{\prime}-1,2k})$ is contained in $[-2^{-\frac{M + N - j^{\prime}}{d}+2}, 2^{-\frac{M + N - j^{\prime}}{d}+2}]^d$. Let $g \in \mathscr{G}$, $r \in \mathscr{R}$. By definition of $\widetilde{\gamma}_{j^{\prime},m}$ and similar arguments to those in the proof of Lemma~\ref{sa-lem: X-process -- SA error},
\begin{align*}
\sum_{m: \mathcal{C}_{j^{\prime},m} \subseteq \mathcal{C}_{j,k}} \left|\widetilde{\gamma}_{j^{\prime},m}(g,r)\right|
& \leq 2^{2(M + N - j^{\prime})} \int_{\mathcal{U}_j^{\prime}} \int_{\mathcal{X}_{j-N,k}} |g(\mathbf{x}) \theta(\mathbf{x},r) - g(\mathbf{x} + \mathbf{s}) \theta(\mathbf{x} + \mathbf{s},r)|d \mathbf{x} d \mathbf{s} \\
& \leq 2^{2(M + N - j^{\prime})} \int_{\mathcal{U}_j^{\prime}} \lVert \mathbf{s} \rVert \lVert \mathcal{X}_{j-N,k} \rVert_{\infty}^{d-1}d \mathbf{s} \,\mathtt{K}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}^*\\
& \leq 2^{2(M + N - j^{\prime})} \operatorname*{\mathfrak{m}}(\mathcal{U}_j^{\prime}) \lVert \mathcal{U}_j^{\prime} \rVert_{\infty} \lVert \mathcal{X}_{j-N,k} \rVert_{\infty}^{d-1} \mathtt{K}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}^*\\
& \leq 2^{\frac{d-1}{d}(j - j^{\prime})} \mathtt{K}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}^*.
\end{align*}
Next, consider $j^{\prime}$ such that $0 \leq j^{\prime} < N$, we know
\begin{align*}
& \sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} |\widetilde{\gamma}_{j^{\prime},k^{\prime}}(g,r)| \\
&\quad = \sum_{j^{\prime}: \mathcal{X}_{0,j^{\prime}} \subseteq \mathcal{X}_{j-N,k}} \sum_{0 \leq m < 2^{j^{\prime}}} |\mathbbm{E}[g(\mathbf{x}_i)|\mathcal{X}_{0,j^{\prime}}]| \cdot |\mathbbm{E}[r(y_i)|\mathcal{X}_{0,j^{\prime}} \times \mathcal{Y}_{j^{\prime},j-1,2m}] - \mathbbm{E}[r(y_i)|\mathcal{X}_{0,j^{\prime}} \times \mathcal{Y}_{j^{\prime},j-1,2m+1}]| \\
&\quad \leq c_{\mathtt{v},\alpha} \sum_{j^{\prime}:\mathcal{X}_{0,j^{\prime}} \subseteq \mathcal{X}_{j-N,k}} |\mathbbm{E}[g(\mathbf{x}_i)|\mathcal{X}_{0,j^{\prime}}]| N^{\alpha} \\
&\quad \leq c_{\mathtt{v},\alpha} 2^{j - N} \mathtt{M}_{\mathscr{G}} N^{\alpha}.
\end{align*}
It follows that
\begin{align*}
& \sum_{j^{\prime} < j} (j-j^{\prime})(j-j^{\prime}+1) 2^{j^{\prime}-j} \sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} |\widetilde{\gamma}_{j^{\prime},k^{\prime}}(g,r)| \\
& \quad \leq \sum_{N \leq j^{\prime} < j} (j - j^{\prime})(j - j^{\prime} + 1) 2^{-\frac{j - j^{\prime}}{d}}\mathtt{K}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}^* + c_{\mathtt{v},\alpha} \sum_{j^{\prime} < N} (j-j^{\prime})(j-j^{\prime}+1) 2^{j^{\prime} - N} \mathtt{M}_{\mathscr{G}}N^{\alpha} \\
& \quad \lesssim \mathtt{K}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}^* + c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}} N^{\alpha}.
\end{align*}
\textbf{Case 2: $j \leq N$.} Then $\mathcal{C}_{j,k} = \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}$ with $k = 2^{N-j}l + m$, and $\mathcal{C}_{j^{\prime},k^{\prime}} = \mathcal{X}_{0,l^{\prime}} \times \mathcal{Y}_{l^{\prime},j^{\prime},m^{\prime}}$ with $k^{\prime} = 2^{N-j^{\prime}}l^{\prime} + m^{\prime}$. In particular, $\mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}$ implies $l^{\prime} = l$ and $\mathcal{Y}_{l^{\prime},j^{\prime},m^{\prime}} \subseteq \mathcal{Y}_{l,j,m}$. By a similar argument to the proof in Lemma~\ref{sa-lem: M- process -- SA error} (Layers $1 \leq j \leq N$), for any $0 \leq j^{\prime} \leq j$,
\begin{align*}
& \sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} |\widetilde{\gamma}_{j^{\prime},k^{\prime}}(g,r)| \\
& \quad = |\mathbbm{E}[g(\mathbf{x}_i)|\mathcal{X}_{0,l}]| \sum_{m^{\prime}: \mathcal{Y}_{l,j^{\prime},m^{\prime}} \subseteq \mathcal{Y}_{l,j,m}} |\mathbbm{E}[r(y_i)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,2m}] - \mathbbm{E}[r(y_i)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,2m+1}]| \\
& \quad \leq c_{\mathtt{v},\alpha} |\mathbbm{E}[g(\mathbf{x}_i)|\mathcal{X}_{0,l}]| N^{\alpha} \\
& \quad \leq c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}} N^{\alpha}.
\end{align*}
Using the elementary inequality that $x(x+1) \leq 30 \cdot 2^{x/4}$ for $x > 0$, we can get
\begin{align*}
\sum_{1 \leq j^{\prime} < j} (j-j^{\prime})(j-j^{\prime}+1) 2^{j^{\prime}-j} \sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} |\widetilde{\gamma}_{j^{\prime},k^{\prime}}(g,r)| \leq 60 c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}} N^{\alpha}.
\end{align*}
Moreover, for all $(j,k)$, we have $\widetilde{\beta}_{j,k}(g,r) \leq c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}} N^{\alpha}$. This implies that
\begin{align*}
\mathtt{C}_{\mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}), M+N} \lesssim c_{\mathtt{v},\alpha}^2 \mathtt{M}_{\mathscr{G}}N^{\alpha} \min\{\mathtt{K}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}^* + \mathtt{M}_{\mathscr{G}}N^{\alpha}, \mathtt{M}_{\mathscr{G}} N^{\alpha}(M + N)\}.
\end{align*}
Since $\mathbf{x}_i \stackrel{i.i.d.}{\thicksim}\mathsf{Uniform}([0,1]^d)$ and the cells $\mathscr{A}_{M,N}(\mathbbm{P}_Z,1)$ are obtained via \emph{axis aligned dyadic expansion}, we have $\lVert \mathcal{X}_{0,k} \rVert_{\infty} \leq 2^{-\lfloor M/d \rfloor}$ for all $0 \leq k < 2^{M}$. Then by Lemma~\ref{sa-lem: M- process -- projection error}, for all $t > N$,
\begin{align*}
\mathbbm{P}\Big[\lVert G_n - \mathtt{\Pi}_{1} G_n \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
\gtrsim \sqrt{c_{\mathtt{v}, 2 \alpha}}\sqrt{N^2 \mathtt{V}_{\mathscr{G}} + 2^{-N} \mathtt{M}_{\mathscr{G}}^2} t^{\alpha + \frac{1}{2}} + c_{\mathtt{v},\alpha}\frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}}t^{\alpha + 1}\Big]
& \leq 4 \mathtt{N}(\delta) n e^{-t},\\
\mathbbm{P}\Big[\lVert Z_n^G - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
\gtrsim \sqrt{c_{\mathtt{v}, 2 \alpha}} \sqrt{N^2 \mathtt{V}_{\mathscr{G}} + 2^{-N} \mathtt{M}_{\mathscr{G}}^2} t^{\frac{1}{2}} + c_{\mathtt{v},\alpha} \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}}t \Big]
& \leq 4 \mathtt{N}(\delta) n e^{-t},
\end{align*}
where $c_{\mathtt{v},\alpha} = \mathtt{v}(1 + (2 \alpha)^{\frac{\alpha}{2}})$ and $c_{\mathtt{v}, 2 \alpha} = \mathtt{v}^2(1 + (4 \alpha)^{\alpha})$, and
\begin{align*}
\mathtt{V}_{\mathscr{G}} = \sqrt{d} \min\{2 \mathtt{M}_{\mathscr{G}}, \mathtt{L}_{\mathscr{G}}2^{-\lfloor M/d\rfloor}\} 2^{-\lfloor M/d\rfloor} \mathtt{TV}_{\mathscr{G}}.
\end{align*}
We find the optimal parameters $M^{\ast}$ and $N^{\ast}$ by balancing the term $\sqrt{\frac{2^M \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}}{n}}$ from the bound on $\|\mathtt{\Pi}_{1} G_n - \mathtt{\Pi}_{1} Z_n^G\|_{(\mathscr{G} \times \mathscr{R})_{\delta}}$ and the term $\mathtt{V}_{\mathscr{G}}$ from the bounds on $\|G_n - \mathtt{\Pi}_{1} G_n\|_{(\mathscr{G} \times \mathscr{R})_{\delta}}$ and $\|Z_n - \mathtt{\Pi}_{1} Z_n^G\|_{(\mathscr{G} \times \mathscr{R})_{\delta}}$, choosing
\begin{align*}
2^{M^{\ast}} = \min \left\{\left(\frac{n \mathtt{TV}_{\mathscr{G}}}{\mathtt{E}_{\mathscr{G}}}\right)^{\frac{d}{d+1}}, \left( \frac{n \mathtt{L}_{\mathscr{G}} \mathtt{TV}_{\mathscr{G}} }{\mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}}\right)^{\frac{d}{d+2}} \right\}, \quad 2^{N^{\ast}} = \max \left\{\left(\frac{n \mathtt{M}_{\mathscr{G}}^{d+1}}{\mathtt{E}_{\mathscr{G}} \mathtt{TV}_{\mathscr{G}}^d}\right)^{\frac{1}{d+1}}, \left(\frac{n^2 \mathtt{M}_{\mathscr{G}}^{2d+2}}{\mathtt{TV}_{\mathscr{G}}^d \mathtt{L}_{\mathscr{G}}^d \mathtt{E}_{\mathscr{G}}^2}\right)^{\frac{1}{d+2}}\right\}.
\end{align*}
It follows that for all $t > N_{\ast}$, with probability at least $1 - 4 n \mathtt{N}(\delta)\exp(-t)$,
\begin{align*}
& \qquad \lVert G_n - Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}} \\
& \leq \sqrt{d} N^{\ast} \min \left\{ \left( \frac{\mathtt{E}_{\mathscr{G}} \mathtt{TV}_{\mathscr{G}}^d \mathtt{M}_{\mathscr{G}}^{d+1}}{n}\right)^{\frac{1}{2(d+1)}}, \left(\frac{\mathtt{E}_{\mathscr{G}}^2 \mathtt{M}_{\mathscr{G}}^2 \mathtt{TV}_{\mathscr{G}}^{d} \mathtt{L}_{\mathscr{G}}^{d}}{n^2} \right)^{\frac{1}{2(d+2)}}\right\}t^{\alpha + \frac{1}{2}} + \sqrt{\frac{\mathtt{C}_{\mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}), M+N}}{n}} t^{\alpha + 1}.
\end{align*}
The result then follows from the decomposition that
\begin{equation*}
\begin{split}
\lVert G_n - Z_n^G \rVert_{\mathscr{G} \times \mathscr{R}}
&= \lVert G_n - Z_n^G \rVert_{\mathscr{G} \times \mathscr{R}} \\
&\leq \lVert G_n - G_n \circ \pi_{(\mathscr{G} \times \mathscr{H})_{\delta}} \rVert_{\mathscr{G}\times \mathscr{R}} + \lVert Z_n^G - Z_n^G \circ \pi_{(\mathscr{G} \times \mathscr{R})_{\delta}} \rVert_{\mathscr{G}\times \mathscr{R}} \\
& \qquad + \lVert G_n - \mathtt{\Pi}_{1} G_n \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}} + \lVert Z_n^G - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
+ \lVert \mathtt{\Pi}_{1} G_n - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}},
\end{split}
\end{equation*}
and Lemma~\ref{sa-lem: normalizing transformation} for the reduction to the case of $\mathsf{Uniform}([0,1]^d)$ distributed $\mathbf{x}_i$'s.
\end{myproof}
\subsection{Additional Results}\label{sa-sec: MULT -- additional result}
This section presents the additional result for the $G_n$-process under VC-type entropy conditions. To simplify notation, the parameters of \(\mathscr{G}\) and \(\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}\) (Definitions 4 to 12, \ref{sa-defn: smooth tv}, \ref{sa-defn: smooth ktv}) are taken with \(\mathcal{C} = \mathcal{Q}_{\mathscr{G}}\), and the index \(\mathcal{Q}_{\mathscr{G}}\) is omitted where there is no ambiguity; the parameters of \(\mathscr{R}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{Y}\), and the index \(\mathcal{Y}\) is omitted where there is no ambiguity; and the parameters of \(\mathscr{G} \times \mathscr{R}\) (Definitions 4 to 12, \ref{sa-defn: uniform covering number for product space}, \ref{sa-defn: uniform entropy integral for product space}) are taken with \(\mathcal{C} = \mathcal{Q}_{\mathscr{G}} \times \mathcal{Y}\), and the index \(\mathcal{Q}_{\mathscr{G}} \times \mathcal{Y}\) is omitted where there is no ambiguity.
\begin{coro}[VC-Type Lipschitz Functions]\label{sa-coro: M- process -- vc type}
Suppose the conditions of Theorem~\ref{sa-thm: M-process -- main theorem} and the following additional conditions hold.
\begin{enumerate}[label=(\roman*)]
\item $\mathscr{G}$ is a VC-type class with respect to envelope $\mathtt{M}_{\mathscr{G}}$ with constant $\mathtt{c}_{\mathscr{G}} \geq e$ and exponent $\mathtt{d}_{\mathscr{G}} \geq 1$ over $\mathcal{Q}_{\mathscr{G}}$.
\item $\mathscr{R}$ is a VC-type class with respect to envelope $M_{\mathscr{R}}$ with constant $\mathtt{c}_{\mathscr{R}} \geq e$ and exponent $\mathtt{d}_{\mathscr{R}} \geq 1$ over $\mathcal{Y}$.
\item There exists a constant $\mathtt{k}$ such that $|\log_2 \mathtt{E}_{\mathscr{G}}| + |\log_2 \mathtt{TV}| + |\log_2 \mathtt{M}_{\mathscr{G}}| \leq \mathtt{k} \log_2 n$, where we take $\mathtt{TV} = \max \{\mathtt{TV}_{\mathscr{G}}, \mathtt{TV}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}\}$.
\end{enumerate}
Then, on a possibly enlarged probability space, there exists a mean-zero Gaussian process $(Z_n^G(g,r): (g,r)\in \mathscr{G}\times \mathscr{R})$ with almost sure continuous trajectories such that:
\begin{itemize}
\item $\mathbbm{E}[G_n(g_1, r_1) G_n(g_2, r_2)] = \mathbbm{E}[Z^G_n(g_1, r_1) Z^G_n(g_2, r_2)]$ for all $(g_1, r_1), (g_2, r_2) \in \mathscr{G} \times \mathscr{R}$, and
\item $\mathbbm{P}\big[\lVert G_n - Z_n^G \rVert_{\mathscr{G} \times \mathscr{R}} > C_1c_{\mathtt{v},\alpha} \mathsf{T}_n^G(t)\big] \leq C_2 e^{-t}$ for all $t > 0$,
\end{itemize}
where $C_1$ and $C_2$ are universal constants, $c_{\mathtt{v},\alpha} = \mathtt{v} \max\{1 + (2 \alpha)^{\frac{\alpha}{2}}, 1 + (4 \alpha)^{\alpha}\}$, and
\begin{align*}
\mathsf{T}^G_n(t)
&= \sqrt{d} \min\Big\{\Big( \frac{\mathtt{c}_1^d \mathtt{E}_{\mathscr{G}} \mathtt{TV}^d_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}^{d+1}}{n}\Big)^{\frac{1}{2(d+1)}}, \Big( \frac{\mathtt{c}_1^d \mathtt{c}_2^d \mathtt{E}_{\mathscr{G}}^2 \mathtt{M}_{\mathscr{G}}^2 \mathtt{TV}_{\mathscr{G}}^d \mathtt{L}_{\mathscr{G}}^d}{n^2}\Big)^{\frac{1}{2(d+2)}} \Big\} (t + \mathtt{k} \log_2(n) + \mathtt{d}\log (\mathtt{c} n))^{\alpha + 1} \\
&\qquad + \sqrt{\frac{\min\{\mathtt{k}\log_2(n)\mathtt{M}_{\mathscr{G}}^2, \mathtt{M}_{\mathscr{G}}(\mathtt{c}_3 \mathtt{K}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}} + \mathtt{M}_{\mathscr{G}})\}}{n}}(\log n)^{\alpha} (t + \mathtt{k} \log_2(n) + \mathtt{d}\log(\mathtt{c}n))^{\alpha + 1},
\end{align*}
with $\mathtt{c} = \mathtt{c}_{\mathscr{G}} \mathtt{c}_{\mathscr{R}}$, $\mathtt{d} = \mathtt{d}_{\mathscr{G}} + \mathtt{d}_{\mathscr{R}}$.
\end{coro}
\begin{myproof}{Corollary~\ref{sa-coro: M- process -- vc type}}
The proof follows by Theorem~\ref{sa-thm: M-process -- main theorem} with $\delta = n^{-1/2}$, and
\begin{align*}
\mathtt{N}(n^{-1/2}) & = \mathtt{N}_{\mathscr{G}}(1/\sqrt{2 n},\mathtt{M}_{\mathscr{G}}) \mathtt{N}_{\mathscr{R}}(1/\sqrt{2 n},M_{\mathscr{R}})
\leq \mathtt{c}_{\mathscr{G}} \mathtt{c}_{\mathscr{R}} (2 \sqrt{n})^{\mathtt{d}_{\mathscr{G}} + \mathtt{d}_{\mathscr{R}}} = \mathtt{c}(2 \sqrt{n})^{\mathtt{d}},
\end{align*}
and
\begin{align*}
J(n^{-1/2}) & = \sqrt{2}J(\mathscr{G},\mathtt{M}_{\mathscr{G}},1/\sqrt{2 n}) + \sqrt{2}J(\mathscr{R},M_{\mathscr{R}}, 1/\sqrt{2 n}) \\
& \leq 3 n^{-1/2} \sqrt{\mathtt{d}_{\mathscr{G}} \log(\mathtt{c}_{\mathscr{G}} \sqrt{n})} + 3 \delta \sqrt{\mathtt{d}_{\mathscr{R}} \log(\mathtt{c}_{\mathscr{R}} \sqrt{n})} \\
& \leq 3 \delta \sqrt{(\mathtt{d}_{\mathscr{G}} + \mathtt{d}_{\mathscr{R}})\log(\mathtt{c}_{\mathscr{G}} \mathtt{c}_{\mathscr{R}} n)}
\leq 3 \delta \sqrt{\mathtt{d} \log(\mathtt{c} n)}.
\end{align*}
The conclusion follows.
\end{myproof}
\section{Residual-Based Empirical Processes}\label{sa-sec: Residual-Based Empirical Process}
Recall that $\mathbf{z}_i=(\mathbf{x}_i, y_i)\in \mathcal{X}\times\mathcal{Y} \subseteq \mathbb{R}^d\times\mathbb{R}$, $i=1,\dots,n$, are i.i.d. random vectors supported on a background probability space $(\Omega,\mathcal{F},\mathbbm{P})$, and the \textit{residual-based empirical process} is
\begin{equation*}
R_n(g, r) = \frac{1}{\sqrt{n}} \sum_{i=1}^n \big( g(\mathbf{x}_i) r(y_i) - \mathbbm{E}[g(\mathbf{x}_i) r(y_i) | \mathbf{x}_i] \big),
\qquad (g, r) \in \mathscr{G} \times \mathscr{R}.
\end{equation*}
In particular, $(R_n(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ can be seen as a combination of the two empirical processes studied in the previous sections: for $r \in \mathscr{R}$ and $\mathbf{x} \in \mathcal{X}$,
\begin{align*}
R_n(g, r) = G_n(g, r) - X_n(g \, \theta(\cdot, r)),
\qquad
\theta(\mathbf{x}, r) = \mathbbm{E}[r(y_i) | \mathbf{x}_i = \mathbf{x}],
\end{align*}
where
\begin{align*}
G_n(g, r) &= \frac{1}{\sqrt{n}} \sum_{i=1}^n \Big[g(\mathbf{x}_i) r(y_i) - \mathbbm{E}[g(\mathbf{x}_i) r(y_i)]\Big], \\
X_n(g \,\theta(\cdot, r)) &= \frac{1}{\sqrt{n}} \sum_{i=1}^n \Big[g(\mathbf{x}_i) \theta(\mathbf{x}_i, r) - \mathbbm{E}[g(\mathbf{x}_i) \theta(\mathbf{x}_i, r)] \Big].
\end{align*}
Results for the $X_n$ process (Section~\ref{sa-sec: General Empirical Process}) and for the $G_n$ process (Section~\ref{sa-sec: Multiplicative Empirical Process}) will be used to handle the terms above. The same error decomposition as in Sections~\ref{sa-sec: General Empirical Process} and \ref{sa-sec: Multiplicative Empirical Process} also applies here:
\begin{align*}
\lVert R_n - Z_n^R \rVert_{\mathscr{G} \times \mathscr{R}}
& \leq \lVert R_n - Z_n^R \rVert_{(\mathscr{G} \times \mathscr{R})_\delta} + \lVert R_n - R_n\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta} \rVert_{\mathscr{G} \times \mathscr{R}}
+ \lVert Z_n^R\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta}-Z_n^R \rVert_{\mathscr{G} \times \mathscr{R}}\\
& \leq \lVert \mathtt{\Pi}_2 Z_n^R - Z_n^R \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}
+ \lVert R_n - \mathtt{\Pi}_2 R_n \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
+ \lVert \mathtt{\Pi}_2 R_n - \mathtt{\Pi}_2 Z_n^R \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}\\
& \qquad + \lVert R_n - R_n\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta} \rVert_{\mathscr{G} \times \mathscr{R}}
+ \lVert Z_n^R\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta}-Z_n^R \rVert_{\mathscr{G} \times \mathscr{R}},
\end{align*}
where $(\mathscr{G} \times \mathscr{R})_\delta$ denotes a discretization (or meshing) of $\mathscr{G} \times \mathscr{R}$ (i.e., $\delta$-net of $\mathscr{G} \times \mathscr{R}$), and the terms $\lVert R_n - R_n\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta} \rVert_{\mathscr{G} \times \mathscr{R}}$ and $\lVert Z_n^R\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta}-Z_n^R \rVert_{\mathscr{G} \times \mathscr{R}}$ capture the fluctuations (or oscillations) of $R_n$ and $Z_n^R$ relative to the meshing for each of the stochastic processes. $\lVert \mathtt{\Pi}_2 R_n - \mathtt{\Pi}_2 Z_n^R \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$ and $\lVert \mathtt{\Pi}_2 Z_n^R - Z_n^R \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$ represent projections onto a Haar function space, where $\mathtt{\Pi}_2 R_n(h) = R_n \circ \mathtt{\Pi}_2 h$. The operator $\mathtt{\Pi}_2$ is a projection onto piecewise constant functions that respects the multiplicative structure of the $R_n$ process. The final term $\lVert \mathtt{\Pi}_2 R_n - \mathtt{\Pi}_2 Z_n^R \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$ captures the coupling between the empirical process and the Gaussian process (on a $\delta$-net of $\mathscr{G} \times \mathscr{R}$, after the projection $\mathtt{\Pi}_2$).
The general result under uniform entropy integral conditions is presented in Section~\ref{sa-sec: REG -- general result}. Theorem 2 and Corollary 4 then follow from that general result. The proofs leverage the existence of a surrogate measure and a normalizing transformation of $\mathscr{G}$ with respect to $\mathbb{P}_X$, the distribution of $\mathbf{x}_1$, as developed in Section~\ref{sa-sec: X-Process -- normalizing transformation}. We will use the same class of cylindered quasi-dyadic cell expansions as in Section~\ref{sa-sec: MULT -- cell expansions}, which explicitly exploits the multiplicative structure of $R_n$. Bounds for each term in the error decomposition are provided in Section~\ref{sa-sec: REG -- preliminary}, which boils down to handle the extra $X_n(g \, \theta(\cdot,r))$ term compared to the results in Section~\ref{sa-sec: MULT -- preliminary} and is organized as follows:
\begin{itemize}[leftmargin=*]
\item Section \ref{sa-sec: REG -- proj} introduces the \emph{conditional mean adjusted product-factorized projection} that combines the \emph{product-factorized projection} for the $G_n(g,r)$ part and the $L_2$ projection for the $X_n(g \,\theta(\cdot,r))$ part.
\item Section~\ref{sa-sec: REG -- sa construction} constructs the Gaussian process $(Z_n^R(g, r): (g, r) \in \mathscr{G} \times \mathscr{R})$. The construction is essentially the same as those in Section~\ref{sa-sec: X-process -- Strong Approximation Constructions}, relying on coupling binomial random variables with Gaussian random variables.
\item Section~\ref{sa-sec: REG -- meshing error} handles the meshing errors $\lVert R_n - R_n\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta} \rVert_{\mathscr{G} \times \mathscr{R}}$ and $\lVert Z_n^R \circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta}-Z_n^R \rVert_{\mathscr{G} \times \mathscr{R}}$ using standard empirical process results.
\item Section~\ref{sa-sec: REG -- sa error} addresses the strong approximation error $\lVert \mathtt{\Pi}_2 R_n - \mathtt{\Pi}_2 Z_n^R \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}$. With the help of the relation between $\mathtt{\Pi}_{1}$ and $\mathtt{\Pi}_2$, we can reuse results from Section~\ref{sa-sec: MULT -- sa error}.
\item Section~\ref{sa-sec: REG -- proj error} addresses the projection errors $\lVert R_n -
\mathtt{\Pi}_2 R_n \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$ and $\lVert Z_n^R - \mathtt{\Pi}_{1} Z_n^R \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$. We use the results from Section~\ref{sa-sec: MULT -- proj error} for $\lVert G_n -
\mathtt{\Pi}_{1} G_n \rVert_{(\mathscr{G} \times \mathscr{R})_\delta}$, and deal with $\lVert X_n(g \theta(\cdot,r)) - \mathtt{\Pi}_{0} X_n(g \theta(\cdot,r)) \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}$ using results from Section~\ref{sa-sec: X-process -- proj error}.
\end{itemize}
\subsection{Preliminary Technical Results}\label{sa-sec: REG -- preliminary}
This section presents preliminary technical results that are used to prove Theorem~\ref{sa-thm: M-process -- main theorem}. Whenever possible, these results are presented at a higher level of generality, and therefore may be of independent theoretical interest. Throughout this section, we assume the same set of conditions (Assumption~\ref{sa-assump: MULT & REG -- step}) on data generate process as in Section~\ref{sa-sec: MULT -- preliminary}.
Compared to the assumptions in Theorem 2, this assumption does not require the existence of a surrogate measure or a normalizing transformation. It will be applied in the analysis of terms in the error decomposition, where we work with the $\mathbbm{P}_Z$ distribution and extra condition on the existence of Lebesgue density of $\mathbbm{P}_X$ is assumed whenever necessary (Section~\ref{sa-sec: REG -- proj error}). The surrogate measure and the normalizing transformation will be used in the proof of Theorem~\ref{sa-thm: M-process -- main theorem} with the help of Section~\ref{sa-sec: X-Process -- normalizing transformation}, providing greater flexibility in the data generating process.
\subsubsection{Projection onto Piecewise Constant Functions}\label{sa-sec: REG -- proj}
For the residual empirical process, we tailor a projection to piecewise constant functions on the quasi-dyadic cells that differs from the mean square projection from Section~\ref{sa-sec: X-process -- Projection onto Piecewise Constant Functions} and the product-factorized projection from Section~\ref{sa-sec: MULT -- proj}. Given a cylindered quasi-dyadic expansion of $\mathbb{R}^{d+1}$, $\mathscr{C}_{M,N}(\mathbbm{P},\rho)$ with $\mathbbm{P}$ the law of random vector $(\mathbf{X},Y) \in \mathbb{R}^d \times \mathbb{R}$, and recall the definition of $\mathscr{E}_{M+N}$ from Section~\ref{sa-sec: X-process -- Projection onto Piecewise Constant Functions}, for any real valued functions $g$ on $\mathbb{R}^d$ and $r$ on $\mathbb{R}$ such that $\int_{\mathbb{R}^d} \int_{\mathbb{R}} g(\mathbf{x})^2 \mathbbm{P}(d y d \mathbf{x}) < \infty$ and $\int_{\mathbb{R}^d}\int_{\mathbb{R}}r(y)^2\mathbbm{P}(d y d \mathbf{x}) < \infty$, the \textit{conditional mean adjusted product-factorized projection} of $g$ and $r$ is defined as
\begin{align}\label{sa-eq: projreg}
\mathtt{\Pi}_2(\mathscr{C}_{M,N}(\mathbbm{P},\rho))[g, r] & = \mathtt{\Pi}_1(\mathscr{C}_{M,N}(\mathbbm{P},\rho))[g, r] - \mathtt{\Pi}_{0}(\mathtt{p}_X[\mathscr{C}_{M,N}(\mathbbm{P}, \rho)])[g \, \theta(\cdot, r)],
\end{align}
where $\theta(\mathbf{x}, r) = \mathbbm{E}[r(Y) | \mathbf{X} = \mathbf{x}]$ for $r \in \mathscr{R}$ and $\mathbf{x} \in \mathcal{X}$, and $\mathtt{p}_X[\mathscr{C}_{M,N}(\mathbbm{P}, \rho)] = \{\mathcal{X}_{l,k}: 0 \leq l \leq M, 0 \leq k < 2^{M-l}\}$ as defined in Definition~\ref{sa-defn: cylindered quasi dyadic expansion}. We denote the collection of conditional mean functions based on $\mathscr{R}$ by $\mathscr{V}_{\mathscr{R}} = \{\theta(\cdot,r): r \in \mathscr{R}\}$.
This projection can also be represented using the Haar basis as
\begin{align*}
\mathtt{\Pi}_2(\mathscr{C}_{M,N}(\mathbbm{P},\rho))[g, r] & = \eta_{M + N,0}(g, r) e_{M + N,0} + \sum_{1 \leq j \leq M + N} \sum_{0 \leq k < 2^{M + N - j}} \widetilde{\eta}_{j,k}(g, r) \widetilde{e}_{j,k},
\end{align*}
with
\begin{align}\label{eq: gamma-eta coeff relation}
\eta_{j,k}(g, r) & =
\begin{cases}
0, & \text{if } N \leq j \leq M+N, \\
\gamma_{j,k}(g, r), & \text{if } j < N.
\end{cases}
\end{align}
We will use $\mathtt{\Pi}_2$ as shorthand for $\mathtt{\Pi}_2(\mathscr{C}_{M,N}(\mathbbm{P},\rho))$.
Next, we define the empirical processes indexed by these projected functions. With a slight abuse of notation, let $(X_n(f) : f \in \mathscr{F})$ be the empirical process based on a random sample $((\mathbf{x}_i, y_i) : 1 \leq i \leq n)$, where $\mathscr{F}$ is a class of real-valued functions on $\mathbb{R}^{d+1}$. Specifically, $X_n(f) = n^{-1/2} \sum_{i = 1}^n (f(\mathbf{x}_i, y_i) - \mathbbm{E}[f(\mathbf{x}_i, y_i)])$ for $f \in \mathscr{F}$. For any real valued functions $g$ on $\mathbb{R}^d$ and $r$ on $\mathbb{R}$ such that $\int_{\mathbb{R}^d} \int_{\mathbb{R}} g(\mathbf{x})^2 \mathbbm{P}(d y d \mathbf{x}) < \infty$ and $\int_{\mathbb{R}^d}\int_{\mathbb{R}}r(y)^2\mathbbm{P}(d y d \mathbf{x}) < \infty$, we define
\begin{equation}\label{sa-eq: proj processes reg}
\begin{aligned}
\mathtt{\Pi}_2 R_n(g, r) & = X_n \circ \mathtt{\Pi}_2(g, r), \\
\mathtt{\Pi}_{0} R_n(g, r) & = X_n \circ \mathtt{\Pi}_{0}[\mathscr{C}_{M,N}(\mathbbm{P}, \rho)](g r) - X_n \circ \mathtt{\Pi}_{0}(\mathtt{p}_X[\mathscr{C}_{M,N}(\mathbbm{P}, \rho)])[g \, \theta(\cdot, r)].
\end{aligned}
\end{equation}
\subsubsection{Strong Approximation Constructions}\label{sa-sec: REG -- sa construction}
\begin{lemma}\label{sa-lem: R- processes pregaussian dyadic}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds, and a cylindered quasi-dyadic expansion $\mathscr{C}_{K}(\mathbbm{P}_Z,\rho)$ is given. Then, $(\mathscr{G} \cdot \mathscr{R}) \cup (\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}) \cup \mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}) \cup \mathtt{\Pi}_2(\mathscr{G} \times \mathscr{R}) \cup \mathtt{\Pi}_{0}[\mathtt{p}_X(\mathscr{C}_{M,N})](\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}})$ is $\mathbbm{P}_Z$-pregaussian.
\end{lemma}
\begin{proof}
Recall we have shown in the proof of Lemma~\ref{sa-lem: M- processes pregaussian dyadic} that for all $0 < \delta < 1$,
\begin{align*}
J_{\mathcal{X} \times \mathcal{Y}}(\mathscr{G} \cdot \mathscr{R}, \mathtt{M}_{\mathscr{G},\mathcal{X}} M_{\mathscr{R},\mathcal{Y}}, \delta) & \lesssim \sqrt{2}J_{\mathcal{X}}(\mathscr{G}, \mathtt{M}_{\mathscr{G},\mathcal{X}}, \delta/\sqrt{2}) + \sqrt{2} J_{\mathcal{Y}}(\mathscr{R}, M_{\mathscr{R},\mathcal{Y}}, \delta/\sqrt{2}), \\
J_{\mathcal{X} \times \mathcal{Y}}(\mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}), c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G},\mathcal{X}} N^{\alpha}, \delta) & \lesssim \sqrt{2} J_{\mathcal{X}}(\mathscr{G}, \mathtt{M}_{\mathscr{G},\mathcal{X}}, \delta/(3 \sqrt{2})) + \sqrt{2} J_{\mathcal{Y}}(\mathscr{R}, M_{\mathscr{R},\mathcal{Y}}, \delta/(3 \sqrt{2})),
\end{align*}
where $c_{\mathtt{v},\alpha} = \mathtt{v}(1 + (2 \alpha)^{\frac{\alpha}{2}})$. Lemma~\ref{sa-lem: entropy conditional mean} implies $J_{\mathcal{X}}(\mathscr{V}_{\mathscr{R}}, \theta(\cdot,M_{\mathscr{R},\mathcal{Y}}), \delta) \leq J_{\mathcal{Y}}(\mathscr{R}, M_{\mathscr{R},\mathcal{Y}}, \delta)$. Since Assumption~\ref{sa-assump: MULT & REG -- step} (iv) implies $\sup_{\mathbf{x} \in \mathcal{X}}\theta(\cdot,M_{\mathscr{R},\mathcal{Y}}) \leq c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}}$, we know for all $0 < \delta < 1$,
\begin{align*}
J_{\mathcal{X}}(\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}, c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}}, \delta) & \leq \sqrt{2} J_{\mathcal{X}}(\mathscr{G}, \mathtt{M}_{\mathscr{G},\mathcal{X}}, \delta/\sqrt{2}) + \sqrt{2} J_{\mathcal{X}}(\mathscr{V}_{\mathscr{R}}, \theta(\cdot,M_{\mathscr{R},\mathcal{Y}}), \delta/\sqrt{2})\\
& \leq \sqrt{2} J_{\mathcal{X}}(\mathscr{G}, \mathtt{M}_{\mathscr{G},\mathcal{X}}, \delta/\sqrt{2}) + \sqrt{2} J_{\mathcal{Y}}(\mathscr{R}, M_{\mathscr{R},\mathcal{Y}}, \delta/\sqrt{2}).
\end{align*}
The same argument for Lemma~\ref{sa-lem: M- processes pregaussian dyadic} implies that for all $0 < \delta < 1$,
\begin{align*}
J_\mathcal{X}(\mathtt{\Pi}_{0}[\mathtt{p}_X(\mathscr{C}_{M,N})](\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}), C_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G},\mathcal{X}} N^{\alpha}, \delta) \leq J_\mathcal{X}(\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}, C_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G},\mathcal{X}} N^{\alpha}, \delta).
\end{align*}
Moreover Lemma~\ref{sa-lem: vc to rho} implies $\mathtt{\Pi}_2(\mathscr{G} \times \mathscr{R}) \subseteq \mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}) + \mathtt{\Pi}_{0}[\mathtt{p}_X(\mathscr{C}_{M,N})](\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}})$. It follows from pointwise separability of $\mathscr{G}$ and $\mathscr{R}$ and Corollary 2.2.9 in \cite{wellner2013weak-SA} that $(\mathscr{G} \cdot \mathscr{R}) \cup (\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}) \cup \mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}) \cup \mathtt{\Pi}_2(\mathscr{G} \times \mathscr{R}) \cup \mathtt{\Pi}_{0}[\mathtt{p}_X(\mathscr{C}_{M,N})](\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}})$ is $\mathbbm{P}_Z$-pregaussian.
\end{proof}
\begin{lemma}\label{sa-lem: R- processes sa for pcw-const}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds and a cylindered dyadic expansion $\mathscr{C}_{M,N}(\mathbbm{P}_Z, 1)$ is given. Then on a possibly enlarged probability space, there exists a $\mathbbm{P}_Z$-Brownian bridge $B_n$ indexed by $\mathscr{F} = (\mathscr{G} \cdot \mathscr{R}) \cup \mathtt{\Pi}_{0}(\mathscr{G} \times \mathscr{R}) \cup \mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R})$ with almost sure continuous trajectories on $(\mathscr{F},\mathfrak{d}_{\mathbbm{P}_Z})$ such that for any $f \in \mathscr{F}$ and any $x > 0$,
\begin{equation*}
\begin{split}
\mathbbm{P} \left(\bigg|\sum_{i = 1}^n f(\mathbf{x}_i, y_i) - \sqrt{n} B_n(f)\bigg| \geq 24 \sqrt{\lVert f \rVert^2_{\mathscr{E}_{M+N}}x} + 4 \sqrt{\mathtt{C}_{\{f\},M+N}}x \right)
\leq 2 \exp(-x),
\end{split}
\end{equation*}
where for both $\lVert f \rVert_{\mathscr{E}_{M+N}}^2$ and $\mathtt{C}_{\{f\},M+N}$ are defined in Lemma~\ref{sa-lem: X-process -- sa for pcw-const}.
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: R- processes sa for pcw-const}}
The result follows from Lemma~\ref{sa-lem: R- processes pregaussian dyadic} and the same argument as Lemma~\ref{sa-lem: M- processes sa for pcw-const}.
\end{myproof}
\begin{lemma}\label{sa-lem: R- processes sa for pcw-const quasi}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds and a cylindered quasi-dyadic expansion $\mathscr{C}_{M,N}(\mathbbm{P}_Z, \rho)$ with $\rho > 1$ is given. Then on a possibly enlarged probability space, there exists a Brownian bridge $B_n$ indexed by $\mathscr{F} = (\mathscr{G} \cdot \mathscr{R}) \cup (\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}) \cup \mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}) \cup \mathtt{\Pi}_2(\mathscr{G} \times \mathscr{R}) \cup \mathtt{\Pi}_{0}[\mathtt{p}_X(\mathscr{C}_{M,N})](\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}})$ with almost sure continuous trajectories on $(\mathscr{F},\mathfrak{d}_{\mathbbm{P}_Z})$ such that for any $f \in \mathscr{F}$ and any $x > 0$,
\begin{equation*}
\begin{split}
\mathbbm{P} \left(\bigg|\sum_{i = 1}^n f(\mathbf{x}_i, y_i) - \sqrt{n} B_n(f)\bigg| \geq C_{\rho} \sqrt{\lVert f \rVert^2_{\mathscr{E}_{M+N}}x} + C_{\rho}\sqrt{\mathtt{C}_{\{f\},M+N}}x \right) \\
\leq 2 \exp(-x) + 2^{M + 2}\exp\left(- C_{\rho} n 2^{-M} \right),
\end{split}
\end{equation*}
where $C_{\rho}$ is a constant that only depends on $\rho$.
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: R- processes sa for pcw-const quasi}}
The result follows from Lemma~\ref{sa-lem: R- processes pregaussian dyadic} and the same argument as Lemma~\ref{sa-lem: M- processes sa for pcw-const quasi}.
\end{myproof}
The above two lemmas enable the construction of Gaussian processes and their projected counterparts as analogs to the empirical processes defined in Section~\ref{sa-sec: X-process -- Strong Approximation Constructions} and Section~\ref{sa-sec: MULT -- sa construction}. In particular, we define \( Z_n^R \) and \( \mathtt{\Pi}_2 Z_n^R \) as Gaussian processes indexed by \( \mathscr{G} \times \mathscr{R} \) such that, for any \( g \in \mathscr{G} \) and \( r \in \mathscr{R} \),
\begin{align}\label{sa-eq: R- Process -- Gaussian Process 1}
\nonumber Z_n^R(g, r) &= B_n(g(r - \theta(\cdot, r))), \\
\mathtt{\Pi}_2 Z_n^R(g, r) &= B_n(\mathtt{\Pi}_2[g, r]).
\end{align}
We also define the following ancillary processes for analysis:
\begin{gather} \label{sa-eq: R- Process -- Gaussian Process 2}
\nonumber Z_n^G(g, r) = B_n(g r), \qquad \mathtt{\Pi}_{1} Z_n^G(g, r) = B_n(\mathtt{\Pi}_{1}[g, r]), \\
Z_n^X(g \, \theta(\cdot, r)) = B_n(g \, \theta(\cdot, r)), \qquad \mathtt{\Pi}_{0} Z_n^X(g \, \theta(\cdot, r)) = B_n(\mathtt{\Pi}_{0}[\mathtt{p}_X(\mathscr{C}_{M,N})][g \, \theta(\cdot, r)]).
\end{gather}
Since for any $g_1,g_2 \in \mathscr{G}$, $r_1, r_2 \in \mathscr{R}$,
\begin{align*}
\mathfrak{d}_{\mathbbm{P}_Z}(g_1(r_1 - \theta(\cdot,r_1)),g_2(r_2 - \theta(\cdot,r_2))) \leq 2 \mathfrak{d}_{\mathbbm{P}_Z}(g_1 r_1,g_2 r_2),
\end{align*}
and $B_n$ has almost sure continuous sample trajectories on $(\mathscr{G} \cdot \mathscr{R}, \mathfrak{d}_{\mathbbm{P}_Z})$, Equation~\eqref{sa-eq: R- Process -- Gaussian Process 1} also implies $(Z_n^R(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ has almost sure continuous sample trajectories on $(\mathscr{G} \times \mathscr{R}, \mathfrak{d}_{\mathbbm{P}_Z})$.
The following ancillary lemma on uniform covering number of the class of conditional means is used for the proof of Lemma~\ref{sa-lem: R- processes pregaussian dyadic}.
\begin{lemma}\label{sa-lem: entropy conditional mean}
Suppose \(\mathscr{S}\) is a class of functions from a measurable space \((\mathcal{Y},\mathcal{B}(\mathcal{Y}))\) to \(\mathbb{R}\), where \(\mathcal{Y} \subseteq \mathbb{R}\), with envelope function \(M_{\mathscr{S},\mathcal{Y}}\). Let \(\mathscr{V}_{\mathscr{S}}\) be the class of conditional means \(\{\theta(\cdot, s): s \in \mathscr{S}\}\) with \(\theta(\mathbf{x}, s) = \mathbbm{E}[s(y_i) | \mathbf{x}_i = \mathbf{x}]\) for \(\mathbf{x} \in \mathcal{X}\). Then
\begin{gather*}
\mathtt{N}_{\mathscr{V}_{\mathscr{S}},\mathcal{X}}(\delta, \theta(\cdot, M_{\mathscr{S},\mathcal{Y}})) \leq \mathtt{N}_{\mathscr{S},\mathcal{Y}}(\delta, M_{\mathscr{S},\mathcal{Y}}).
\end{gather*}
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: entropy conditional mean}}
Let \(\mathcal{Q}\) be a finite discrete measure on \(\mathbb{R}^d\), and let \(r, s \in \mathscr{S}\). Define a new probability measure \(\widetilde{P}\) on \(\mathbb{R}\) by
\begin{align*}
\widetilde{P}(A) = \int \mathbbm{E}[\mathbbm{1}((\mathbf{x}_i, y_i) \in \mathbb{R}^{d} \times A) | \mathbf{x}_i = \mathbf{x}] \, d \mathcal{Q}(\mathbf{x}), \qquad \forall A \subseteq \mathbb{R}^{d}.
\end{align*}
Then \(\int |\theta(\cdot, M_{\mathscr{S},\mathcal{Y}})| \, d \widetilde{P} \leq \int_{\mathbb{R}^{d}} \mathbbm{E}[M_{\mathscr{S},\mathcal{Y}}(y_i) | \mathbf{x}_i = \mathbf{x}] \, d \mathcal{Q}(z) < \infty\), since \(\sup_{m \in \mathscr{V}_S} \|m\|_{\infty} < \infty\).
For \(r, s \in \mathscr{S}\), we have
\begin{align*}
\int |\theta(\cdot, r) - \theta(\cdot, s)|^2 \, d \mathcal{Q} \leq \int_{\mathbb{R}^{d}} \mathbbm{E}[|r(y_i) - s(y_i)|^2 | \mathbf{x}_i = \mathbf{x}] \, d \mathcal{Q}(x) = \int |r - s|^2 \, d \widetilde{P}.
\end{align*}
Here, \(\widetilde{P}\) is not necessarily finite or discrete, but by a similar argument as in Lemma~\ref{sa-lem: vc to rho}, there exists a subset \(\mathscr{S}_{\varepsilon} \subseteq \mathscr{S}\) with cardinality no greater than \(\mathtt{N}_{\mathscr{S},\mathcal{Y}}(\delta, M_{\mathscr{S},\mathcal{Y}})\), such that for any \(s \in \mathscr{S}\), there exists \(r \in \mathscr{S}_{\varepsilon}\) with \(\|r - s\|_{\widetilde{P},2} \leq \varepsilon \|\theta(\cdot, M_{\mathscr{S},\mathcal{Y}})\|_{\widetilde{P},2}\). Hence, \(\|m_r - m_s\|_{\mathcal{Q},2} \leq \varepsilon \|\theta(\cdot, M_{\mathscr{S},\mathcal{Y}})\|_{\widetilde{P},2} = \varepsilon \|\theta(\cdot, M_{\mathscr{S},\mathcal{Y}})\|_{\mathcal{Q},2}\). The conclusion then follows.
\end{myproof}
\subsubsection{Meshing Error}\label{sa-sec: REG -- meshing error}
To simplify notation, the parameters of \(\mathscr{G}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{X}\), and the index \(\mathcal{X}\) is omitted where there is no ambiguity; the parameters of \(\mathscr{R}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{Y}\), and the index \(\mathcal{Y}\) is omitted where there is no ambiguity; and the parameters of \(\mathscr{G} \times \mathscr{R}\) (Definitions 4 to 12, \ref{sa-defn: uniform covering number for product space}, \ref{sa-defn: uniform entropy integral for product space}) are taken with \(\mathcal{C} = \mathcal{X} \times \mathcal{Y}\), and the index \(\mathcal{X} \times \mathcal{Y}\) is omitted where there is no ambiguity. We also define
\begin{gather*}
J(\delta) = \sqrt{2}J(\mathscr{G},\mathtt{M}_{\mathscr{G}},\delta/\sqrt{2}) + \sqrt{2}J(\mathscr{R},M_{\mathscr{R}}, \delta/\sqrt{2}), \qquad \delta \in (0,1],\\
\mathtt{N}(\delta) = \mathtt{N}_{\mathscr{G}}(\delta/\sqrt{2}, \mathtt{M}_{\mathscr{G}}) \mathtt{N}_{\mathscr{R}}(\delta/\sqrt{2},M_{\mathscr{R}}), \qquad \delta \in (0,1].
\end{gather*}
For \(0 < \delta \leq 1\), consider a \(\delta \mathtt{M}_{\mathscr{G}} \|M_{\mathscr{R}}\|_{\mathbbm{P}_Y
,2}\)-net of \((\mathscr{G} \times \mathscr{R}, \|\cdot\|_{\mathbbm{P}_Z,2})\), denoted by \((\mathscr{G} \times \mathscr{R})_{\delta}\), with cardinality at most \(\mathtt{N}_{\mathscr{G} \times \mathscr{R}}(\delta, \mathtt{M}_{\mathscr{G}} \|M_{\mathscr{R}}\|_{\mathbbm{P}_Y,2})\). Define the projection onto the \(\delta\)-net as a mapping \(\pi_{(\mathscr{G} \times \mathscr{R})_{\delta}} : \mathscr{G} \times \mathscr{R} \to \mathscr{G} \times \mathscr{R}\) such that \(\|\pi_{(\mathscr{G} \times \mathscr{R})_{\delta}}(g, r) - g r\|_{\mathbbm{P}_Z,2} \leq \delta \mathtt{M}_{\mathscr{G}} \|M_{\mathscr{R}}\|_{\mathbbm{P}_Y,2}\) for all \(g \in \mathscr{G}\) and \(r \in \mathscr{R}\).
\begin{lemma}\label{sa-lem: R- Process -- Fluctuation off the net}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds, a cylindered quasi-dyadic expansion $\mathscr{C}_{M,N}(\mathbbm{P}_Z, \rho)$ is given, $(Z_n^R(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ is the Gaussian process constructed as in \eqref{sa-eq: R- Process -- Gaussian Process 1} on a possibly enlarged probability space, and $(\mathscr{G} \times \mathscr{R})_\delta$ is chosen in Section~\ref{sa-sec: MULT -- meshing error}. For all \(t > 0\) and \(0<\delta<1\),
\begin{align*}
\mathbbm{P}\big[\lVert R_n - R_n \circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta} \rVert_{\mathscr{G} \times \mathscr{R}} + \lVert Z_n^R\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta}-Z_n^R \rVert_{\mathscr{G} \times \mathscr{R}}> C_1 c_{\mathtt{v},\alpha} \mathsf{F}_n^R(t,\delta)\big] &\leq \exp(-t),
\end{align*}
where \(c_{\mathtt{v},\alpha} = \mathtt{v}(1 + (2 \alpha)^{\frac{\alpha}{2}})\) and
\begin{align*}
\mathtt{F}_n^R(t,\delta) = J(\delta) \mathtt{M}_{\mathscr{G}} + \frac{(\log n)^{\alpha/2} \mathtt{M}_{\mathscr{G}} J^2(\delta)}{\delta^2 \sqrt{n}} + \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} t + (\log n)^{\alpha} \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} t^{\alpha}.
\end{align*}
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: R- Process -- Fluctuation off the net}}
Recall for any \(g \in \mathscr{G}\), \(r \in \mathscr{R}\),
\begin{align*}
R_n(g,r) = G_n(g,r) + X_n[\mathtt{p}_X(\mathcal{C}_{M,N}(\mathbbm{P}_Z,\rho))](g \, \theta(\cdot, r)).
\end{align*}
Lemma~\ref{sa-lem: M- Process -- Fluctuation off the net} implies that for any \(t > 0\) and \(0 < \delta < 1\),
\begin{align*}
\mathbbm{P}\big[\lVert G_n - G_n \circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta} \rVert_{\mathscr{G} \times \mathscr{R}} + \lVert Z_n^G\circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta}-Z_n^G \rVert_{\mathscr{G} \times \mathscr{R}}> C_1 c_{\mathtt{v},\alpha} \mathsf{F}_n^R(t,\delta)\big] &\leq \exp(-t).
\end{align*}
For \(g \in \mathscr{G}\), \(r \in \mathscr{R}\), and take \((g_0, r_0) = \pi_{(\mathscr{G} \times \mathscr{R})_\delta}\), Jensen's inequality implies
\begin{align*}
& \qquad \lVert X_n(g \, \theta(\cdot, r)) - X_n(g_0 \, \theta(\cdot,r_0)) \rVert_{2}^2 \\
& = \frac{1}{n}\sum_{i = 1}^n \mathbbm{E} [(g(\mathbf{x}_i) \mathbbm{E}[r(y_i)|\mathbf{x}_i] - g_0(\mathbf{x}_i) \mathbbm{E}[r_0(y_i)|\mathbf{x}_i])^2] \\
& \leq \frac{1}{n} \sum_{i = 1}^n \mathbbm{E} [(g(\mathbf{x}_i) r(y_i) - g_0(\mathbf{x}_i) r_0(y_i))^2] = \lVert G_n(g,r) - G_n(g_0,r_0) \rVert_{2}^2.
\end{align*}
Thus,
\begin{align*}
\lVert \lVert X_n(g \, \theta(\cdot, r)) - X_n \circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta}(g \, \theta(\cdot, r)) \rVert_2 \rVertVert_2}_{\mathscr{G} \times \mathscr{R}} \leq \lVert \lVert G_n - G_n \circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta} \rVert_2 \rVertVert_2}_{\mathscr{G} \times \mathscr{R}}.
\end{align*}
Lemma~\ref{sa-lem: entropy conditional mean} implies that if we define \(\mathscr{G} \times \overline{\mathscr{R}} = \{g (r - \theta(\cdot,r)): g \in \mathscr{G}, r \in \mathscr{R}\}\), then
\begin{align*}
\mathtt{N}_{\mathscr{G} \times \overline{\mathscr{R}},\mathcal{X} \times \mathcal{Y}}(\delta,\mathtt{M}_{\mathscr{G}}M_{\mathscr{R}}) \leq 2 \mathtt{N}(\delta).
\end{align*}
The conclusion then follows by applying the same empirical process argument to \(\lVert X_n(g \, \theta(\cdot, r)) - X_n \circ\pi_{(\mathscr{G} \times \mathscr{R})_\delta} \rVert_{\mathscr{G} \times \mathscr{R}}\) as in Lemma~\ref{sa-lem: M- Process -- Fluctuation off the net}.
\end{myproof}
\subsubsection{Strong Approximation Errors}\label{sa-sec: REG
-- sa error}
To simplify notation, the parameters of \(\mathscr{G}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{X}\), and the index \(\mathcal{X}\) is omitted where there is no ambiguity; the parameters of \(\mathscr{R}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{Y}\), and the index \(\mathcal{Y}\) is omitted where there is no ambiguity; and the parameters of \(\mathscr{G} \times \mathscr{R}\) (Definitions 4 to 12, \ref{sa-defn: uniform covering number for product space}, \ref{sa-defn: uniform entropy integral for product space}) are taken with \(\mathcal{C} = \mathcal{X} \times \mathcal{Y}\), and the index \(\mathcal{X} \times \mathcal{Y}\) is omitted where there is no ambiguity. Recall we also define
\begin{gather*}
J(\delta) = \sqrt{2}J(\mathscr{G},\mathtt{M}_{\mathscr{G}},\delta/\sqrt{2}) + \sqrt{2}J(\mathscr{R},M_{\mathscr{R}}, \delta/\sqrt{2}), \qquad \delta \in (0,1],\\
\mathtt{N}(\delta) = \mathtt{N}_{\mathscr{G}}(\delta/\sqrt{2}, \mathtt{M}_{\mathscr{G}}) \mathtt{N}_{\mathscr{R}}(\delta/\sqrt{2},M_{\mathscr{R}}), \qquad \delta \in (0,1].
\end{gather*}
\begin{lemma}\label{sa-lem: R- process -- SA error}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds, a cylindered dyadic expansion $\mathscr{C}_{M,N}(\mathbbm{P}_Z, 1)$ is given, $(Z_n^R(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ and $(\mathtt{\Pi}_2 Z_n^R(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ are the Gaussian processes constructed as in Equations \eqref{sa-eq: M- Process -- Gaussian Process 1} and \eqref{sa-eq: M- Process -- Gaussian Process 2} on a possibly enlarged probability space, and $(\mathscr{G} \times \mathscr{R})_\delta$ is chosen in Section~\ref{sa-sec: MULT -- meshing error}. Then for all $t > 0$,
\begin{align*}
\mathbbm{P}\Big[\lVert \mathtt{\Pi}_2 R_n - \mathtt{\Pi}_2 Z_n^R \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
> C_1 c_{\mathtt{v},\alpha} \sqrt{\frac{N^{2\alpha + 1} 2^{M} \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}}{n} t}
+ C_1 c_{\mathtt{v},\alpha}\sqrt{\frac{\mathtt{C}_{\mathtt{\Pi}_2(\mathscr{G} \times \mathscr{R})_{\delta},M+N}}{n}}t \Big]
\leq 2 \mathtt{N}(\delta) e^{-t},
\end{align*}
where $C_1 > 0$ is a universal constant and $c_{\mathtt{v},\alpha} = \mathtt{v}(1 + (2 \alpha)^{\alpha/2})$.
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: R- process -- SA error}} We have shown in the proof of Lemma~\ref{sa-lem: M- process -- SA error} that for any $(g,r) \in \mathscr{G} \times \mathscr{R}$,
\begin{align*}
\sum_{j = 1}^{M + N} \sum_{0 \leq k < 2^{M + N -j}}|\widetilde{\gamma}_{j,k}(g,r)|^2 \leq c_{\mathtt{v},\alpha}^2 N^{2 \alpha + 1} 2^{M} \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}.
\end{align*}
It then follows from the relation between $\gamma_{j,k}$ and $\eta_{j,k}$ in Equation~\eqref{eq: gamma-eta coeff relation} that for any $(g,r) \in \mathscr{G} \times \mathscr{R}$,
\begin{align*}
\sum_{j = 1}^{M + N} \sum_{0 \leq k < 2^{M + N -j}}|\widetilde{\eta}_{j,k}(g,r)|^2 \leq c_{\mathtt{v},\alpha}^2 N^{2 \alpha + 1} 2^{M} \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}},
\end{align*}
and hence by Lemma~\ref{sa-lem: R- processes sa for pcw-const}, for any $x > 0$, with probability at least $1 - 2 \exp(-x)$,
\begin{align*}
& |\mathtt{\Pi}_2 R_n(g, r) - \mathtt{\Pi}_2 Z_n(g, r) | \leq c_{\mathtt{v},\alpha} \sqrt{\frac{N^{2\alpha + 1} 2^{M} \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}}{n}x} + c_{\mathtt{v},\alpha}\sqrt{\frac{\mathtt{C}_{\mathtt{\Pi}_2\{(g,r)\},M+N}}{n}}x.
\end{align*}
The conclusion then follows from a union bound on $(\mathscr{G} \times \mathscr{R})_{\delta}$.
\end{myproof}
\begin{lemma}\label{sa-lem: R- process -- SA error quasi-dyadic}
Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds, a cylindered quasi-dyadic expansion $\mathscr{C}_{M,N}(\mathbbm{P}_Z, \rho)$ is given with $\rho > 1$, $(Z_n^R(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ and $(\mathtt{\Pi}_2 Z_n^R(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ are the Gaussian processes constructed as in Equations \eqref{sa-eq: M- Process -- Gaussian Process 1} and \eqref{sa-eq: M- Process -- Gaussian Process 2} on a possibly enlarged probability space, and $(\mathscr{G} \times \mathscr{R})_\delta$ is chosen in Section~\ref{sa-sec: MULT -- meshing error}. Then for all $t > 0$,
\begin{align*}
& \mathbbm{P}\Big[\lVert \mathtt{\Pi}_2 R_n - \mathtt{\Pi}_2 Z_n^R \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
> C_\rho c_{\mathtt{v},\alpha} \sqrt{\frac{N^{2\alpha + 1} 2^{M} \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}}{n} t}
+ C_\rho c_{\mathtt{v},\alpha}\sqrt{\frac{\mathtt{C}_{\mathtt{\Pi}_2(\mathscr{G} \times \mathscr{R})_{\delta},M+N}}{n}} t \Big] \\
& \quad \leq 2 \mathtt{N}(\delta) e^{-t} + 2^{M}\exp \left(-C_{\rho} n 2^{-M}\right),
\end{align*}
where $C_\rho > 0$ is a constant that only depends on $\rho$ and $c_{\mathtt{v},\alpha} = \mathtt{v}(1 + (2 \alpha)^{\alpha/2})$.
\end{lemma}
\begin{proof}
Since $\mathscr{C}_{M,N}(\mathbbm{P}_Z,\rho)$ is a cylindered quasi-dyadic expansion, $\rho^{-1} 2^{-M-N+j} \leq \mathbbm{P}_Z(\mathcal{C}_{j,k}) \leq \rho 2^{-M-N+j}$, for all $0 \leq j \leq M + N$, $0 \leq k < 2^{M + N - j}$. Hence following the argument in the proof for Lemma~\ref{sa-lem: M- process -- SA error}, for any $g \in \mathscr{G}, r \in \mathscr{R}$,
\begin{align*}
\sum_{j = 1}^{M + N} \sum_{k = 0}^{2^{M + N - j}} \widetilde{\eta}_{j,k}^2 (g,r)
\leq \sum_{j = 1}^{M + N} \sum_{k = 0}^{2^{M + N - j}} \widetilde{\gamma}_{j,k}^2 (g,r)
\leq c_{\mathtt{v},\alpha}^2 N^{2\alpha + 1} 2^{M}\mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}.
\end{align*}
The result then follows from Lemma~\ref{sa-lem: M- processes sa for pcw-const quasi}.
\end{proof}
\subsubsection{Projection Error}\label{sa-sec: REG -- proj error}
To simplify notation, the parameters of \(\mathscr{G}\) and \(\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}\) (Definitions 4 to 12, \ref{sa-defn: smooth tv}, \ref{sa-defn: smooth ktv}) are taken with \(\mathcal{C} = \mathcal{X}\), and the index \(\mathcal{X}\) is omitted where there is no ambiguity; the parameters of \(\mathscr{R}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{Y}\), and the index \(\mathcal{Y}\) is omitted where there is no ambiguity; and the parameters of \(\mathscr{G} \times \mathscr{R}\) (Definitions 4 to 12, \ref{sa-defn: uniform covering number for product space}, \ref{sa-defn: uniform entropy integral for product space}) are taken with \(\mathcal{C} = \mathcal{X} \times \mathcal{Y}\), and the index \(\mathcal{X} \times \mathcal{Y}\) is omitted where there is no ambiguity. Recall we also define
\begin{gather*}
J(\delta) = \sqrt{2}J(\mathscr{G},\mathtt{M}_{\mathscr{G}},\delta/\sqrt{2}) + \sqrt{2}J(\mathscr{R},M_{\mathscr{R}}, \delta/\sqrt{2}), \qquad \delta \in (0,1],\\
\mathtt{N}(\delta) = \mathtt{N}_{\mathscr{G}}(\delta/\sqrt{2}, \mathtt{M}_{\mathscr{G}}) \mathtt{N}_{\mathscr{R}}(\delta/\sqrt{2},M_{\mathscr{R}}), \qquad \delta \in (0,1].
\end{gather*}
The projection errors for the $R_n$ and $Z_n^R$ processes can be decomposed by the observation that, for any $g \in \mathscr{G}$ and $r \in \mathscr{R}$,
\begin{align}\label{sa-eq: REG -- proj decomposition}
\nonumber \mathtt{\Pi}_2 R_n(g, r) - R_n(g, r)
& = \Big(\mathtt{\Pi}_{1} G_n(g, r) - G_n(g, r)\Big) - \Big(\mathtt{\Pi}_{0}[\mathtt{p}_X(\mathscr{C}_{M, N})] X_n(g \, \theta(\cdot, r)) - X_n(g \, \theta(\cdot, r))\Big), \\
\mathtt{\Pi}_2 Z_n^R(g, r) - Z_n^R(g, r)
& = \Big(\mathtt{\Pi}_{1} Z_n^G(g, r) - Z_n^G(g, r)\Big) - \Big(\mathtt{\Pi}_{0}[\mathtt{p}_X(\mathscr{C}_{M, N})] Z_n^X(g \, \theta(\cdot, r)) - Z_n^X(g \, \theta(\cdot, r))\Big),
\end{align}
where, in each line, the first term in parentheses is the projection error for the $G_n$-process, as discussed in Section~\ref{sa-sec: MULT -- proj error}, and the second term is the projection error for the $X_n$-process, detailed in Section~\ref{sa-sec: X-process -- proj error}, with
\begin{align*}
X_n(g \, \theta(\cdot,r)) & = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \Big[g(\mathbf{x}_i) \theta(\mathbf{x}_i,r) - \mathbbm{E}[g(\mathbf{x}_i) \theta(\mathbf{x}_i,r)]\Big], \\
\mathtt{\Pi}_{0}[\mathtt{p}_X(\mathscr{C}_{M, N})] X_n(g \, \theta(\cdot, r)) & = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \Big[ \mathtt{\Pi}_{0}[\mathtt{p}_X(\mathscr{C}_{M, N})](g \, \theta(\cdot, r))(\mathbf{x}_i) - \mathbbm{E}[\mathtt{\Pi}_{0}[\mathtt{p}_X(\mathscr{C}_{M, N})](g \, \theta(\cdot, r))(\mathbf{x}_i)]\Big].
\end{align*}
This decomposition allows us to leverage previously established error bounds and convergence results for $G_n$ and $X_n$ processes, thus facilitating the analysis of the $R_n$ and $Z_n^R$ processes. By utilizing known results from Sections~\ref{sa-sec: MULT -- proj error} and \ref{sa-sec: X-process -- proj error}, this approach simplifies the treatment of the projection errors for these new processes.
\begin{lemma}\label{sa-lem: R- process -- projection error} Suppose Assumption~\ref{sa-assump: MULT & REG -- step} holds, a cylindered dyadic expansion $\mathscr{C}_{M,N}(\mathbbm{P}_Z, 1)$ is given, $(Z_n^R(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ and $(\mathtt{\Pi}_2 Z_n^R(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ are the Gaussian processes constructed as in Equations~\eqref{sa-eq: R- Process -- Gaussian Process 1} on a possibly enlarged probability space, and $(\mathscr{G} \times \mathscr{R})_\delta$ is chosen in Section~\ref{sa-sec: MULT -- meshing error}. Suppose $\mathbbm{P}_X$ admits a Lebesgue density $f_X$ supported on $\mathcal{X} \subseteq \mathbb{R}^d$. Then for all $t > N$, with probability at least $1 - 4 \mathtt{N}(\delta) n e^{-t}$,
\begin{gather*}
\lVert R_n - \mathtt{\Pi}_2 R_n \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
\lesssim \sqrt{\mathtt{V}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}}t^{\frac{1}{2}}+ \sqrt{c_{\mathtt{v}, 2 \alpha}} \sqrt{N^2 \mathtt{V}_{\mathscr{G}} + 2^{-N} \mathtt{M}_{\mathscr{G}}^2} t^{\alpha + \frac{1}{2}} + c_{\mathtt{v},\alpha}\frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}}t^{\alpha + 1},\\
\lVert Z_n^R - \mathtt{\Pi}_2 Z_n^R \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
\lesssim \sqrt{\mathtt{V}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}}t^{\frac{1}{2}} + \sqrt{c_{\mathtt{v}, 2 \alpha}} \sqrt{N^2 \mathtt{V}_{\mathscr{G}} + 2^{-N} \mathtt{M}_{\mathscr{G}}^2} t^{\frac{1}{2}} + c_{\mathtt{v},\alpha}\frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}}t,
\end{gather*}
where $c_{\mathtt{v},\alpha} = \mathtt{v}(1 + (2 \alpha)^{\frac{\alpha}{2}})$, $c_{\mathtt{v}, 2 \alpha} = \mathtt{v}^2(1 + (4 \alpha)^{\alpha})$, and
\begin{align*}
\mathtt{V}_{\mathscr{G}} & = \min\{2 \mathtt{M}_{\mathscr{G}}, \mathtt{L}_{\mathscr{G}}\lVert \mathcal{V}_M \rVert_{\infty}\} \Big(\sup_{\mathbf{x} \in \mathcal{X}} f_X(\mathbf{x})\Big)^2 2^M \mathfrak{m}(\mathcal{V}_M) \lVert \mathcal{V}_M \rVert_{\infty} \mathtt{TV}_{\mathscr{G}}^*, \\
\mathtt{V}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}} & = \min\{2 \mathtt{M}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}, \mathtt{L}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}\lVert \mathcal{V}_M \rVert_{\infty}\} \left(\sup_{\mathbf{x} \in \mathcal{X}}f_X(\mathbf{x})\right)^2 2^{M} \mathfrak{m}(\mathcal{V}_M) \lVert \mathcal{V}_M \rVert_{\infty} \mathtt{TV}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}^*,
\end{align*}
with $\mathcal{V}_M = \cup_{0 \leq l < 2^{M}} (\mathcal{X}_{0,l} - \mathcal{X}_{0,l})$ the upper level quasi-dyadic variation set as in Section~\ref{sa-sec: X-process -- proj error}.
\end{lemma}
\begin{proof}
By Equations~\eqref{sa-eq: projreg} and \eqref{sa-eq: proj processes reg}, we can show the decomposition in Equation~\eqref{sa-eq: REG -- proj decomposition} holds. The terms $\mathtt{\Pi}_{1} G_n - G_n$ and $\mathtt{\Pi}_{1} Z_n^G - Z_n^G$ can be bounded from results in Lemma~\ref{sa-lem: M- process -- projection error}. Recall $\mathscr{G} \cdot \mathcal{V}_{\mathscr{R}} = \{g \, \theta(\cdot,r): g \in \mathscr{G}, r \in \mathscr{R}\}$. We know from Lemma~\ref{sa-lem: X-process -- projection error} for all $t > 0$,
\begin{gather*}
\mathbbm{P} \left( |\mathtt{\Pi}_{0}[\mathtt{p}_X(\mathscr{C}_{M,N})] X_n(g \, \theta(\cdot,r)) - X_n(g \, \theta(\cdot,r))|
\geq 2\sqrt{\mathtt{V}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}t} + \frac{4}{3}\cdot \frac{\mathtt{M}_{\mathscr{G} \cdot \mathcal{V}_{\mathscr{R}}}}{\sqrt{n}} t\right) \leq 2 \exp(-t), \\
\mathbbm{P} \left( |\mathtt{\Pi}_{0}[\mathtt{p}_X(\mathscr{C}_{M,N})] Z_n^X(g \, \theta(\cdot,r)) - Z_n^X(g \, \theta(\cdot,r))|
\geq 2\sqrt{\mathtt{V}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}t}\right) \leq 2 \exp(-t).
\end{gather*}
Moreover, suppose $\alpha > 0$ in (iv) from Assumption~\ref{sa-assump: MULT & REG -- step}, $\sup_{\mathbf{x} \in \mathcal{X}}\mathbbm{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] \leq 2$, hence by moment properties of sub-Gaussian random variables,
\begin{align*}
\sup_{r \in \mathscr{R}} \sup_{\mathbf{x}\in \mathcal{X}} \mathbbm{E}[|r(y_i)||\mathbf{x}_i = \mathbf{x}] \leq \mathtt{v}(1 + \sup_{\mathbf{x} \in \mathcal{X}}\mathbbm{E}[|y_i|^{\alpha}|\mathbf{x}_i = \mathbf{x}] ) \leq c_{\mathtt{v},\alpha}.
\end{align*}
Hence $\mathtt{M}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}} \leq c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}}$. Suppose $\alpha = 0$ from (iv) from Assumption~\ref{sa-assump: MULT & REG -- step} holds, $\sup_{r \in \mathscr{R}} \sup_{\mathbf{x} \in \mathcal{X}}|r(\mathbf{x})| \leq 2 \mathtt{v}$, hence we also have $\mathtt{M}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}} \leq c_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}}$. The result then follows from a union bound over $(\mathscr{G} \times \mathscr{R})_{\delta}$.
\end{proof}
\subsection{General Result}\label{sa-sec: REG -- general result}
The following theorem presents a generalization of Theorem 2 in the paper. To simplify notation, the parameters of \(\mathscr{G}\) and \(\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}\) (Definitions 4 to 12, \ref{sa-defn: smooth tv}, \ref{sa-defn: smooth ktv}) are taken with \(\mathcal{C} = \mathcal{Q}_{\mathscr{G}}\), and the index \(\mathcal{Q}_{\mathscr{G}}\) is omitted where there is no ambiguity; the parameters of \(\mathscr{R}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{Y}\), and the index \(\mathcal{Y}\) is omitted where there is no ambiguity; and the parameters of \(\mathscr{G} \times \mathscr{R}\) (Definitions 4 to 12, \ref{sa-defn: uniform covering number for product space}, \ref{sa-defn: uniform entropy integral for product space}) are taken with \(\mathcal{C} = \mathcal{Q}_{\mathscr{G}} \times \mathcal{Y}\), and the index \(\mathcal{Q}_{\mathscr{G}} \times \mathcal{Y}\) is omitted where there is no ambiguity.
\begin{thm}\label{sa-thm: R-process -- main theorem}
Suppose $(\mathbf{z}_i=(\mathbf{x}_i, y_i): 1 \leq i \leq n)$ are i.i.d. random vectors taking values in $(\mathbb{R}^{d+1}, \mathcal{B}(\mathbb{R}^{d+1}))$ with common law $\mathbbm{P}_Z$, where $\mathbf{x}_i$ has distribution $\mathbbm{P}_X$ supported on $\mathcal{X}\subseteq\mathbb{R}^d$, $y_i$ has distribution $\mathbbm{P}_Y$ supported on $\mathcal{Y}\subseteq\mathbb{R}$, and the following conditions hold.
\begin{enumerate}[label=\emph{(\roman*)}]
\item $\mathscr{G}$ is a real-valued pointwise measurable class of functions on $(\mathbb{R}^d, \mathcal{B}(\mathbb{R}^d), \mathbbm{P}_X)$.
\item There exists a surrogate measure $\mathbb{Q}_\mathscr{G}$ for $\mathbbm{P}_X$ with respect to $\mathscr{G}$ such that $\mathbb{Q}_\mathscr{G} = \operatorname*{\mathfrak{m}} \circ \phi_\mathscr{G}$, where the \textit{normalizing transformation} $\phi_{\mathscr{G}}: \mathcal{Q}_\mathscr{G} \mapsto [0,1]^d$ is a diffeomorphism.
\item $\mathtt{M}_{\mathscr{G}} < \infty$ and $J(\mathscr{G}, \mathtt{M}_{\mathscr{G}},1) < \infty$.
\item $\mathscr{R}$ is a real-valued pointwise measurable class of functions on $(\mathbb{R}, \mathcal{B}(\mathbb{R}),\mathbbm{P}_Y)$.
\item $J(\mathscr{R},M_{\mathscr{R}},1) < \infty$, where $M_{\mathscr{R}}(y) + \mathtt{pTV}_{\mathscr{R},(-|y|,|y|)} \leq \mathtt{v} (1 + |y|^{\alpha})$ for all $y \in \mathcal{Y}$, for some $\mathtt{v}>0$, and for some $\alpha\geq0$. Furthermore, if $\alpha>0$, then $\sup_{\mathbf{x} \in \mathcal{X}}\mathbbm{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] \leq 2$.
\end{enumerate}
Then, on a possibly enlarged probability space, there exists a sequence of mean-zero Gaussian processes $(Z_n^R(g,r): (g,r) \in \mathscr{G} \times \mathscr{R}))$ with almost surely continuous trajectories on $(\mathscr{G} \times \mathscr{R},\mathfrak{d}_{\mathbbm{P}_X,\mathbbm{P}_Y})$ such that:
\begin{itemize}
\item $\mathbbm{E}[R_n(g_1,r_1) R_n(g_2, r_2)] = \mathbbm{E}[Z_n^R(g_1, r_1) Z_n^R(g_2, r_2)]$ for all $(g_1,r_1), (g_2, r_2) \in \mathscr{G} \times \mathscr{R}$.
\item $\mathbbm{P}[\lVert R_n - Z_n^R \rVert_{\mathscr{G} \times \mathscr{R}} > C_1 C_{\mathtt{v},\alpha} \mathsf{T}^R_n(t)] \leq C_2 e^{-t}$ for all $t > 0$,
\end{itemize}
where $C_1$ and $C_2$ are universal constants, $C_{\mathtt{v},\alpha} = \mathtt{v} \max\{1 + (2 \alpha)^{\frac{\alpha}{2}}, 1 + (4 \alpha)^{\alpha}\}$, and
and
\begin{align*}
\mathsf{T}^R_n(t) = \min_{\delta \in (0,1)}\{\mathsf{A}^R_n(t,\delta) + \mathsf{F}^R_n(t,\delta)\}
\end{align*}
with
\begin{align*}
\mathsf{A}^R_n(t,\delta)
&= \sqrt{d} \min \Big\{ \Big( \frac{\mathtt{c}_1^{d} \mathtt{E}_{\mathscr{G}} \mathtt{TV}^{d} \mathtt{M}_{\mathscr{G}}^{d+1}}{n}\Big)^{\frac{1}{2(d+1)}}, \Big(\frac{\mathtt{c}_1^{d} \mathtt{c}_2^{d}\mathtt{E}_{\mathscr{G}}^2 \mathtt{M}_{\mathscr{G}}^2 \mathtt{TV}^{d} \mathtt{L}^{d}}{n^2} \Big)^{\frac{1}{2(d+2)}} \Big\} (t + \log(n \mathtt{N}_{\mathscr{G}}(\delta/2) \mathtt{N}_{\mathscr{R}}(\delta/2) N_{\ast}))^{\alpha + 1}\\
&\qquad + \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} (\log n)^{\alpha} (t + \log(n \mathtt{N}_{\mathscr{G}}(\delta/2) \mathtt{N}_{\mathscr{R}}(\delta/2) N_{\ast}))^{\alpha + 1},\\
\mathsf{F}^R_n(t,\delta)
&= J(\delta) \mathtt{M}_{\mathscr{G}} + \frac{\log(n)^{\alpha/2} \mathtt{M}_{\mathscr{G}} J^2(\delta)}{\delta^2 \sqrt{n}} + \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} \sqrt{t} + (\log n)^{\alpha}\frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} t^{\alpha},
\end{align*}
and
\begin{align*}
\mathscr{V}_{\mathscr{R}} &= \{\theta(\cdot,r): r \in \mathscr{R}\},\\
\mathtt{TV} &= \max \{\mathtt{TV}_{\mathscr{G}},\mathtt{TV}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}\}, \qquad
\mathtt{L} = \max\{\mathtt{L}_{\mathscr{G}}, \mathtt{L}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}}\},\\
M_{\ast} &= \Big\lfloor\log_2\min \Big\{\Big(\frac{n \mathtt{TV}}{\mathtt{E}_{\mathscr{G}}}\Big)^{\frac{d}{d+1}}, \Big( \frac{n \mathtt{L} \mathtt{TV}}{\mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}}\Big)^{\frac{d}{d+2}} \Big\} \Big\rfloor,\\
N_{\ast} &= \Big\lceil \log_2 \max \Big\{\Big(\frac{n \mathtt{M}_{\mathscr{G}}^{d+1}}{\mathtt{E}_{\mathscr{G}} \mathtt{TV}^d}\Big)^{\frac{1}{d+1}}, \Big(\frac{n^2 \mathtt{M}_{\mathscr{G}}^{2d+2}}{\mathtt{TV}^d \mathtt{L}^d \mathtt{E}_{\mathscr{G}}^2}\Big)^{\frac{1}{d+2}}\Big\}\Big\rceil.
\end{align*}
\end{thm}
\begin{myproof}{Theorem~\ref{sa-thm: R-process -- main theorem}}
To simplify notation, we will use $\mathbbm{E} [\cdot |\mathcal{X}_{0,l}]$ in short for $\mathbbm{E} [\cdot | \mathbf{x}_i \in \mathcal{X}_{0,l}]$, and $\mathbbm{E}[\cdot|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}]$ in short for $\mathbbm{E}[\cdot|(\mathbf{x}_i, y_i) \in \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}]$ in this proof.
First, we make a reduction via the surrogate measure and normalizing transformation. Since $\operatorname{Supp}(\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}) \subseteq \operatorname{Supp}(\mathscr{G})$, we know $\mathcal{Q}_{\mathscr{G}}$ is also a surrogate measure for $\mathbbm{P}_X$ with respect to $\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}$, and $\phi_{\mathscr{G}}$ remains a valid normalizing transformation. By the same argument as in the proof for Theorem 1, assumption (ii) implies that on a possibly enriched probability space, there exists $(\mathbf{u}_i: 1 \leq i \leq n)$ i.i.d distributed with law $\mathbbm{P}_U = \mathsf{Uniform}([0,1]^d)$, and
\begin{align*}
g(\mathbf{x}_i) = g(\phi_{\mathscr{H}}^{-1}(\mathbf{u}_i)), \qquad \forall g \in \mathscr{G}, 1 \leq i \leq n,
\end{align*}
and if $g(\mathbf{x}_i) \neq 0$ for any $g \in \mathscr{G}$, then $\mathbf{x}_i = \phi_{\mathscr{H}}^{-1}(\mathbf{u}_i)$, $1 \leq i \leq n$.
Define $\widetilde{R}_n$ to be the empirical process based on $((\mathbf{u}_i,y_i): 1 \leq i \leq n)$, and
\begin{align*}
\widetilde{R}_n(f,s) = \frac{1}{\sqrt{n}}\sum_{i = 1}^n \Big[f(\mathbf{u}_i)s(y_i) - \mathbbm{E}[f(\mathbf{u}_i)s(y_i)|\mathbf{u}_i] \Big],
\end{align*}
and take $\widetilde{\mathscr{G}} = \{g \circ \phi_{\mathscr{H}}^{-1}: g \in \mathscr{G}\}$, then
\begin{align*}
R_n(g,r) = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \Big[g(\mathbf{x}_i)r(y_i) - \mathbbm{E}[g(\mathbf{x}_i)r(y_i)|\mathbf{x}_i]\Big] = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \Big[\widetilde{g}(\mathbf{u}_i) r(y_i) - \mathbbm{E}[\widetilde{g}(\mathbf{u}_i) r(y_i)|\mathbf{u}_i]\Big] = \widetilde{R}_n(\widetilde{g},r).
\end{align*}
The relation between constants for $\widetilde{\mathscr{R}}$ and constants for $\mathscr{R}$ can be deduced from Lemma~\ref{sa-lem: normalizing transformation}. Hence, without loss of generality, we assume $(\mathbf{x}_i: 1 \leq i \leq n)$ are i.i.d under common law $\mathbbm{P}_X = \mathsf{Uniform}([0,1]^d)$ distributed and $\mathcal{X} = [0,1]^d$.
Take $\mathscr{A}_{M,N}(\mathbbm{P}_Z,1)$ to be the axis-aligned cylindered quasi-dyadic expansion of $\mathbb{R}^{d+1}$. By Lemma~\ref{sa-lem: R- process -- SA error} and Lemma~\ref{sa-lem: R- process -- projection error}, for all $t > N$,
\begin{align*}
\mathbbm{P}\Big[\lVert \mathtt{\Pi}_2 R_n - \mathtt{\Pi}_2 Z_n^R \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
> C_{\mathtt{v},\alpha} \sqrt{\frac{N^{2\alpha + 1} 2^{M} \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}}{n} t}
+ C_{\mathtt{v},\alpha}\sqrt{\frac{\mathtt{C}_{\mathtt{\Pi}_2(\mathscr{G} \times \mathscr{R}),M+N}}{n}}t \Big]
& \leq 2 \mathtt{N}(\delta) e^{-t}, \\
\mathbbm{P}\Big[\lVert R_n - \mathtt{\Pi}_2 R_n \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
> C_{\mathtt{v},\alpha} \sqrt{2 N^2 \mathtt{V} + 2^{-N} \mathtt{M}_{\mathscr{G}}^2} t^{\alpha + \frac{1}{2}} + C_{\mathtt{v},\alpha} \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}}t^{\alpha + 1}\Big]
& \leq 4 \mathtt{N}(\delta) n e^{-t},\\
\mathbbm{P}\Big[\lVert Z_n^R - \mathtt{\Pi}_2 Z_n^R \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
> C_{\mathtt{v},\alpha} \sqrt{2 N^2 \mathtt{V} + 2^{-N} \mathtt{M}_{\mathscr{G}}^2} t^{\frac{1}{2}} + C_{\mathtt{v},\alpha} \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}}t \Big]
& \leq 4 \mathtt{N}(\delta) n e^{-t},
\end{align*}
where $\mathtt{V} = \sqrt{d} \min\left\{2 \mathtt{M}_{\mathscr{G}}, \mathtt{L} 2^{-\lfloor M/ d \rfloor} \right\} 2^{-\lfloor M/d \rfloor} \mathtt{TV}$, and
\begin{align*}
\mathtt{C}_{\mathtt{\Pi}_2(\mathscr{G} \times \mathscr{R})} = \sup_{f \in \mathtt{\Pi}_2(\mathscr{G} \times \mathscr{R})}\min\left\{\sup_{(j,k)} \left[\sum_{j^{\prime} < j} (j-j^{\prime})(j-j^{\prime}+1) 2^{j^{\prime}-j} \sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} \widetilde{\beta}_{j^{\prime},k^{\prime}}^2(f) \right], \lVert f \rVert_{\infty}^2(M + N)\right\}.
\end{align*}
Let $f \in \mathtt{\Pi}_2(\mathscr{G} \times \mathscr{R})$. Then there exists $g \in \mathscr{G}$ and $r \in \mathscr{R}$ such that $f = \mathtt{\Pi}_2[g,r]$. Since $f$ is already piecewise-constant, by definition of $\beta_{j,k}$'s and $\eta_{j,k}$'s, we know $\widetilde{\beta}_{l,m}(f) = \widetilde{\eta}_{l,m}(g,r)$. Fix $(j,k)$. We consider two cases.
\textbf{Case 1:} $j > N$. Then by the design of cell expansions (Section \ref{sa-sec: MULT -- cell expansions}), $\mathcal{C}_{j,k} = \mathcal{X}_{j-N,k} \times \mathcal{Y}_{*,N,0}$. By definition of $\eta_{l,m}$, for any $N \leq j^{\prime} \leq j$, we have $(j - j^{\prime})(j - j^{\prime} + 1)2^{j^{\prime}-j} \sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} \widetilde{\eta}_{j^{\prime},k^{\prime}}^2(g,r) = 0$. Now consider $0 \leq j^{\prime} < N$. Then
\begin{align*}
& \quad \sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} |\widetilde{\eta}_{j^{\prime},k^{\prime}}(g,r)| \\
&= \sum_{l: \mathcal{X}_{0,l} \subseteq \mathcal{X}_{j-N,k}} \sum_{0 \leq m < 2^{j^{\prime}}} |\mathbbm{E}[g(\mathbf{x}_i)|\mathcal{X}_{0,l}]| \cdot |\mathbbm{E}[r(y_i)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,2m}] - \mathbbm{E}[r(y_i)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,2m+1}]| \\
&\leq C_{\mathtt{v},\alpha} \sum_{l:\mathcal{X}_{0,l} \subseteq \mathcal{X}_{j-N,k}} |\mathbbm{E}[g(\mathbf{x}_i)|\mathcal{X}_{0,l}]| N^{\alpha} \leq C_{\mathtt{v},\alpha} 2^{j - N} \mathtt{M}_{\mathscr{G}} N^{\alpha}.
\end{align*}
It follows that
\begin{align*}
\sum_{j^{\prime} < j} (j-j^{\prime})(j-j^{\prime}+1) 2^{j^{\prime}-j} \sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} |\widetilde{\eta}_{j^{\prime},k^{\prime}}(g,r)| \leq \sum_{j^{\prime} < j} (j-j^{\prime})(j-j^{\prime}+1) 2^{j^{\prime} - N} C_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}}N^{\alpha} \lesssim C_{\mathtt{v},\alpha}\mathtt{M}_{\mathscr{G}} N^{\alpha}.
\end{align*}
\textbf{Case 2:} $j \leq N$. Then $\mathcal{C}_{j,k} = \mathcal{X}_{0,l} \times \mathcal{Y}_{l,j,m}$. Hence for any $0 \leq j^{\prime} \leq j$, we have
\begin{align*}
\sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} |\widetilde{\eta}_{j^{\prime},k^{\prime}}(g,r)|
&= |\mathbbm{E}[g(\mathbf{x}_i)|\mathcal{X}_{0,l}]| \sum_{m^{\prime}: \mathcal{Y}_{l,j^{\prime},m^{\prime}} \subseteq \mathcal{Y}_{l,j,m}} |\mathbbm{E}[r(y_i)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,2m}] - \mathbbm{E}[r(y_i)|\mathcal{X}_{0,l} \times \mathcal{Y}_{l,j-1,2m+1}]| \\
\lesssim & C_{\mathtt{v},\alpha} |\mathbbm{E}[g(\mathbf{x}_i)|\mathcal{X}_{0,l}]| N^{\alpha} \lesssim C_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}} N^{\alpha}.
\end{align*}
It follows that
\begin{align*}
\sum_{j^{\prime} < j} (j-j^{\prime})(j-j^{\prime}+1) 2^{j^{\prime}-j} \sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} |\widetilde{\eta}_{j^{\prime},k^{\prime}}(g,r)| \lesssim C_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}} N^{\alpha}.
\end{align*}
Moreover, for all $(j,k)$, we have $\widetilde{\beta}_{j,k}(g,r) \lesssim C_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}} N^{\alpha}$. Hence $\mathtt{C}_{\mathtt{\Pi}_2(\mathscr{G} \times \mathscr{R})} \lesssim (C_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}} N^{\alpha})^2$.
The rest of the proofs follow from choosing optimal $M, N$ and Lemma~\ref{sa-lem: M- Process -- Fluctuation off the net} in the same way as in the proof for Theorem~\ref{sa-thm: M-process -- main theorem}.
\end{myproof}
\subsection{Proof of Theorem 2}\label{sa-sec: REG -- proof of theorem 2}
The proof follows by Theorem~\ref{sa-thm: R-process -- main theorem} with $\delta = n^{-1/2}$, and
\begin{align*}
\mathtt{N}(n^{-1/2}) & = \mathtt{N}_{\mathscr{G}}(1/\sqrt{2 n},\mathtt{M}_{\mathscr{G}}) \mathtt{N}_{\mathscr{R}}(1/\sqrt{2 n},M_{\mathscr{R}})
\leq \mathtt{c}_{\mathscr{G}} \mathtt{c}_{\mathscr{R}} (2 \sqrt{n})^{\mathtt{d}_{\mathscr{G}} + \mathtt{d}_{\mathscr{R}}} = \mathtt{c}(2 \sqrt{n})^{\mathtt{d}},
\end{align*}
and
\begin{align*}
J(n^{-1/2}) & = \sqrt{2}J(\mathscr{G},\mathtt{M}_{\mathscr{G}},1/\sqrt{2 n}) + \sqrt{2}J(\mathscr{R},M_{\mathscr{R}}, 1/\sqrt{2 n}) \\
& \leq 3 n^{-1/2} \sqrt{\mathtt{d}_{\mathscr{G}} \log(\mathtt{c}_{\mathscr{G}} \sqrt{n})} + 3 \delta \sqrt{\mathtt{d}_{\mathscr{R}} \log(\mathtt{c}_{\mathscr{R}} \sqrt{n})} \\
& \leq 3 \delta \sqrt{(\mathtt{d}_{\mathscr{G}} + \mathtt{d}_{\mathscr{R}})\log(\mathtt{c}_{\mathscr{G}} \mathtt{c}_{\mathscr{R}} n)}
\leq 3 \delta \sqrt{\mathtt{d} \log(\mathtt{c} n)}.
\end{align*}
This completes the proof.\qed
\subsection{Proof of Corollary 4}\label{sa-sec: REG -- proof of corollary 4}
Take $t = C\log n$ with $C>1$ in Theorem 2.\qed
\subsection{Example: Local Polynomial Estimators}
The following lemma provides sufficient conditions for the rate of \emph{non-linearity error} and \emph{smoothing bias} claimed in Section 4.1.
\begin{lemma}\label{sa-lem: local polynomial -- non-linearity and bias}
Consider the setup of Section 4.1. Recall we assume that $((\mathbf{x}_i,y_i): 1 \leq i \leq n)$ are i.i.d random vectors taking values in $(\mathbb{R}^{d+1},\mathcal{B}(\mathbb{R}^{d+1}))$, with $\mathbf{x}_i \sim \mathbbm{P}_X$ admitting a continuous Lebesgue density $f_X$ on its support $\mathcal{X} = [0,1]^d$. Assume in addition that $\mathbf{w} \mapsto \theta(\mathbf{w};r)$ is $(\mathfrak{p} + 1)$-times continuously differentiable with $(\mathfrak{p}+1)$th partial derivatives bounded uniformly over $\mathbf{w} \in \mathcal{W} \subseteq \mathcal{X}$ and $r \in \mathscr{R}_{l}$, $l=1,2$, for some $\mathfrak{p}\geq0$.
If $(n b^d)^{-1} \log n \to 0$, then
\begin{align*}
\sup_{\mathbf{w}\in\mathcal{W},r\in\mathscr{R}_2} \big|\mathbf{e}_1^{\top} (\widehat{\mathbf{H}}_{\mathbf{w}}^{-1} - \mathbf{H}_{\mathbf{w}}^{-1}) \mathbf{S}_{\mathbf{w},r}\big|
&= O((n b^d)^{-1}\log{n}) \qquad \text{a.s.}, \quad and\\
\sup_{\mathbf{w}\in\mathcal{W},r\in\mathscr{R}_l} \big| \mathbbm{E}[\widehat{\theta}(\mathbf{w},r)|\mathbf{x}_1,\cdots,\mathbf{x}_n] - \theta(\mathbf{w},r)\big|
&= O(b^{1+\mathfrak{p}}) \qquad \text{a.s.}, \quad l=1,2.
\end{align*}
If, in addition, $\sup_{\mathbf{x} \in \mathcal{X}}\mathbbm{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] \leq 2$, then
\begin{align*}
\sup_{\mathbf{w}\in\mathcal{W},r\in\mathscr{R}_1} \big|\mathbf{e}_1^{\top} (\widehat{\mathbf{H}}_{\mathbf{w}}^{-1} - \mathbf{H}_{\mathbf{w}}^{-1}) \mathbf{S}_{\mathbf{w},r}\big|
&= O((n b^d)^{-1} \log n + (n b^d)^{-3/2}(\log n)^{5/2}) \qquad \text{a.s.}
\end{align*}
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: local polynomial -- non-linearity and bias}}
We concisely flash out the arguments that are standard from the empirical process literature.
\paragraph*{Convergence rate for each entry of $\widehat{\mathbf{H}}_{\mathbf{w}} - \mathbf{H}_{\mathbf{w}}$:} Consider $\mathbf{u}_1^{\top}(\widehat{\mathbf{H}}_{\mathbf{w}} - \mathbf{H}_{\mathbf{w}})\mathbf{u}_2$, where $\mathbf{u}_1, \mathbf{u}_2$ are multi-indices such that $|\mathbf{u}_1|, |\mathbf{u}_2| \leq p$. Take $\mathbf{v} = \mathbf{u}_1 + \mathbf{u}_2$. Define $$g_n(\xi,\mathbf{w}) = \left(\frac{\xi - \mathbf{w}}{h}\right)^{\mathbf{v}} \frac{1}{h^d} K \left(\frac{\xi - \mathbf{w}}{h}\right), \qquad \xi \in \mathcal{X}, \mathbf{w} \in \mathcal{W}.$$ Define $\mathscr{F} = \{g_n(\cdot, \mathbf{w}): \mathbf{w} \in \mathcal{W}\}$. Then $\sup_{\mathbf{w} \in \mathcal{W}}|\mathbf{u}_1^{\top}(\widehat{\mathbf{H}}_{\mathbf{w}} - \mathbf{H}_{\mathbf{w}})\mathbf{u}_2| = \sup_{f \in \mathscr{F}}|\mathbbm{E}_n[f(\mathbf{x}_i)] - \mathbbm{E}[f(\mathbf{x}_i)]|$. By standard arguments from kernel regression literature, we can show $\mathscr{F}$ forms a VC-type class over $\mathcal{X}$ with exponent $d$ and constant $\lVert \mathcal{X} \rVert_{\infty}/b$, with $\mathtt{M}_{\mathscr{F},\mathcal{X}} = O(b^{-d})$, $\sigma_n^2 = \sup_{f \in \mathscr{F}} \mathbbm{V}[f(\mathbf{x}_i)] = O(b^{-d/2})$. By Corollary 5.1 in \cite{chernozhukov2014gaussian-SA}, we can show $\mathbbm{E}[\sup_{f \in \mathscr{F}}|\mathbbm{E}_n[f(\mathbf{x}_i)] - \mathbbm{E}[f(\mathbf{x}_i)]|] = O((n b^d)^{-1/2}\sqrt{\log n} + (n b^d)^{-1} \log n)$. Since $\mathscr{F}$ is separable, we can use Talagrand's inequality \citep[Theorem 3.3.9]{Gine-Nickl_2016_Book-SA} to get for all $t > 0$,
\begin{equation*}
\mathbbm{P} \Big( \sup_{f \in \mathscr{F}}|\mathbbm{E}_n[f(\mathbf{x}_i)] - \mathbbm{E}[f(\mathbf{x}_i)]| \geq C_1 (n b^d)^{-1/2}\sqrt{t + \log n} + C_1 (n b^d)^{-1} (t + \log n)\Big) \leq \exp(-t),
\end{equation*}
where $C_1$ is a constant not depending on $n$. This shows for any multi-indices $\mathbf{u}_1, \mathbf{u}_2$ with $|\mathbf{u}_1|, |\mathbf{u}_2| \leq p$,
\begin{align*}
\sup_{\mathbf{w} \in \mathcal{W}}|\mathbf{u}_1^{\top}(\widehat{\mathbf{H}}_{\mathbf{w}} - \mathbf{H}_{\mathbf{w}}) \mathbf{u}_2| = O((n b^d)^{-1/2} \sqrt{\log n} + (n b^d)^{-1} \log n), \text{ a.s. }
\end{align*}
\paragraph*{Convergence rate for $\sup_{\mathbf{w} \in \mathcal{W}} \lVert \widehat{\mathbf{H}}_{\mathbf{w}}^{-1} - \mathbf{H}_{\mathbf{w}}^{-1} \rVert$:} Since $\mathbf{H}_{\mathbf{w}}$ and
$\widehat{\mathbf{H}}_{\mathbf{w}}$ are finite-dimensional, $\sup_{\mathbf{w} \in \mathcal{W}} \lVert \widehat{\mathbf{H}}_{\mathbf{w}} - \mathbf{H}_{\mathbf{w}} \rVert = O((n b^d)^{-1/2} \sqrt{\log n} + (n b^d)^{-1} \log n)$ a.s.. By Weyl's Theorem, $\sup_{\mathbf{w} \in \mathcal{W}}|\sigma_{d}(\widehat{\mathbf{H}}_{\mathbf{w}}) - \sigma_{d}(\mathbf{H}_{\mathbf{w}})| = O((n b^d)^{-1/2} \sqrt{\log n} + (n b^d)^{-1} \log n)$ a.s., which also implies $\inf_{\mathbf{w} \in \mathcal{W}}\sigma_{d}(\widehat{\mathbf{H}}_{\mathbf{w}}) = \Omega(1)$ a.s.. Hence
\begin{align*}
\sup_{\mathbf{w} \in \mathcal{W}} \lVert \widehat{\mathbf{H}}_{\mathbf{w}}^{-1} - \mathbf{H}_{\mathbf{w}}^{-1} \rVert \leq \sup_{\mathbf{w} \in \mathcal{W}} \lVert \widehat{\mathbf{H}}_{\mathbf{w}}^{-1} \rVert \lVert \widehat{\mathbf{H}}_{\mathbf{w}} - \mathbf{H}_{\mathbf{w}} \rVert \lVert \mathbf{H}_{\mathbf{w}}^{-1} \rVert = O((n b^d)^{-1/2} \sqrt{\log n}), \quad a.s..
\end{align*}
\paragraph*{Convergence rate for $\sup_{\mathbf{w} \in \mathcal{W}} \sup_{r \in \mathscr{R}_{\ell}} \lVert \mathbf{S}_{\mathbf{w},r} \rVert$, $\ell = 1,2$:} Consider $\mathbf{v}^{\top} \mathbf{S}_{\mathbf{w},r}$ where $|\mathbf{v}| \leq p$. Define $\mathscr{H}_{\ell} = \{(\mathbf{z},y) \mapsto g_n(\mathbf{z},\mathbf{w})(r(y) - \theta(\mathbf{z},r)): \mathbf{w} \in \mathcal{W}, r \in \mathscr{R}_{\ell}\}$, $\ell = 1,2$. It is not hard to check both $\mathscr{H}_1$ and $\mathscr{H}_2$ are VC-type classes over $\mathcal{X}$. By similar arguments as in $\widehat{\mathbf{H}}_{\mathbf{w}} - \mathbf{H}_{\mathbf{w}}$, for all $t > 0$,
\begin{align*}
\mathbbm{P} \Big(\sup_{h \in \mathscr{H}_2}|\mathbbm{E}_n[h(\mathbf{x}_i, y_i)] - \mathbbm{E}[h(\mathbf{x}_i, y_i)]| \geq C_2 (n b^d)^{-1/2}\sqrt{t + \log n} + C_2 (n b^d)^{-1} (t + \log n)\Big) \leq \exp(-t),
\end{align*}
where $C_2$ is a constant that does not depend on $n$. And if we further assume $\sup_{\mathbf{x} \in \mathcal{X}} \mathbbm{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] \leq 2$, then for all $t > 0$,
\begin{align*}
\mathbbm{P} \Big(\sup_{h \in \mathscr{H}_1}|\mathbbm{E}_n[h(\mathbf{x}_i, y_i)] - \mathbbm{E}[h(\mathbf{x}_i, y_i)]| \geq C_2 (n b^d)^{-1/2}\sqrt{t + \log n} + C_2 (n b^d)^{-1} (\log n) (t + \log n)\Big) \leq \exp(-t).
\end{align*}
Together with finite dimensionality of the vector $\mathbf{S}_{\mathbf{w},r}$,
\begin{align*}
\sup_{\mathbf{w} \in \mathcal{W}} \sup_{r \in \mathscr{R}_1} \lVert \mathbf{S}_{\mathbf{w},r} \rVert & = O((n b^d)^{-1/2} \sqrt{\log n} + (n b^d)^{-1} (\log n)^2), \quad a.s., \\
\sup_{\mathbf{w} \in \mathcal{W}} \sup_{r \in \mathscr{R}_2} \lVert \mathbf{S}_{\mathbf{w},r} \rVert & = O((n b^d)^{-1/2} \sqrt{\log n}), \quad a.s.
\end{align*}
\paragraph*{Putting together for Non-Linearity Errors: }
\begin{align*}
& \sup_{\mathbf{w} \in \mathcal{W}} \sup_{r \in \mathscr{R}_1} |\mathbf{e}_1^{\top} (\widehat{\mathbf{H}}_{\mathbf{w}}^{-1} - \mathbf{H}_{\mathbf{w}}^{-1}) \mathbf{S}_{\mathbf{w},r}| = O((n b^d)^{-1} \log n + (n b^d)^{-3/2}(\log n)^{5/2}), \quad a.s.,\\
& \sup_{\mathbf{w} \in \mathcal{W}} \sup_{r \in \mathscr{R}_2} |\mathbf{e}_1^{\top} (\widehat{\mathbf{H}}_{\mathbf{w}}^{-1} - \mathbf{H}_{\mathbf{w}}^{-1}) \mathbf{S}_{\mathbf{w},r}| = O((n b^d)^{-1} \log n), \quad a.s..
\end{align*}
\paragraph*{Smoothing Error: }
Take $\mathbf{R}_{\mathbf{w},r} = \mathbbm{E}_n \left[\mathbf{r}_p\left(\frac{\mathbf{X}_i - \mathbf{w}}{h}\right) K_h(\mathbf{X}_i - \mathbf{w})\mathfrak{r}_{\mathbf{w}}(\mathbf{X}_i;r) \right]$ where $$\mathfrak{r}_{\mathbf{w}}(\xi;r) = \theta(\xi;r) - \sum_{0 \leq |\boldsymbol{\nu}| \leq \mathfrak{p}}\frac{\partial_{\boldsymbol{\nu}}\theta(\mathbf{w};r)}{\boldsymbol{\nu}!}(\xi - \mathbf{w})^{\boldsymbol{\nu}}.$$ Since all $\theta(\cdot;r), r \in \mathscr{R}_{\ell}$ are $(\mathfrak{p} + 1)$-times continuously differentiable with derivatives bounded uniformly over $\mathcal{X}$ and $\mathscr{R}_\ell$, we have almost surely $\sup_{r \in \mathscr{R}_{\ell}}\sup_{\mathbf{w} \in \mathcal{W}}|\mathbf{R}_{\mathbf{w},r}| = O(b^{\mathfrak{p} +1})$, $\ell = 1,2$. We have proved that $\inf_{\mathbf{w} \in \mathcal{W}}\sigma_{d}(\widehat{\mathbf{H}}_{\mathbf{w}}) = \Omega(1)$ a.s.. Hence
\begin{align*}
\sup_{r \in \mathscr{R}_{\ell}}\sup_{\mathbf{w} \in \mathcal{W}} |\mathbbm{E}[\widehat{\theta}(\mathbf{w},r)|\mathbf{x}_1, \cdots, \mathbf{x}_n] - \theta(\mathbf{w},r)|
& = \sup_{r \in \mathscr{R}_{\ell}}\sup_{\mathbf{w} \in \mathcal{W}} |\mathbf{e}_1^{\top} \widehat{\mathbf{H}}_{\mathbf{w}}^{-1} \mathbf{R}_{\mathbf{w},r}| = O(b^{\mathfrak{p}+1}), \qquad a.s., \text{ for } \ell = 1,2.
\end{align*}
This completes the proof.
\end{myproof}
The following two examples provide the omitted details concerning uniform Gaussian strong approximation rates obtained via other methods, which are discussed in Section 4.1 of the paper.
\begin{example}[Strong Approximation via \cite{Rio_1994_PTRF-SA}]\label{sa-example: local polynomial estimators -- rio}
Consider the setup of Section 4.1, and assume the following regularity conditions hold:
\begin{enumerate}[label=\emph{(\alph*)}]
\item $(\mathbf{x}_i,y_i) = (\mathbf{x}_i, \varphi(\mathbf{x}_i, u_i))$, where the law of $\mathbf{b}_i = (\mathbf{x}_i, u_i)$, $\mathbbm{P}_B$, has continuous and positive Lebesgue density $f_B$ on its support $\mathcal{B} = [0,1]^{d+1}$.
\item $\mathtt{M}_{\{\varphi\},\mathcal{B}} = O(1)$, $\mathtt{K}_{\{\varphi\},\mathcal{B}} = O(1)$, and $\sup_{g \in \mathscr{G}}\mathtt{TV}_{\{\varphi\},\operatorname{Supp}(g) \times [0,1]} = O(\sup_{g \in \mathscr{G} }\mathfrak{m}(\operatorname{Supp}(g)))$.
\item $\sup_{g \in \mathscr{G}}\mathtt{TV}_{\mathscr{V}_{\mathscr{R}_l},\operatorname{Supp}(g)} = O(\sup_{g \in \mathscr{G} }\mathfrak{m}(\operatorname{Supp}(g)))$ and $\mathtt{K}_{\mathscr{V}_{\mathscr{R}_l},\mathcal{B}} = O(1)$, for $l = 1,2$.
\end{enumerate}
Recall $\mathscr{G} = \{b^{-d/2}\mathfrak{K}_\mathbf{w}(\frac{\cdot - \mathbf{w}}{b}):\mathbf{w}\in \mathcal{W}\}$ with $\mathfrak{K}_\mathbf{w}(\mathbf{u})=\mathbf{e}_1^{\top} \mathbf{H}_{\mathbf{w}}^{-1} \mathbf{p}(\mathbf{u})K(\mathbf{u})$. For $\mathscr{R}_1$, take $\mathscr{H}_1 = \{h \circ T_{\mathbbm{P}_B}^{-1}: h \in \mathscr{H}_1^o\}$, where $\mathscr{H}_1^o = \{(\mathbf{x},u) \in \mathcal{B} \mapsto g(\mathbf{x}) \varphi(\mathbf{x},u) - g(\mathbf{x}) \theta(\mathbf{x},\operatorname{Id}): g \in \mathscr{G}\}$, $T_{\mathbbm{P}_B}$ is the Rosenblatt transformation based on $\mathbbm{P}_B$ given in Section 3.1. Recall we denote $\mathcal{X} = [0,1]^d$. Then,
\begin{equation}\label{sa-eq: local polynomial estimators -- rio 1}
\begin{aligned}
\mathtt{M}_{\mathscr{H}_1,\mathcal{B}} & \leq \mathtt{M}_{\mathscr{G},\mathcal{X}}\mathtt{M}_{\{\varphi\},\mathcal{B}} = O(b^{-d/2}),\\
\mathtt{TV}_{\mathscr{H}_1,\mathcal{B}} & \leq \frac{\overline{f}_B^2}{\underline{f}_B}(\mathtt{TV}_{\mathscr{G},\mathcal{X}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \sup_{g \in \mathscr{G}} \mathfrak{m}(\operatorname{Supp}(g))) = O(b^{d/2-1}),\\
\mathtt{K}_{\mathscr{H}_1,\mathcal{B}} &\leq (2 \sqrt{d})^{d-1} \frac{\overline{f}_B^{d+1}}{\underline{f}_B^d}(\mathtt{M}_{\{\varphi\},\mathcal{B}}\mathtt{K}_{\mathscr{G},\mathcal{X}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{K}_{\{\varphi\},\mathcal{B}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{K}_{\mathscr{V}_{\mathscr{R}_1},\mathcal{X}}) = O(b^{-d/2}),\\
\mathtt{N}_{\mathscr{H}_1,\mathcal{B}}(\delta,\mathtt{M}_{\mathscr{H}_1,\mathcal{B}}) & = O(\delta^{-d-1}), \qquad 0 < \delta < 1,
\end{aligned}
\end{equation}
where $\overline{f}_B = \sup_{\mathbf{x} \in \mathcal{B}} f_B(\mathbf{x})$ and $\underline{f}_B = \inf_{\mathbf{x} \in \mathcal{B}} f_B(\mathbf{x})$. \citet[Theorem 1.1]{Rio_1994_PTRF-SA} implies that $(X_n(h): h \in \mathscr{H}_1) = (\sqrt{n b^d}\mathbf{e}_1^{\top} \mathbf{H}_{\mathbf{w}}^{-1} \mathbf{S}_{\mathbf{w},r}: \mathbf{w} \in \mathcal{W}, r \in \mathscr{R}_1)$ admits a uniform Gaussian strong approximation with rate
\begin{align*}
\mathsf{S}_n(t) = C_{d,\varphi,\mathbbm{P}_B} (n b^{d+1})^{-1/(2d+2)}\sqrt{t + d \log n} + C_{d,\varphi,\mathbbm{P}_B} (n b^d)^{-1/2}(t + d \log n),
\end{align*}
where $C_{d,\varphi,\mathbbm{P}_B}$ is a quantity that only depends on $d$, $\varphi$ and $\mathbbm{P}_B$.
For $\mathscr{R}_2$, take $\mathscr{H}_2 = \{h \circ T_{\mathbbm{P}_B}^{-1}: h \in \mathscr{H}_2^o\}$, where $\mathscr{H}_2^o = \{(\mathbf{x},u) \in \mathcal{B} \mapsto g(\mathbf{x})r \circ \varphi(\mathbf{x},u) - g(\mathbf{x})\theta(\mathbf{x},r): g \in \mathscr{G}, r \in \mathscr{R}_2\}$. Then
\begin{align*}
\mathtt{M}_{\mathscr{H}_2} &= \mathtt{M}_{\mathscr{G},\mathcal{X}}\mathtt{M}_{\{\varphi\},\mathcal{B}} = O(b^{-d/2}),\\
\mathtt{TV}_{\mathscr{H}_2} &\leq \frac{\overline{f}_B^2}{\underline{f}_B} (\mathtt{TV}_{\mathscr{G},\mathcal{X}} + \mathtt{E}_{\mathscr{G},\mathcal{X}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \sup_{g \in \mathscr{G}} \mathfrak{m}(\operatorname{Supp}(g))) \max\{\mathtt{L}_{\{\varphi\},\mathcal{B}},1\}^{d-1} = O(b^{d/2-1}), \\
\mathtt{N}_{\mathscr{H}_2,\mathcal{B}}(\delta,\mathtt{M}_{\mathscr{H}_2,\mathcal{B}}) & = O(\delta^{-d-1}), \qquad 0 < \delta < 1.
\end{align*}
\citet[Theorem 1.1]{Rio_1994_PTRF-SA} implies that $(X_n(h): h \in \mathscr{H}_2) = (\sqrt{n b^d}\mathbf{e}_1^{\top} \mathbf{H}_{\mathbf{w}}^{-1} \mathbf{S}_{\mathbf{w},r}: \mathbf{w} \in \mathcal{W}, r \in \mathscr{R}_2)$ admits a Gaussian strong approximation with rate function \begin{align*}
& \mathsf{S}_n(t) = C_{d,\varphi,\mathbbm{P}_B} (n b^{d+1})^{-1/(2d+2)}\sqrt{t + d \log n} + C_{d,\varphi,\mathbbm{P}_B} \sqrt{\frac{\log n}{n b^d}}(t + d \log n),
\end{align*}
where $C_{d,\varphi,\mathbbm{P}_B}$ is a quantity that only depends on $d$, $\varphi$ and $\mathbbm{P}_B$.
The strong approximation rates stated in Section 4.1 now follow directly from the strong approximation results above.
\end{example}
\begin{myproof}{Example~\ref{sa-example: local polynomial estimators -- rio}}
Recall $\mathscr{G} = \{b^{-d/2} \mathfrak{K}_{\mathbf{w}}(\frac{\cdot - \mathbf{w}}{b}): \mathbf{w} \in \mathcal{W}\}$ with $\mathfrak{K}_{\mathbf{w}}(\mathbf{u}) = \mathbf{e}_1^{\top} \mathbf{H}_{\mathbf{x}}^{-1} \mathbf{p}(\mathbf{u}) K(\mathbf{u})$. \newline
\textbf{(1) Properties of $\mathscr{G}$}
Since $\sup_{\mathbf{w} \in \mathcal{W}}\lVert \mathbf{H}_{\mathbf{w}}^{-1} \rVert \lesssim 1$ and $K$ is continuous with compact support, we know
\begin{align*}
\mathtt{M}_{\mathscr{G},\mathcal{X}} = O(b^{-d/2}).
\end{align*}
By a change of variables, we can show
\begin{align*}
\mathtt{E}_{\mathscr{G},\mathcal{X}} & = \sup_{\mathbf{w} \in \mathcal{W}} \mathbbm{E} \left[\left|b^{-d/2}\mathfrak{K}_{\mathbf{w}}\Big(\frac{\mathbf{x}_i - \mathbf{w}}{b}\Big)\right|\right] = O(b^{d/2}).
\end{align*}
And $\sup_{\mathbf{w} \in \mathcal{W}} \sup_{\mathbf{u}, \mathbf{u}^{\prime}}\frac{|\mathbf{r}_p\left(\frac{\mathbf{u} - \mathbf{w}}{b}\right) - \mathbf{r}_p(\frac{\mathbf{u}^{\prime} - \mathbf{w}}{b})|}{\lVert \mathbf{u} - \mathbf{u}^{\prime} \rVert_{\infty}} = O (b^{-1})$, and $\sup_{\mathbf{w} \in \mathcal{W}} \sup_{\mathbf{u}, \mathbf{u}^{\prime}} \frac{|K(\frac{\mathbf{u} - \mathbf{w}}{b}) - K(\frac{\mathbf{u}^{\prime} - \mathbf{w}}{b})|}{\lVert \mathbf{u} - \mathbf{u}^{\prime} \rVert_{\infty}} = O(b^{-1})$, hence
\begin{align*}
\mathtt{L}_{\mathscr{G},\mathcal{X}} = O(b^{-\frac{d}{2}-1}).
\end{align*}
Notice that the support of functions in $\mathscr{G}$ has uniformly bounded volume, i.e. $\sup_{g \in \mathscr{G}} \operatorname*{\mathfrak{m}}\left(\operatorname{Supp}(g) \right) = O(b^d)$. Together with the rate for $\mathtt{L}_{\mathscr{G},\mathcal{X}}$, we know
\begin{align*}
\mathtt{TV}_{\mathscr{G},\mathcal{X}} \leq \mathtt{L}_{\mathscr{G},\mathcal{X}} \sup_{g \in \mathscr{G}} \operatorname*{\mathfrak{m}}\left(\operatorname{Supp}(g) \right) = O(b^{\frac{d}{2} - 1}).
\end{align*}
Now we will show that $\mathtt{M}_{\mathscr{G},\mathcal{X}}^{-1} \mathscr{G}$ is a VC-type class. We know $\sup_{\mathbf{w}, \mathbf{w}^{\prime} \in \mathcal{W}} \lVert \mathbf{H}_{\mathbf{w}} - \mathbf{H}_{\mathbf{w}^{\prime}} \rVert/ \lVert \mathbf{w} - \mathbf{w}^{\prime} \rVert_{\infty} = O(b^{-1})$. Since $\inf_{\mathbf{w} \in \mathcal{W}} \lVert \mathbf{H}_{\mathbf{w}} \rVert = \Omega(1)$, we also have $\sup_{\mathbf{w}, \mathbf{w}^{\prime} \in \mathcal{W}} \lVert \mathbf{H}_{\mathbf{w}}^{-1} - \mathbf{H}_{\mathbf{w}^{\prime}}^{-1} \rVert/ \lVert \mathbf{w} - \mathbf{w}^{\prime} \rVert_{\infty} = O(b^{-1})$. It follows that
\begin{align*}
\mathtt{L}_{\mathscr{G},\mathcal{X}} = \sup_{\mathbf{w} \in \mathcal{W}} \sup_{\mathbf{x}, \mathbf{x}^{\prime} \in \mathcal{X}} \Big|b^{-d/2}\mathfrak{K}_{\mathbf{w}}\Big(\frac{\mathbf{x} - \mathbf{w}}{b}\Big) - b^{-d/2}\mathfrak{K}_{\mathbf{w}}\Big(\frac{\mathbf{x}^{\prime} - \mathbf{w}}{b}\Big)\Big|/ \lVert \mathbf{x} - \mathbf{x}^{\prime} \rVert_{\infty} = O(b^{-\frac{d}{2}- 1}).
\end{align*}
To upper bound $\mathtt{K}_{\mathscr{G},\mathcal{X}}$, let $\mathcal{D} \subseteq \mathcal{X}$ be a cube with edges of length $\mathtt{a}$ parallel to the coordinate axises. Consider the following two cases: (i) if $\mathtt{a} < b$, then $\mathtt{TV}_{\mathscr{G},\mathcal{D}} \leq C_K b^{-d/2-1}\mathtt{a}^d \leq C_K b^{-d/2} \mathtt{a}^{d-1}$; (ii) if $\mathtt{a} > b$, then $\mathtt{TV}_{\mathscr{G},\mathcal{D}} \leq C_K \sup_{\mathbf{w} \in \mathcal{W}} \operatorname*{\mathfrak{m}}(\operatorname{Supp}(\mathfrak{K}_{\mathbf{w}})) \mathtt{L}_{\mathscr{G},\mathcal{X}}
\leq C_K b^{d} b^{-d/2-1}
\leq C_K b^{-d/2} b^{d-1}
\leq C_K b^{-d/2} \mathtt{a}^{d-1}$.
This shows $$\mathtt{K}_{\mathscr{G},\mathcal{X}} \leq C_K b^{-d/2}.$$
Consider $h_{\mathbf{w}}(\cdot) = \sqrt{ b^d} \mathbf{e}_1^T \mathbf{H}_{\mathbf{w}}^{-1} \mathbf{r}_p(\cdot)K(\cdot)$, $\mathbf{w} \in \mathcal{W}$. Then $b^{-d/2} \mathfrak{K}_{\mathbf{w}}(\frac{\cdot - \mathbf{w}}{b}) = h_{\mathbf{w}} \left(\frac{\cdot - \mathbf{w}}{b}\right)$, $\mathbf{w} \in \mathcal{W}$. Recall that $\mathbf{x}_i$ has common law $\mathbbm{P}_X$ with Lebesgue density $f_X$. Then there exists a constant $\mathbf{c}$ only depending on $\sup_{\mathbf{x} \in \mathcal{X}}K(\mathbf{x})$, $\mathtt{L}_{\{K\},\mathcal{X}}, \sigma_K = (\int K(\mathbf{x}) d \mathbf{x})^{1/2}, \overline{f}_X = \sup_{\mathbf{x} \in \mathcal{X}} f_X(\mathbf{x}), \underline{f}_X = \inf_{\mathbf{x} \in \mathcal{X}} f_X(\mathbf{x})$ that
\begin{align*}
& \sup_{\mathbf{w} \in \mathcal{W}} \lVert h_{\mathbf{w}} \rVert_{\infty} \leq \mathbf{c}, \\
& \sup_{\mathbf{w} \in \mathcal{W}} \sup_{\mathbf{u}, \mathbf{v} \in \mathcal{W}} \frac{|h_{\mathbf{w}}(\mathbf{u}) - h_{\mathbf{w}}(\mathbf{v})|}{\lVert \mathbf{u} - \mathbf{v} \rVert_{\infty}} \leq \mathbf{c}, \\
& \sup_{\mathbf{w},\mathbf{w}^{\prime} \in \mathcal{W}} \sup_{\mathbf{u} \in \mathcal{W}} \frac{|h_{\mathbf{w}}(\mathbf{u}) - h_{\mathbf{w}^{\prime}}(\mathbf{u})|}{\lVert \mathbf{w} - \mathbf{w}^{\prime} \rVert_{\infty}} \leq \mathbf{c}.
\end{align*}
We can again apply Lemma 7 from \cite{Cattaneo-Chandak-Jansson-Ma_2024_Bernoulli-SA} to show that, for all $0 < \delta < 1$,
\begin{align*}
N_{\mathcal{X}}(\mathtt{M}_{\mathscr{G},\mathcal{X}}^{-1}\mathscr{G}, \lVert \cdot \rVert_{\mathbbm{P}_X,2}, \delta) \leq \mathbf{c} \delta^{-2d-2} + 1.
\end{align*}
\textbf{(2) Properties of $\mathscr{H}_1^o$}
Let $g \in \mathscr{G}$. Take $\mathscr{H}_1^a = \{g \cdot \varphi: g \in \mathscr{G}\}$ and $\mathscr{H}_1^b = \{g \cdot \theta(\cdot,\operatorname{Id}): g \in \mathscr{G}\}$. Then
\begin{align*}
\mathtt{M}_{\mathscr{H}_1^a,\mathcal{B}} \leq \mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{M}_{\{\varphi\},\mathcal{B}}.
\end{align*}
We have shown that all functions in $\mathscr{G}$ are Lipschitz and $\mathtt{L}_{\mathscr{G},\mathcal{X}} = O(b^{-d/2-1})$, \citet[Proposition 3.2 (b)]{Ambrosio-Fusco-Pallara_2000_book} then implies
\begin{align*}
\mathtt{TV}_{\mathscr{H}_1^a,\mathcal{B}} \leq \mathtt{M}_{\{\varphi\},\mathcal{B}}\mathtt{TV}_{\mathscr{G},\mathcal{X}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \sup_{g \in \mathscr{G}} \mathtt{TV}_{\{\varphi\},\operatorname{Supp}(g) \times [0,1]}.
\end{align*}
Let $\mathcal{C}$ be any cube of side-length $a$ in $\mathbb{R}^{d+1}$. By \citet[Proposition 3.2 (b)]{Ambrosio-Fusco-Pallara_2000_book},
\begin{align*}
\mathtt{TV}_{\mathscr{H}_1^a,\mathcal{C}}
&\leq \mathtt{M}_{\{\varphi\},\mathcal{B}} \mathtt{TV}_{\mathscr{G},\mathcal{C}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \sup_{g \in \mathscr{G}}\operatorname{TV}_{\{\varphi\}, \operatorname{Supp}(g) \times [0,1] \cap \mathcal{C}}
\leq \mathtt{M}_{\{\varphi\},\mathcal{B}} \mathtt{K}_{\mathscr{G},\mathcal{X}} a^d + \mathtt{K}_{\{\varphi\},\mathcal{B}} \mathtt{M}_{\mathscr{G},\mathcal{X}} a^d,
\end{align*}
which implies
\begin{align*}
\mathtt{K}_{\mathscr{H}_1^a,\mathcal{B}} \leq \mathtt{M}_{\{\varphi\},\mathcal{B}} \mathtt{K}_{\mathscr{G},\mathcal{X}} + \mathtt{K}_{\{\varphi\},\mathcal{B}} \mathtt{M}_{\mathscr{G},\mathcal{X}}.
\end{align*}
Similar argument shows
\begin{gather*}
\mathtt{M}_{\mathscr{H}_1^b,\mathcal{X}} \leq \mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{M}_{\{\varphi\},\mathcal{B}}, \qquad \mathtt{TV}_{\mathscr{H}_1^b,\mathcal{X}} \leq \mathtt{TV}_{\mathscr{G},\mathcal{X}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \sup_{g \in \mathscr{G}}\mathtt{TV}_{\{\theta(\cdot,\operatorname{Id})\},\operatorname{Supp}(g)}, \\
\mathtt{K}_{\mathscr{H}_1^b,\mathcal{X}} \leq \mathtt{M}_{\{\varphi\},\mathcal{B}} \mathtt{K}_{\mathscr{G},\mathcal{X}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{K}_{\{\theta(\cdot,\operatorname{Id})\},\mathcal{X}}.
\end{gather*}
Then by assumptions $\sup_{g \in \mathscr{G}}\mathtt{TV}_{\{\varphi\},\operatorname{Supp}(g) \times [0,1]} = O(\sup_{g \in \mathscr{G} }\mathfrak{m}(\operatorname{Supp}(g)))$ and $\sup_{g \in \mathscr{G}}\mathtt{TV}_{\{\theta(\cdot,\operatorname{Id})\},\operatorname{Supp}(g)} = O(\sup_{g \in \mathscr{G} }\mathfrak{m}(\operatorname{Supp}(g)))$, we have
\begin{gather*}
\mathtt{M}_{\mathscr{H}_1^o} \leq \mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{M}_{\{\varphi\},\mathcal{B}}, \qquad \mathtt{TV}_{\mathscr{H}_1^o} = O( \mathtt{TV}_{\mathscr{G},\mathcal{X}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \sup_{g \in \mathscr{G}}\operatorname*{\mathfrak{m}}(\operatorname{Supp}(g))), \\ \mathtt{K}_{\mathscr{H}_1^o} \leq \mathtt{M}_{\{\varphi\},\mathcal{B}} \mathtt{K}_{\mathscr{G},\mathcal{X}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{K}_{\{\varphi\},\mathcal{B}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{K}_{\{\theta(\cdot,\operatorname{Id})\},\mathcal{X}}.
\end{gather*}
By standard empirical process argument, $\mathscr{H}_1^o$ is a VC-type class with constant $\mathbf{c} 2^{d + 1}$ and exponent $2d + 2$ with respect to envelope function $\mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{M}_{\{\varphi\},\mathcal{B}}$ over $\mathcal{B}$.\newline
\textbf{(3) Properties of $\mathscr{H}_2^o$}
The main challenge is that $\mathscr{R}_2$ contains non-differentiable indicator. First, we study properties of $\mathscr{G} \cdot \mathscr{R}_2$. By Definition 4,
\begin{align*}
\mathtt{TV}_{\mathscr{G} \cdot \mathscr{R}_2,\mathcal{B}} & = \sup_{g \in \mathscr{G}} \sup_{y \in \mathbb{R}}\sup_{\substack{\phi \in \mathscr{D}_{d+1}([0,1]^{d+1})\\\lVert \lVert \phi \rVert_2 \rVert_{\infty} \leq 1}} \int_{[0,1]^d} \int_{[0,1]} g(\mathbf{x}) \mathbbm{1}(u \leq y)\operatorname{div}(\phi)(\mathbf{x},u) du d \mathbf{x} \\
& \leq \sup_{g \in \mathscr{G}} \sup_{y \in \mathbb{R}}\sup_{\substack{\phi \in \mathscr{D}_{d}([0,1]^{d})\\\lVert \lVert \phi \rVert_2 \rVert_{\infty} \leq 1}} \sup_{\substack{\psi \in \mathscr{D}_{1}([0,1])\\\lVert \psi \rVert_{\infty} \leq 1}}\int_{[0,1]^d} \int_{[0,1]} g(\mathbf{x}) \mathbbm{1}(u \leq y)(\operatorname{div} \phi(\mathbf{x}) + \psi^{\prime}(u)) du d \mathbf{x} \\
& = \sup_{g \in \mathscr{G}} \sup_{y \in \mathbb{R}} \sup_{\substack{\phi \in \mathscr{D}_{d}([0,1]^{d})\\\lVert \lVert \phi \rVert_2 \rVert_{\infty} \leq 1}} \int_{[0,1]^d} g(\mathbf{x}) \operatorname{div} \phi(\mathbf{x}) d \mathbf{x} + \sup_{g \in \mathscr{G}} \sup_{y \in \mathbb{R}}\sup_{\substack{\psi \in \mathscr{D}_{1}([0,1])\\\lVert \psi \rVert_{\infty} \leq 1}} \int_{[0,1]^d} g(\mathbf{x}) d \mathbf{x} (\psi(1) - \psi(0)) \\
&\leq \mathtt{TV}_{\mathscr{G},\mathcal{X}} + 2 \mathtt{E}_{\mathscr{G},\mathcal{X}},
\end{align*}
where $\mathscr{D}_{d+1}([0,1]^{d+1})$ denotes the space of infinitely differentiable functions from $[0,1]^{d+1}$ to $\mathbb{R}^{d+1}$, and $\mathscr{D}_d([0,1]^d)$ is analogously defined. Similar argument as in the proof for properties of $\mathscr{H}_1^o$ gives
\begin{align*}
\mathtt{TV}_{\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}},\mathcal{B}} \leq \mathtt{TV}_{\mathscr{G},\mathcal{X}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \sup_{g \in \mathscr{G}} \mathtt{TV}_{\mathscr{V}_{\mathscr{R}},\operatorname{Supp}(g)} = O(\mathtt{TV}_{\mathscr{G},\mathcal{X}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \sup_{g \in \mathscr{G}} \mathfrak{m}(\operatorname{Supp}(g))).
\end{align*}
It follows that
\begin{align*}
\mathtt{TV}_{\mathscr{G} \cdot \mathscr{R}_2 + \mathscr{G} \cdot \mathscr{V}_{\mathscr{R}},\mathcal{B}} = O(\mathtt{TV}_{\mathscr{G},\mathcal{X}} + \mathtt{E}_{\mathscr{G},\mathcal{X}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \sup_{g \in \mathscr{G}} \mathfrak{m}(\operatorname{Supp}(g))).
\end{align*}
Consider the change of variables function $T:[0,1]^{d+1} \to \mathbb{R}^{d+1}$ given by $T(\mathbf{x},u) = (\mathbf{x},\varphi(\mathbf{x},u))$. Since $\mathtt{L}_{\{T\},\mathcal{B}} \leq \max\{\mathtt{L}_{\{\varphi\},\mathcal{B}},1\}$, Theorem 3.16 from \cite{Ambrosio-Fusco-Pallara_2000_book} implies
\begin{align*}
\mathtt{TV}_{\mathscr{H}_2^o,\mathcal{B}} \leq \mathtt{L}_{\{T\},\mathcal{B}}^{d-1} \mathtt{TV}_{\mathscr{G} \cdot \mathscr{R}_2 + \mathscr{G} \cdot \mathscr{V}_{\mathscr{R}},\mathcal{B}} = O(\max\{\mathtt{L}_{\{\varphi\},\mathcal{B}},1\}^{d-1} \mathtt{TV}_{\mathscr{G} \cdot \mathscr{R}_2 + \mathscr{G} \cdot \mathscr{V}_{\mathscr{R}},\mathcal{B}}).
\end{align*}
By standard empirical process argument, $\mathscr{H}_2^o$ is a VC-type class with constant $C_1 2^{d+1}$ and exponent $2d+2$ with respect to envelope function $\mathtt{M}_{\mathscr{G},\mathcal{X}}$, where $C_1$ is a constant that does not depend on $n$. \newline
\textbf{(4) Effects of Rosenblatt Transformation}
By Lemma~\ref{sa-lem: normalizing transformation} with $\mathbb{Q}_{\mathscr{G}} = \mathbbm{P}_X$ and $\phi_{\mathscr{G}} = \operatorname{Id}$, we have $\mathtt{TV}_{\mathscr{H}_1} \leq \mathtt{TV}_{\mathscr{H}_1^o} \overline{f}_B^2 \underline{f}_B^{-1}$, $\mathtt{TV}_{\mathscr{H}_2} \leq \mathtt{TV}_{\mathscr{H}_2^o} \overline{f}_B^2 \underline{f}_B^{-1}$, $\mathtt{M}_{\mathscr{H}_1} = \mathtt{M}_{\mathscr{H}_1^o}$, $\mathtt{M}_{\mathscr{H}_2} = \mathtt{M}_{\mathscr{H}_2^o}$. Moreover, $\mathscr{H}_1$ and $\mathscr{H}_2$ are VC-type classes with constant $C_2 2^{d+1}$ and exponent $2d+2$ with respect to envelope functions $\mathtt{M}_{\mathscr{G},\mathcal{X}}\mathtt{M}_{\{\varphi\},\mathcal{B}}$ and $\mathtt{M}_{\mathscr{G},\mathcal{X}}$ respectively, with $C_2$ a constant that does not depend on $n$. \newline
\textbf{(5) Application of Theorem 1.1 in \cite{Rio_1994_PTRF-SA}}
We can now apply Theorem 1.1 in \cite{Rio_1994_PTRF-SA} to get $\{X_n(h): h \in \mathscr{H}_1\}$ admits a Gaussian strong approximation with rate function
\begin{align*}
& C_{d,\varphi}\sqrt{\frac{d\overline{f}_B^2}{\underline{f}_B}}\frac{\sqrt{\mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{M}_{\{\varphi\},\mathcal{B}}(\mathtt{TV}_{\mathscr{G},\mathcal{X}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \sup_{g \in \mathscr{G}} \operatorname*{\mathfrak{m}}(\operatorname{Supp}(g)))}}{n^{\frac{1}{2d+2}}}\sqrt{t + d \log n} \\
& + C_{d,\varphi}\sqrt{\frac{\mathtt{M}_{\mathscr{G},\mathcal{X}}\mathtt{M}_{\{\varphi\},\mathcal{B}}}{n}}\min\bigg\{\sqrt{\log(n) \mathtt{M}_{\mathscr{G},\mathcal{X}}\mathtt{M}_{\{\varphi\},\mathcal{B}}}, \sqrt{\frac{(2 \sqrt{d})^{d-1}\overline{f}_B^{d+1}}{\underline{f}_B^d}(\mathtt{K}_{\mathscr{G},\mathcal{X}} \mathtt{M}_{\{\varphi\},\mathcal{B}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{K}_{\{\varphi\},\mathcal{B}}))}\bigg\}(t + d\log n),
\end{align*}
where $C_{d,\varphi}$ is a quantity that only depends on $d$ and $\varphi$. And $\{X_n(h): h \in \mathscr{H}_2\}$ admits a Gaussian strong approximation with rate function
\begin{align*}
& C_{d,\varphi} \sqrt{\frac{d\overline{f}_B^2}{\underline{f}_B}}\frac{\sqrt{\mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{TV}_{\mathscr{G},\mathcal{X},\{\varphi\},\mathcal{B}}}}{n^{\frac{1}{2d+2}}}\sqrt{t + d \log n} + C_{d,\varphi}\frac{\mathtt{M}_{\mathscr{G},\mathcal{X}}\mathtt{M}_{\{\varphi\},\mathcal{B}}}{\sqrt{n}}(t +d \log n),
\end{align*}
where $\mathtt{TV}_{\mathscr{G},\mathcal{X},\{\varphi\},\mathcal{B}} = \max\{\mathtt{L}_{\{\varphi\},\mathcal{B}},1\}^{d-1} (\mathtt{TV}_{\mathscr{G},\mathcal{X}} + \mathtt{E}_{\mathscr{G},\mathcal{X}} + \mathtt{M}_{\mathscr{G},\mathcal{X}}\sup_{g \in \mathscr{G}}\operatorname*{\mathfrak{m}}(\operatorname{Supp}(g)))$.
\end{myproof}
\begin{example}[Strong Approximation via Theorem 1]\label{sa-example: local polynomial estimators -- thm1}
Consider the setup of Section 4.1, and assume the following regularity conditions hold:
\begin{enumerate}[label=\emph{(\alph*)}]
\item $(\mathbf{x}_i,y_i) = (\mathbf{x}_i, \varphi(\mathbf{x}_i, u_i))$, where the law of $\mathbf{b}_i = (\mathbf{x}_i, u_i)$, $\mathbbm{P}_B$, has a continuous and positive Lebesgue density $f_B$ on its support $\mathcal{B} = [0,1]^{d+1}$.
\item $\mathtt{M}_{\{\varphi\},\mathcal{B}} = O(1)$, $\sup_{g \in \mathscr{G}}\mathtt{TV}_{\{\varphi\},\operatorname{Supp}(g)} = O(\sup_{g \in \mathscr{G} }\mathfrak{m}(\operatorname{Supp}(g)))$, $\mathtt{K}_{\{\varphi\},\mathcal{B}} = O(1)$, and $\mathtt{L}_{\{\varphi\},\mathcal{B}} = O(1)$.
\item $\sup_{r \in \mathscr{R}_{\ell}} \sup_{\mathbf{x},\mathbf{y} \in \mathcal{X}}|\theta(\mathbf{x},r) - \theta(\mathbf{y},r)|/\lVert \mathbf{x} - \mathbf{y} \rVert_{\infty} < \infty$ for $\ell = 1,2$.
\end{enumerate}
Recall $\mathscr{G} = \{b^{-d/2}\mathfrak{K}_\mathbf{w}(\frac{\cdot - \mathbf{w}}{b}):\mathbf{w}\in \mathcal{W}\}$ with $\mathfrak{K}_\mathbf{w}(\mathbf{u})=\mathbf{e}_1^{\top} \mathbf{H}_{\mathbf{w}}^{-1} \mathbf{p}(\mathbf{u})K(\mathbf{u})$. For $\mathscr{R}_1$, take $\mathscr{H}_1 = \{h \circ T_{\mathbbm{P}_B}^{-1}: h \in \mathscr{H}_1^o\}$, where $\mathscr{H}_1^o = \{(\mathbf{x},u) \in \mathcal{B} \mapsto g(\mathbf{x}) \varphi(\mathbf{x},u) - g(\mathbf{x}) \theta(\mathbf{x},\operatorname{Id}): g \in \mathscr{G}\}$, $T_{\mathbbm{P}_B}$ is the Rosenblatt transformation based on $\mathbbm{P}_B$ given in Section 3.1. Recall we denote $\mathcal{X} = [0,1]^d$. Then, Equation~\eqref{sa-eq: local polynomial estimators -- rio 1} holds, and
\begin{align*}
\mathtt{L}_{\mathscr{H}_1} \leq \mathtt{L}_{\mathscr{H}_1^o}\frac{\overline{f}_B} {\underline{f}_B} \leq (\mathtt{L}_{\mathscr{G},\mathcal{X}} \mathtt{M}_{\{\varphi\},\mathcal{B}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{L}_{\{\varphi\},\mathcal{B}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{L}_{\mathscr{V}_{\mathscr{R}_1},\mathcal{X}})\frac{\overline{f}_B} {\underline{f}_B} = O(b^{-d/2-1}).
\end{align*}
Theorem 1 implies $(X_n(h): h \in \mathscr{H}_1) = (\sqrt{n b^d}\mathbf{e}_1^{\top} \mathbf{H}_{\mathbf{x}}^{-1} \mathbf{S}_{\mathbf{w},r}: \mathbf{x} \in [0,1]^d, r \in \mathscr{R}_1)$ admits a uniform Gaussian strong approximation with rate
\begin{align*}
\mathsf{S}_n(t) = C_{d, \varphi,\mathbbm{P}_B} (n b^{d+1})^{-1/(d+1)}\sqrt{t + d\log n} + C_{d,\varphi,\mathbbm{P}_B} (n b^d)^{-1/2}(t + d\log n),
\end{align*}
where $C_{d,\varphi,\mathbbm{P}_B}$ is a quantity that only depends on $d$, $\varphi$ and $\mathbbm{P}_B$.
The strong approximation rate stated in Section 4.1 in the paper now follows directly from the strong approximation result above.
\end{example}
\begin{myproof}{Example~\ref{sa-example: local polynomial estimators -- thm1}}
Besides the properties given in the proof of Example~\ref{sa-example: local polynomial estimators -- rio}, using product rule we can show $\mathtt{L}_{\mathscr{H}_1^o} \leq \mathtt{L}_{\mathscr{G},\mathcal{X}} \mathtt{M}_{\{\varphi\},\mathcal{B}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{L}_{\{\varphi\},\mathcal{B}} + \mathtt{M}_{\mathscr{G},\mathcal{X}} \mathtt{L}_{\mathscr{V}_{\mathscr{R}_1},\mathcal{X}} = O(b^{-d/2-1})$. By the discussion on Rosenblatt transformation in Section 3.1, $\mathtt{L}_{\mathscr{H}_1,\mathcal{X}} \leq \mathtt{L}_{\mathscr{H}_1^o,\mathcal{X}} \overline{f}_B/\underline{f}_B$. Take the surrogate measure to be $\mathcal{Q}_{\mathscr{H}_1} = \mathsf{Uniform}([0,1]^{d+1})$ with $\phi_{\mathscr{H}_1} = \operatorname{Id}$. The result then follows from application of Theorem~\ref{sa-thm: M-process -- main theorem} on
\begin{align*}
X_n(h) = \frac{1}{\sqrt{n}}\sum_{i = 1}^n [h(\mathbf{x}_i,u_i) - \mathbbm{E}[h(\mathbf{x}_i,u_i)]], \qquad h \in \mathscr{H}_1.
\end{align*}
This completes the proof.
\end{myproof}
\begin{example}[Strong Approximation via Theorem 2]\label{sa-example: local polynomial estimators -- thm3}
Consider the setup of Section 4.1 and assume the following regularity conditions hold:
\begin{enumerate}[label=\emph{(\alph*)}]
\item $\mathbf{x}_i$ has $\mathbbm{P}_X$ with Lebesgue density $f_X$ continuous on its support $\mathcal{X}$, which is a compact subset of $\mathbb{R}^d$, and $\sup_{\mathbf{x} \in \mathcal{X}}\mathbbm{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] \leq 2$.
\item $\sup_{r \in \mathscr{R}_{\ell}} \sup_{\mathbf{x},\mathbf{y} \in \mathcal{X}}|\theta(\mathbf{x},r) - \theta(\mathbf{y},r)|/\lVert \mathbf{x} - \mathbf{y} \rVert_{\infty} < \infty$ for $\ell = 1,2$.
\end{enumerate}
Recall that $\mathscr{G} = \{b^{-d/2}\mathfrak{K}_\mathbf{x}(\frac{\cdot - \mathbf{x}}{b}):\mathbf{x}\in \mathcal{X}\}$. Take the surrogate measure $\mathbb{Q}_\mathscr{G} = \mathbbm{P}_X$ and the normalizing transformation $\phi_{\mathscr{G}} = \operatorname{Id}$. Then, using the notation introduced in the paper,
\begin{align*}
\mathtt{c}_1 = d \frac{\overline{f}_X^2}{\underline{f}_X}, \qquad \mathtt{c}_2 = \frac{\overline{f}_X}{\underline{f}_X},
\end{align*}
where $\overline{f}_X = \sup_{\mathbf{x} \in \mathcal{X}}f_X(\mathbf{x})$, $\underline{f}_X = \inf_{\mathbf{x} \in \mathcal{X}}f_X(\mathbf{x})$, and
\begin{gather*}
\mathtt{M}_{\mathscr{G}} = O(b^{-d/2}), \quad \mathtt{E}_{\mathscr{G}} = O(b^{d/2}), \quad \mathtt{TV}_{\mathscr{G}} = O(b^{d/2-1}), \quad
\mathtt{L}_{\mathscr{G}} = O(b^{-d/2-1}), \\
\mathtt{N}_{\mathscr{G}}(\delta) = O(\delta^{-d-1}), \quad 0 < \delta < 1.
\end{gather*}
Theorem 2 implies that $(R_n(g,r): g \in \mathscr{G}, r \in \mathscr{R}_1) = (\sqrt{n b^d}\mathbf{e}_1^{\top} \mathbf{H}_{\mathbf{x}}^{-1} \mathbf{S}_{\mathbf{w},r}: \mathbf{x} \in [0,1]^d, r \in \mathscr{R}_1)$ admits a uniform Gaussian strong approximation with rate function
\begin{align*}
\mathsf{S}_n(t) = \bigg(\frac{\overline{f}_X^3}{\underline{f}_X^2}\bigg)^{\frac{d}{2(d+2)}} \sqrt{d}(n b^d)^{-1/(d+2)}(t + d\log n)^{3/2} + (n b^d)^{-1/2}(t + d\log n).
\end{align*}
If, in addition, $\sup_{\mathbf{x} \in [0,1]^d}\mathbbm{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] \leq 2$, then Theorem 2 implies $(R_n(g,r): g \in \mathscr{G}, r \in \mathscr{R}_1) = (\sqrt{n b^d}\mathbf{e}_1^{\top} \mathbf{H}_{\mathbf{x}}^{-1} \mathbf{S}_{\mathbf{w},r}: \mathbf{x} \in [0,1]^d, r \in \mathscr{R}_1)$ admits a uniform Gaussian strong approximation with rate function
\begin{align*}
\mathsf{S}_n(t) = \bigg(\frac{\overline{f}_X^3}{\underline{f}_X^2}\bigg)^{\frac{d}{2(d+2)}} \sqrt{d}(n b^d)^{-1/(d+2)}(t + d\log n)^{5/2} + (n b^d)^{-1/2}(t + d\log n).
\end{align*}
The strong approximation rate stated in Section 4.1 in the paper now follow directly from the strong approximation result above.
\end{example}
\begin{myproof}{Example~\ref{sa-example: local polynomial estimators -- thm3}}
The conditions of $\mathscr{G}$ can be verified from Part (1) Properties of $\mathscr{G}$ in Section~\ref{sa-example: local polynomial estimators -- rio}. It is easy to check that $\mathscr{R}_1$ satisfies the conditions in Theorem 2 with $\mathtt{c}_{\mathscr{R}_1} = 1$, $\mathtt{d}_{\mathscr{R}_1} = 1$ and $\alpha = 1$. Moreover, $\mathscr{R}_2$ satisfies the conditions in Theorem 2 with $\mathtt{c}_{\mathscr{R}_2}$ some universal constant and $\mathtt{d}_{\mathscr{R}_2} = 2$ by \citet[][Theorem 2.6.7]{wellner2013weak-SA}. The results then follow from Theorem 2.
\end{myproof}
\section{Quasi-Uniform Haar Basis}\label{sa-sec: Quasi-Uniform Haar Basis}
This section provides the proofs and additional results for Section 5. In Section~\ref{sa-sec: Quasi-Uniform Haar Basis -- GEP}, we present the proofs of Theorem 3 and Corollary 5, and verify the claims for Example 2. In Section~\ref{sa-sec: Quasi-Uniform Haar Basis -- MULT & REG}, we present the proofs of Theorem 4, Corollary 6 and the Haar Partitioning-based Regression example in Section 5.3, with the additional results for $M_n$ and $R_n$ processes under generic entropy conditions.
\subsection{General Empirical Process}\label{sa-sec: Quasi-Uniform Haar Basis -- GEP}
\subsubsection{Proof of Theorem 3}
First, we make a reduction through the surrogate measure $\mathbb{Q}_\mathscr{H}$ (Definition 2). Denote $\mathcal{E}_{\mathscr{H}} = \mathcal{X} \cap \operatorname{Supp}(\mathscr{H})$. The definition of surrogate measure implies $\mathbbm{P}_X|_{\mathcal{E}_{\mathscr{H}}} = \mathbb{Q}_{\mathscr{H}}|_{\mathcal{E}_{\mathscr{H}}}$. We use a coupling argument similar to the proof of Theorem 1. Define a probability measure $\mathbb{O}$ on $(\mathbb{R}^d \times \mathbb{R}^d, \mathcal{B}(\mathbb{R}^{2d}))$ such that for all $A \in \mathcal{B}(\mathbb{R}^{2d})$,
$\mathbb{O}|_{\mathcal{E}_{\mathscr{H}} \times \mathcal{E}_{\mathscr{H}}}(A) = \mathbbm{P}_X(\Pi_{1:d}(A \cap \{(\mathbf{x},\mathbf{x}): \mathbf{x} \in \mathcal{E}_{\mathscr{H}}\}))$, $\mathbb{O}|_{\mathcal{E}_{\mathscr{H}} \times \mathcal{E}_{\mathscr{H}}^c}(A) = \mathbb{O}|_{\mathcal{E}_{\mathscr{H}}^c \times \mathcal{E}_{\mathscr{H}}}(A) = 0$, $\mathbb{O}|_{\mathcal{E}_{\mathscr{H}}^c \times \mathcal{E}_{\mathscr{H}}^c}(A) = \int_{\mathcal{E}_{\mathscr{H}}^c} \mathbbm{P}_X(A^{\mathbf{y}} \cap \mathcal{E}_{\mathscr{H}}^c)d \mathbb{Q}(\mathbf{y})$ where $A^{\mathbf{y}} = \{\mathbf{x} \in \mathbb{R}^d: (\mathbf{x},\mathbf{y}) \in A\}$, where we take $\Pi_{1:d}(E) = \{\mathbf{x} \in \mathbb{R}^d: (\mathbf{x},\mathbf{y}) \in E \text{ for some } \mathbf{y} \in \mathbb{R}^{d}\}$ for any $E \in \mathcal{B}(\mathbb{R}^{2d})$.
The definition of $\mathbb{O}$ implies the marginals are $\mathbbm{P}_X$ and $\mathbb{Q}_{\mathscr{H}}$, respectively. By Skorohod embedding \citep[Lemma 3.35]{dudley2014uniform}, on a possibly enlarged probability space, there exists $(\mathbf{z}_i: 1 \leq i \leq n)$ i.i.d. with law $\mathbb{Q}_{\mathscr{H}}$ such that $(\mathbf{x}_i, \mathbf{z}_i)$ has joint law $\mathbb{O}$ for each $1 \leq i \leq n$. In particular, when $\mathbf{x}_i \in \mathcal{E}_{\mathscr{H}}$, $\mathbf{z}_i = \mathbf{x}_i$; and $\mathbb{O}(\{\mathbf{x}_i \in \mathcal{E}_\mathscr{H}^c\} \bigtriangleup \{\mathbf{z}_i \in \mathcal{E}_\mathscr{H}^c\}) = 0$. Moreover, since $\mathbb{Q}_\mathscr{H}(\operatorname{Supp}(\mathscr{H}) \setminus \mathcal{X}) = 0$, and the definition of $\mathbb{O}$ on $\mathcal{E}_\mathscr{H} \times \mathcal{E}_\mathscr{H}$ as a product measure between $\mathbbm{P}_X$ and $\mathbb{Q}_\mathscr{H}$, we know $\mathbb{O}(\{\mathbf{x}_i \in \mathcal{E}_\mathscr{H}^c\} \bigtriangleup \{\mathbf{z}_i \in \operatorname{Supp}(\mathscr{H})^c\}) = 0$. This allows for the reduction to $\mathbf{z}_i$-based processes, since for $1 \leq i \leq n$, almost surely
\begin{align*}
h(\mathbf{x}_i) & = h(\mathbf{x}_i) \mathbbm{1}(\mathbf{x}_i \in \mathcal{E}_\mathscr{H}) + 0 \cdot \mathbbm{1}(\mathbf{x}_i \in \mathcal{E}_\mathscr{H}^c) \\
& = h(\mathbf{z}_i) \mathbbm{1}(\mathbf{z}_i \in \mathcal{E}_\mathscr{H}) + 0 \cdot \mathbbm{1}(\mathbf{z}_i \in \operatorname{Supp}(\mathscr{H})^c) \\
& = h(\mathbf{z}_i) \mathbbm{1}(\mathbf{z}_i \in \mathcal{E}_\mathscr{H}) + h(\mathbf{z}_i) \cdot \mathbbm{1}(\mathbf{z}_i \in \operatorname{Supp}(\mathscr{H})^c) \\
& = h(\mathbf{z}_i) \mathbbm{1}(\mathbf{z}_i \in \mathcal{E}_\mathscr{H}) + h(\mathbf{z}_i) \cdot \mathbbm{1}(\mathbf{z}_i \in \mathcal{E}_\mathscr{H}^c) \\
& = h(\mathbf{z}_i), \qquad \forall h \in \mathscr{H},
\end{align*}
where the first line is due to $h = 0$ on $\operatorname{Supp}(\mathscr{H})^c$, the second line is by $\mathbb{O}(\{\mathbf{x}_i \in \mathcal{E}_\mathscr{H}^c\} \bigtriangleup \{\mathbf{z}_i \in \operatorname{Supp}(\mathscr{H})^c\}) = 0$, the third line is due to $h = 0$ on $\operatorname{Supp}(\mathscr{H})^c$, and the fourth line by $\mathbb{Q}_\mathscr{H}(\operatorname{Supp}(\mathscr{H}) \setminus \mathcal{X}) = 0$. Almost surely,
\begin{align*}
X_n(h) & = \frac{1}{\sqrt{n}}\sum_{i =1}^n \big[h(\mathbf{x}_i) - \mathbbm{E}[h(\mathbf{x}_i)]\big] = \frac{1}{\sqrt{n}}\sum_{i =1}^n \big[h(\mathbf{z}_i) - \mathbbm{E}[h(\mathbf{z}_i)] \big], \qquad \forall h \in \mathscr{H}.
\end{align*}
Hence we reduce the problem to coupling for $(\widetilde{X}_n(h): h \in \mathscr{H})$, with the process defined by
\begin{align*}
\widetilde{X}_n(h) & = \frac{1}{\sqrt{n}}\sum_{i =1}^n \big[h(\mathbf{z}_i) - \mathbbm{E}[h(\mathbf{z}_i)] \big], \qquad h \in \mathscr{H},
\end{align*}
with $(\mathbf{z}_i: 1 \leq i \leq n)$ i.i.d $\sim \mathbb{Q}_{\mathscr{H}}$. Suppose $2^{K} \leq L < 2^{K+1}$. For each $l \in \{1,2,\dots,d\}$, we can divide at most $2^K$ cells into two intervals of equal measure under $\mathbb{Q}_{\mathscr{H}}$ such that we get a new partition of $\mathcal{Q}_{\mathscr{H}} = \sqcup_{0 \leq j < 2^{K+1}} \Delta_l^{\prime}$ and satisfies
\begin{align*}
\frac{\max_{0 \leq l < 2^{K+1}}\mathbb{Q}_{\mathscr{H}}(\Delta_l^{\prime})}{\min_{0 \leq l < 2^{K+1}} \mathbb{Q}_{\mathscr{H}}(\Delta_l^{\prime})} \leq 2 \rho.
\end{align*}
By construction, there exists an axis-aligned quasi-dyadic expansion $\mathscr{A}_{K+1}(\mathbb{Q}_{\mathscr{H}}, 2 \rho) = \{\mathcal{C}_{j,k}: 0 \leq j \leq K+1, 0 \leq k < 2^{K+1 - j}\}$ such that
\begin{align*}
\left\{\mathcal{C}_{0,k}: 0 \leq k < 2^{K+1} \right\} = \left\{\Delta_{l}^{\prime}: 0 \leq l < 2^{K+1}\right\},
\end{align*}
and $\mathscr{H} \subseteq \operatorname{Span}\{\mathbbm{1}_{\Delta_j}: 0 \leq j < L\} \subseteq \operatorname{Span}\{\mathbbm{1}_{\mathcal{C}_{0,k}}: 0 \leq k < 2^{K+1}\}$. Now we consider the term $\mathtt{C}_{\mathscr{H}}$ from Lemma~\ref{sa-lem: X-process -- sa for pcw-const non-dyadic}. Let $h \in \mathscr{H}$. By definition of $S$ and the step of splitting each cell into at most two, there exists $l_1, \cdots, l_{2S} \in \{0, \cdots, 2^{K+1}-1\}$ such that $h = \sum_{q = 1}^{2S} c_q \mathbbm{1}(\Delta_{l_q}^{\prime})$ where $|c_q| \leq \mathtt{M}_{\{h\}}$. Fix $(j,k)$. Let $(l,m)$ be an index such that $\mathcal{C}_{l,m} \subseteq \mathcal{C}_{j,k}$. Since each $\Delta_{l_q}^{\prime}$ belongs to at most one $\mathcal{C}_{l-1,k}$, $\widetilde{\beta}_{l,m}(\mathbbm{1}(\Delta_{l_q}^{\prime})) = 0$ if $\Delta_{l_q}^{\prime}$ is not contained in $\mathcal{C}_{l,m}$ and $\widetilde{\beta}_{l,m}(\mathbbm{1}(\Delta_{l_q}^{\prime})) = 2^{-l+1}$ if $\Delta_{l_q}^{\prime} \subseteq \mathcal{C}_{l,m}$. Hence
\begin{align*}
\sum_{m: \mathcal{C}_{l,m} \subseteq \mathcal{C}_{j,k}} \big|\widetilde{\beta}_{l,m}(h)\big|^2 \leq 2 S \sum_{q = 1}^{2S} \sum_{m: \mathcal{C}_{l,m} \subseteq \mathcal{C}_{j,k}} \big(c_q \widetilde{\beta}_{l,k} (\mathbbm{1}(\Delta_{l_q}))\big)^2 \leq 2 S \sum_{q = 1}^{2S} c_q^2 2^{-2l} \leq 4 S^2 \mathtt{M}_{\mathscr{H}}^2 2^{-2l}.
\end{align*}
It follows that
\begin{align*}
\mathtt{C}_{\mathscr{H}} = \sup_{h \in \mathscr{H}}\min\left\{\sup_{(j,k)} \left[\sum_{l < j} (j-l)(j-l+1) 2^{l-j} \sum_{m: \mathcal{C}_{l,m} \subseteq \mathcal{C}_{j,k}} \widetilde{\beta}_{l,m}^2(h) \right], \mathtt{M}_{\mathscr{H}}^2 (K + 1) \right\} \lesssim \mathtt{M}_{\mathscr{H}}^2 \min\{K, S^2\}.
\end{align*}
Then apply Lemma~\ref{sa-lem: X-process -- sa for pcw-const non-dyadic}, we get there exists a mean-zero Gaussian process $\widetilde{Z}_n^X$ with the same covariance structure as $\widetilde{X}_n$ such that with probability at least $1 - 2 \exp(-t) - 2^{K+1} \exp(-C_{\rho} n 2^{-K-1})$,
\begin{align*}
\lVert \widetilde{X}_n - \widetilde{Z}_n^X \rVert_{\mathscr{H}} \leq \min_{\delta \in (0,1)} & \bigg\{C_{\rho} \sqrt{\frac{2^{K+2} \mathtt{M}_{\mathscr{H}} \mathtt{E}_{\mathscr{H}}}{n}(t + \log N_{\mathscr{H}}(\delta,\mathtt{M}_{\mathscr{H}}))} \\
& + C_{\rho} \sqrt{\frac{\min\{K,S^2\}}{n}} \mathtt{M}_{\mathscr{H}}(t + \log N_{\mathscr{H}}(\delta,\mathtt{M}_{\mathscr{H}})) + F_n(t, \delta)\bigg\},
\end{align*}
where $K \leq \log_2(L)$, and $C_{\rho}$ is a constant that only depends on $\rho$. The conclusion then follows from taking $(Z_n(h): h \in \mathscr{H}) = (\widetilde{Z}_n(h): h \in \mathscr{H})$ and the fact that $(X_n(h): h \in \mathscr{H}) = (\widetilde{X}_n(h): h \in \mathscr{H})$ almost surely.\qed
\subsubsection{Proof of Corollary 5}
Take $t = C\log n$ with $C>1$ and $\delta = n^{-\frac{1}{2}}$ in Theorem 3.\qed
\subsubsection{Example 2: Histogram Density Estimation}
Recall for $\mathbf{w} \in \mathcal{W}$, we define
\begin{align*}
h_{\mathbf{w}}(\mathbf{u}) = \sqrt{L} \sum_{0 \leq l < P} \mathbbm{1} \left(\mathbf{u} \in \Delta_l \right) \mathbbm{1} \left(\mathbf{w} \in \Delta_l \right), \qquad \mathbf{u} \in \mathcal{X}, \mathbf{w} \in \mathcal{W},
\end{align*}
and $\mathscr{H} = \left\{h_{\mathbf{w}}(\cdot): \mathbf{w} \in \mathcal{W} \right\}$. Then $\mathscr{H} \subseteq \operatorname{Span}(\mathbbm{1}_{\Delta_l}: 0 \leq l < P)$. In particular, for every $\mathbf{u} \in \mathcal{X}$ and $\mathbf{w} \in \mathcal{W}$, at most one of $\mathbbm{1} \left(\mathbf{u} \in \Delta_l \right) \mathbbm{1} \left(\mathbf{x} \in \Delta_l \right)$ will be non-zero. Hence $\mathtt{M}_{\mathscr{H},\mathbb{R}^d} = L^{1/2}$. Each function in $\mathscr{H}$ can be written as $c \mathbbm{1}(\Delta_l)$ for some $l \leq L$, which implies we can take $\mathtt{S}_\mathscr{H} = 1$.
If $\mathcal{W} = \mathcal{X}$, since we assume the partition is quasi-uniform of $\mathcal{Q}_\mathscr{H} = \mathcal{X}$ with $\mathbb{Q}_\mathscr{H} = \mathbbm{P}_X$, we know $\max_{0 \leq l < P}\mathbbm{P}_X \left(\Delta_l \right) \leq c_\rho L^{-1}$ for some constant $c_\rho > 0$ that only depends on $\rho$, which implies
\begin{align*}
\mathtt{E}_{\mathscr{H}} & \leq \max_{0 \leq l < P}\mathbbm{P}_X \left(\Delta_l \right) \cdot \mathtt{M}_{\mathscr{H}} \leq c_{\rho} L^{-1} \sqrt{L}
\leq c_{\rho} L^{-1/2},
\end{align*}
where in this case $P=L$.
If $\mathcal{W} \subsetneq \mathcal{X}$, take $\mathring{P}$ to be the unique number in $\mathbb{Z}$ such that $\mathring{P} \leq \frac{\mathbbm{P}_X((\cup_{0 \leq l < P}\Delta_l)^c)}{\min_{0 \leq l < P}\mathbbm{P}_X(\Delta_l)} < \mathring{P} + 1$. Then we consider the following two cases. Set $L = P +\mathring{P}$.
\textit{Case 1}: $\mathring{P} \geq 1$. The construction in Example 2 implies for every $P \leq l < L$,
\begin{align*}
1 \leq \frac{\mathbb{Q}_\mathscr{H}(\Delta_l)}{\min_{0 \leq k < P}\mathbbm{P}_X(\Delta_k)} \leq 1 + \mathring{P}^{-1} \leq 2.
\end{align*}
In particular, $\min_{0 \leq k < L}\mathbbm{P}_X(\Delta_l) = \min_{0 \leq k < P}\mathbbm{P}_X(\Delta_l)$. Combined with quasi-uniformity of $\{\Delta_l: 0 \leq l < P\}$, we have
\begin{align*}
\frac{\max_{0 \leq k < L}\mathbbm{P}_X(\Delta_k)}{\min_{0 \leq k < L}\mathbbm{P}_X(\Delta_k)} \leq \max\{\rho,2\}.
\end{align*}
Since $\mathbbm{P}_X$ agrees with $\mathbb{Q}_\mathscr{H}$ on $\sqcup_{0 \leq l < P}\Delta_l$, and $\sqcup_{P \leq l < L} \Delta_l \subseteq \mathcal{X} \cup \operatorname{Supp}(\mathscr{H})^c$, $\mathbb{Q}_\mathscr{H}$ is a surrogate measure of $\mathbbm{P}_X$ with respect to $\mathscr{H}$. And we verified that $\{\Delta_l: 0 \leq l < L\}$ is a quasi-uniform partition of $\mathcal{Q}_\mathscr{H}$ with respect to $\mathbb{Q}_\mathscr{H}$.
\textit{Case 2}: $\mathring{P} = 0$. Then for any $0 \leq l < P$, there exists $\mathring{P}_l \in \mathbb{N}$ such that
\begin{align*}
\mathring{P}_l \leq \frac{\mathbbm{P}_X(\Delta_l)}{\mathbbm{P}_X((\sqcup_{0 \leq l < P}\Delta_l)^c)} < \mathring{P}_l + 1.
\end{align*}
Taking arbitrary $\Delta_P$ with $\mathbbm{P}_X(\Delta_P) = \mathbbm{P}_X((\sqcup_{0 \leq l < P}\Delta_l)^c)$, and for $0 \leq l < P$ break $\Delta_l$ into $\mathring{P}_l$ pieces of equal measure by $\mathbbm{P}_X$, we can show by similar arguments as above that the refined cells with the additional $\Delta_P$ together forms a quasi-uniform partition of $\mathcal{X}$ with respect to $\mathbbm{P}_X$. Suppose also in this case, the number of cells in the quasi-uniform partition is $L$ after refinement.
In both cases, we know $\max_{0 \leq l < L}\mathbbm{P}_X \left(\Delta_l \right) \leq c_{\rho} L^{-1}$ for some constant $c_{\rho}$ that only depends on $\rho$, which implies
\begin{align*}
\mathtt{E}_{\mathscr{H}} & \leq \max_{0 \leq l < P}\mathbbm{P}_X \left(\Delta_l \right) \cdot \mathtt{M}_{\mathscr{H}} \leq c_{\rho} L^{-1} \sqrt{L}
\leq c_{\rho} L^{-1/2}.
\end{align*}
We can then apply Theorem 3 to get the stated rates.
\qed
\subsection{Residual-Based (and Multiplicative Separable) Empirical Process}\label{sa-sec: Quasi-Uniform Haar Basis -- MULT & REG}
For $\delta \in (0,1]$, define
\begin{align*}
\mathtt{N}(\delta) = \mathtt{N}_{\mathscr{G}}(\delta/\sqrt{2}, \mathtt{M}_{\mathscr{G}}) \mathtt{N}_{\mathscr{R}}(\delta/\sqrt{2},M_{\mathscr{R}})
\end{align*}
and
\begin{align*}
J(\delta) = \sqrt{2}J(\mathscr{G},\mathtt{M}_{\mathscr{G}},\delta/\sqrt{2}) + \sqrt{2}J(\mathscr{R},M_{\mathscr{R}}, \delta/\sqrt{2}).
\end{align*}
To simplify notation, the parameters of \(\mathscr{G}\) and \(\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}\) (Definitions 4 to 12, \ref{sa-defn: smooth tv}, \ref{sa-defn: smooth ktv}) are taken with \(\mathcal{C} = \mathcal{Q}_{\mathscr{G}}\), and the index \(\mathcal{Q}_{\mathscr{G}}\) is omitted where there is no ambiguity; the parameters of \(\mathscr{R}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{Y}\), and the index \(\mathcal{Y}\) is omitted where there is no ambiguity; and the parameters of \(\mathscr{G} \times \mathscr{R}\) (Definitions 4 to 12, \ref{sa-defn: uniform covering number for product space}, \ref{sa-defn: uniform entropy integral for product space}) are taken with \(\mathcal{C} = \mathcal{Q}_{\mathscr{G}} \times \mathcal{Y}\), and the index \(\mathcal{Q}_{\mathscr{G}} \times \mathcal{Y}\) is omitted where there is no ambiguity.
\begin{thm}\label{sa-thm: M-Process -- Approximate Dyadic Thm}
Suppose $(\mathbf{z}_i=(\mathbf{x}_i, y_i): 1 \leq i \leq n)$ are i.i.d. random vectors taking values in $(\mathbb{R}^{d+1}, \mathcal{B}(\mathbb{R}^{d+1}))$, where $\mathbf{x}_i$ has distribution $\mathbbm{P}_X$ supported on $\mathcal{X}\subseteq\mathbb{R}^d$, $y_i$ has distribution $\mathbbm{P}_Y$ supported on $\mathcal{Y}\subseteq\mathbb{R}$, and the following conditions hold.
\begin{enumerate}[label=(\roman*)]
\item $\mathscr{G} \subseteq \operatorname{Span}\{\mathbbm{1}_{\Delta_l}: 0 \leq l < L\}$ is a class of Haar functions on $(\mathbb{R}^d, \mathcal{B}(\mathbb{R}^d),\mathbbm{P}_X)$.
\item There exists a surrogate measure $\mathbb{Q}_\mathscr{G}$ for $\mathbbm{P}_X$ with respect to $\mathscr{G}$ such that $\{\Delta_l: 0 \leq l < L\}$ forms a \textit{quasi-uniform partition} of $\mathcal{Q}_\mathscr{G}$ with respect to $\mathbb{Q}_\mathscr{G}$:
\begin{align*}
\mathcal{Q}_\mathscr{G} \subseteq \sqcup_{0\leq l < L} \Delta_l \qquad\text{and}\qquad
\frac{\max_{0 \leq l < L}\mathbb{Q}_\mathscr{G}(\Delta_l)}{\min_{0 \leq l < L}\mathbb{Q}_\mathscr{G}(\Delta_l)} \leq \rho < \infty.
\end{align*}
\item $\mathscr{G}$ is a VC-type class with envelope function $\mathtt{M}_{\mathscr{G}}$ over $\mathcal{Q}_\mathscr{G}$ with $\mathtt{c}_{\mathscr{G}} \geq e$ and $\mathtt{d}_{\mathscr{G}} \geq 1$.
\item $\mathscr{R}$ is a real-valued pointwise measurable class of functions on $(\mathbb{R}, \mathcal{B}(\mathbb{R}),\mathbbm{P}_Y)$.
\item $\mathscr{R}$ is a VC-type class with envelope $M_{\mathscr{R},\mathcal{Y}}$ over $\mathcal{Y}$ with $\mathtt{c}_{\mathscr{R},\mathcal{Y}}\geq e$ and $\mathtt{d}_{\mathscr{R},\mathcal{Y}}\geq 1$, where $M_{\mathscr{R},\mathcal{Y}}(y) + \mathtt{pTV}_{\mathscr{R},(-|y|,|y|)} \leq \mathtt{v} (1 + |y|^{\alpha})$ for all $y \in \mathcal{Y}$, for some $\mathtt{v}>0$, and for some $\alpha\geq0$. Furthermore, if $\alpha>0$, then $\sup_{\mathbf{x} \in \mathcal{X}}\mathbbm{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] \leq 2$.
\end{enumerate}
Then, on a possibly enlarged probability space, there exists mean-zero Gaussian processes $(Z_n^G(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ with almost sure continuous trajectory such that:
\begin{itemize}
\item $\mathbbm{E}[G_n(g_1, r_1) G_n(g_2, r_2)] = \mathbbm{E}[Z^G_n(g_1, r_1) Z^G_n(g_2, r_2)]$ for all $(g_1, r_1), (g_2, r_2) \in \mathscr{G} \times \mathscr{R}$, and
\item $\mathbbm{P}[\lVert G_n - Z_n^G \rVert_{\mathscr{G} \times \mathscr{R}} > C_1 C_{\mathtt{v},\alpha} C_{\rho} \min_{\delta \in (0,1)}(\mathsf{H}^G_n(t,\delta) + \mathsf{F}^G_n(t,\delta))] \leq C_2 e^{-t} + L e^{-C_{\rho}n/L}$ for all $t > 0$,
\end{itemize}
where $C_1$ and $C_2$ are universal constants, $C_{\mathtt{v},\alpha} = \mathtt{v} \max\{1 + (2 \alpha)^{\frac{\alpha}{2}}, 1 + (4 \alpha)^{\alpha}\}$, $C_{\rho}$ is a constant that only depends on $\rho$, and
\begin{equation*}
\begin{aligned}
\mathsf{H}^G_n(t,\delta) &= \sqrt{\frac{L \mathtt{M}_{\mathscr{G}} \mathtt{E}_{\mathscr{G}}}{n}}\left(t + \log \mathtt{N}_{\mathscr{G}}(\delta/2) + \log \mathtt{N}_{\mathscr{R}}(\delta/2) + \log_2 N^{\ast}\right)^{\alpha + \frac{1}{2}} \\
&\qquad + \sqrt{\frac{\min\{L+N^{\ast},\mathtt{S}_{\mathscr{G}}^2\}}{n}} \mathtt{M}_{\mathscr{G}} (\log n)^{\alpha}\left(t + \log \mathtt{N}_{\mathscr{G}}(\delta/2) + \log \mathtt{N}_{\mathscr{R}}(\delta/2) + \log_2 N^{\ast}\right)^{\alpha + 1},
\end{aligned}
\end{equation*}
and recall
\begin{align*}
\mathsf{F}^G_n(t,\delta)
&= J(\delta) \mathtt{M}_{\mathscr{G}} + \frac{(\log n)^{\alpha/2} \mathtt{M}_{\mathscr{G}} J^2(\delta)}{\delta^2 \sqrt{n}} + \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} \sqrt{t} + (\log n)^{\alpha}\frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} t^{\alpha},
\end{align*}
with $N^{\ast}=\Big\lceil \log_2 \Big(\frac{n \mathtt{M}_{\mathscr{G}}}{2^L \mathtt{E}_{\mathscr{G}}}\Big) \Big\rceil$, $\mathtt{S}_{\mathscr{G}} = \sup_{g\in\mathscr{G}} \sum_{l=1}^L \mathbbm{1}(\operatorname{Supp}(g)\cap\Delta_l \neq \emptyset)$.
\end{thm}
\begin{myproof}{Theorem~\ref{sa-thm: M-Process -- Approximate Dyadic Thm}}
First, we make a reduction through the surrogate measure and normalizing transformation.
Let $\mathcal{Z}_{\mathscr{G}} = \mathcal{X} \cap \operatorname{Supp}(\mathscr{G})$. Definition 2 implies $\mathbbm{P}_X|_{\mathcal{Z}_\mathscr{G}} = \mathbb{Q}_{\mathscr{G}}|_{\mathcal{Z}_\mathscr{G}}$. Define a joint probability measure $\mathbb{O}$ on $(\mathbb{R}^d \times \mathbb{R}^d, \mathcal{B}(\mathbb{R}^{2d}))$ such that for all $A \in \mathcal{B}(\mathbb{R}^{2d})$
\begin{align*}
\mathbb{O}(A \cap (\mathcal{Z}_{\mathscr{G}} \times \mathcal{Z}_{\mathscr{G}})) & = \mathbbm{P}_X(\Pi_{1:d}(A \cap \{(\mathbf{x},\mathbf{x}): \mathbf{x} \in \mathcal{Z}_{\mathscr{G}}\})), \\
\mathbb{O}(A \cap (\mathcal{Z}_{\mathscr{G}} \times \mathcal{Z}_{\mathscr{G}}^c)) & = \mathbb{O}(A \cap (\mathcal{Z}_{\mathscr{G}}^c \times \mathcal{Z}_{\mathscr{G}})) = 0, \\
\mathbb{O}(A \cap (\mathcal{Z}_{\mathscr{G}}^c \times \mathcal{Z}_{\mathscr{G}}^c)) & = \int_{\mathcal{Z}_{\mathscr{G}}^c} \mathbbm{P}_X(A^{\mathbf{y}} \cap \mathcal{Z}_{\mathscr{G}}^c)d \mathbb{Q}_{\mathscr{G}}(\mathbf{y}),
\end{align*}
where for $A \in \mathcal{B}(\mathbb{R}^{2d})$, $\Pi_{1:d}(A) = \{\mathbf{x} \in \mathbb{R}^d: (\mathbf{x},\mathbf{y}) \in A \text{ for some } \mathbf{y} \in \mathbb{R}^{d}\}$, $A^{\mathbf{y}} = \{\mathbf{x} \in \mathbb{R}^d: (\mathbf{x},\mathbf{y}) \in A\}$.
Then we can check that (i) the marginals of $\mathbb{O}$ are $\mathbbm{P}_X$ and $\mathbb{Q}_{\mathscr{G}}$, respectively; (ii) $\mathbb{O}|_{\mathcal{Z}_{\mathscr{G}} \times \mathbb{R}^d \cup \mathbb{R}^d \times \mathcal{Z}_{\mathscr{G}}}$ is supported on $\{(\mathbf{x},\mathbf{x}): \mathbf{x} \in \mathcal{Z}_{\mathscr{G}}\}$. By Skorohod embedding \citep[Lemma 3.35]{dudley2014uniform}, on a possibly enlarged probability space, there exists a $\mathbf{u}_i, 1 \leq i \leq n$ i.i.d. $\sim \mathbb{Q}_{\mathscr{G}}$ such that $(\mathbf{x}_i,\mathbf{u}_i)$ has joint law $\mathbb{O}$. In particular, if $\mathbf{x}_i \in \mathcal{Z}_{\mathscr{G}}$, then $\mathbf{x}_i = \mathbf{u}_i$; if $\mathbf{x}_i \in \mathcal{Z}_{\mathscr{G}}^c$, then $\mathbf{u}_i \in \mathcal{Z}_{\mathscr{G}}^c$, and since $\mathcal{Q}_\mathscr{G} \subseteq \mathcal{X} \cup (\cap_{g \in \mathscr{G}} \operatorname{Supp}(g)^c)$, $\mathbf{u}_i \in \cap_{g \in \mathscr{G}} \operatorname{Supp}(g)^c$. Thus for any $g \in \mathscr{G}$, $r \in \mathscr{R}$,
\begin{align*}
G_n(g,r) & = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \big[g(\mathbf{x}_i)r(y_i) - \mathbbm{E}[g(\mathbf{x}_i)r(y_i)]\big]
= \frac{1}{\sqrt{n}} \sum_{i = 1}^n \big[g(\mathbf{u}_i) r(y_i) - \mathbbm{E}[g(\mathbf{u}_i) r(y_i)]\big],
\end{align*}
where the second equality follows because $\mathbf{x}_i=\mathbf{u}_i$ on the event $\{\mathbf{x}_i \in \mathcal{Z}_{\mathscr{G}}\}$, and $g(\mathbf{x}_i) = g(\mathbf{u}_i) = 0$ (a.s.) on the event $\{\mathbf{x}_i \in \mathcal{Z}_{\mathscr{G}}^c\}$. Hence, we work with an equivalent empirical process
\begin{align*}
\widetilde{G}_n(g,r) = \frac{1}{\sqrt{n}} \sum_{i = 1}^n \big[g(\mathbf{u}_i)r(y_i) - \mathbbm{E}[g(\mathbf{u}_i)r(y_i)] \big], \qquad g \in \mathscr{G}, r \in \mathscr{R}.
\end{align*}
In particular $(\widetilde{G}_n(g,r): g \in \mathscr{G}, r \in \mathscr{R}) = (G_n(g,r): g \in \mathscr{G}, r \in \mathscr{R})$. Hence w.l.o.g. assume $\mathbb{Q}_{\mathscr{G}} = \mathbbm{P}_X$ and we work with the $(G_n(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ process.
Suppose $2^{M} \leq L < 2^{M+1}$. For each $l \in \{1,2,\dots,d\}$, we can divide at most $2^M$ cells into two intervals of equal measure under $\mathbbm{P}_X$ such that we get a new partition of $\mathcal{X} = \sqcup_{0 \leq j < 2^{M+1}} \Delta_l^{\prime}$ and satisfies
\begin{align*}
\frac{\max_{0 \leq l < 2^{M+1}}\mathbbm{P}_X(\Delta_l^{\prime})}{\min_{0 \leq l < 2^{M+1}} \mathbbm{P}_X(\Delta_l^{\prime})} \leq 2 \rho.
\end{align*}
By construction, for each $N \in \mathbb{N}$, there exists an axis-aligned quasi-dyadic expansion $\mathscr{A}_{M+1,N}(\mathbbm{P}_Z, 2 \rho) = \{\mathcal{C}_{j,k}: 0 \leq j \leq M+1 + N, 0 \leq k < 2^{M+1+N - j}\}$ such that
\begin{align*}
\left\{\mathcal{X}_{0,k}: 0 \leq k < 2^{M+1} \right\} = \left\{\Delta_{l}^{\prime}: 0 \leq l < 2^{M+1}\right\},
\end{align*}
and $\mathscr{G} \subseteq \operatorname{Span}\{\mathbbm{1}_{\Delta_j}: 0 \leq j < J\} \subseteq \operatorname{Span}\{\mathbbm{1}_{\mathcal{X}_{0,k}}: 0 \leq k < 2^{M+1}\}$. Hence
\begin{align}\label{eq: M- and R- processes -- projection characterization}
\mathtt{\Pi}_{0}(g,r) = \mathtt{\Pi}_{1}(g,r) = \sum_{0 \leq l < 2^{K+1}} \sum_{0 \leq m < 2^N} \mathbbm{1}(\mathcal{X}_{0,l} \times \mathcal{Y}_{j,l,m}) g|_{\mathcal{X}_{0,l}} \mathbbm{E}[r(y_i)|\mathbf{x}_i \in \mathcal{X}_{0,l}, y_i \in \mathcal{Y}_{j,l,m}].
\end{align}
Again, consider $(\mathscr{G} \times \mathscr{R})_{\delta}$ which is a $\delta \lVert \mathtt{M}_{\mathscr{G}} M_{\mathscr{R}} \rVert_{\mathbbm{P}_Z}$ of $\mathscr{G} \times \mathscr{R}$ of cardinality no greater than $\mathtt{N}_{\mathscr{G} \times \mathscr{R}}(\delta,\mathtt{M}_{\mathscr{G}} M_{\mathscr{R}})$, $0 < \delta \leq 1$. The SA error for projected process on the $\delta$-net is given by Lemma~\ref{sa-lem: M- process -- SA error quasi-dyadic}: For all $t > 0$,
\begin{align*}
& \mathbbm{P}\Big[\lVert \mathtt{\Pi}_{1} G_n - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}
> C_{\mathtt{v},\alpha} \sqrt{\frac{N^{2\alpha + 1} 2^{M+1} \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}}{n} t}
+ C_{\mathtt{v},\alpha}\sqrt{\frac{\mathtt{C}_{\mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}), M+N}}{n}} t \Big] \\
\leq \quad & 2 \mathtt{N}_{\mathscr{G} \times \mathscr{R}}(\delta, \mathtt{M}_{\mathscr{G}} M_{\mathscr{R}}) e^{-t} + 2^{M}\exp \left(-C_{\rho} n 2^{-M}\right).
\end{align*}
Now we find an upper bound for $\mathtt{C}_{\mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}),M+N}$. Consider the following two cases.
\paragraph*{Case 1: $j \geq N$} Let $g \in \mathscr{G}, r \in \mathscr{R}$. Fix $(j,k)$. Let $(j^{\prime},m^{\prime})$ be an index such that $\mathcal{C}_{j^{\prime},m^{\prime}} \subseteq \mathcal{C}_{j,k}$. If $N \leq j^{\prime} \leq M + N$, then by definition of $S$ and the step of splitting each cell into at most two, there exists $l_1, \cdots, l_{2S} \in \{0, \cdots, 2^{M+1}-1\}$ with possible duplication such that $g = \sum_{q = 1}^{2S} c_q \mathbbm{1}(\Delta_{l_q}^{\prime})$ where $|c_q| \leq \mathtt{M}_{\{g\}}$. Since each $\Delta_{l_q}^{\prime}$ belongs to at most one $\mathcal{X}_{j^{\prime}-N,k}$, $\widetilde{\gamma}_{j^{\prime},m^{\prime}}(\mathbbm{1}(\Delta_{l_q}^{\prime}),r) = 0$ if $\Delta_{l_q}^{\prime}$ is not contained in $\mathcal{X}_{j^{\prime}-N,m^{\prime}}$ and $|\widetilde{\gamma}_{j^{\prime},m^{\prime}}(\mathbbm{1}(\Delta_{l_q}^{\prime}),r)| \leq C_{\mathtt{v},\alpha} 2^{-l+1}$ if $\Delta_{l_q}^{\prime} \subseteq \mathcal{X}_{j^{\prime}-N,m^{\prime}}$ where $C_{\mathtt{v},\alpha} = \mathtt{v}(1 + (2 \sqrt{\alpha})^{\alpha})$. For $j^{\prime}$ such that $N \leq j^{\prime} \leq j$,
\begin{align*}
\sum_{m^{\prime}: \mathcal{C}_{j^{\prime},m^{\prime}} \subseteq \mathcal{C}_{j,k}} \big|\widetilde{\gamma}_{j^{\prime},m^{\prime}}(g, r)\big|^2 \leq 2 S \sum_{q = 1}^{2S} \sum_{m^{\prime}: \mathcal{C}_{j^{\prime},m^{\prime}} \subseteq \mathcal{C}_{j,k}} \big(c_q \widetilde{\gamma}_{j^{\prime},m^{\prime}} (\mathbbm{1}(\Delta_{l_q}),r)\big)^2 \leq 2 C_{\mathtt{v},\alpha}^2 S \sum_{q = 1}^{2S} c_q^2 2^{-2l} \leq 4 C_{\mathtt{v},\alpha}^2 S^2 \mathtt{M}_{\mathscr{G}}^2 2^{-2l}.
\end{align*}
For $0 \leq j^{\prime} \leq j$,
\begin{align*}
& \sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} |\widetilde{\gamma}_{j^{\prime},k^{\prime}}(g,r)| \\
&= \sum_{l: \mathcal{X}_{0,l} \subseteq \mathcal{X}_{j-N,k}} \sum_{0 \leq m < 2^{j^{\prime}}} |\mathbbm{E}[g(\mathbf{x}_i)|\mathbf{x}_i \in \mathcal{X}_{0,l}]| \cdot |\mathbbm{E}[r(y_i)|\mathbf{x}_i \in \mathcal{X}_{0,l}, y_i \in \mathcal{Y}_{l,j-1,2m}] \\
& \phantom{aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa} - \mathbbm{E}[r(y_i)|\mathbf{x}_i \in \mathcal{X}_{0,l}, y_i \in \mathcal{Y}_{l,j-1,2m+1}]| \\
&\leq C_{\mathtt{v},\alpha} \sum_{l:\mathcal{X}_{0,l} \subseteq \mathcal{X}_{j-N,k}} |\mathbbm{E}[g(\mathbf{x}_i)|\mathbf{x}_i \in \mathcal{X}_{0,l}]| N^{\alpha} \leq C_{\mathtt{v},\alpha} 2^{j - N} \mathtt{M}_{\mathscr{G}} N^{\alpha}.
\end{align*}
Since $|\widetilde{\gamma}_{l,m}(g,r)| \lesssim C_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}} N^{\alpha}$ for all $(l,m)$, $\sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} \widetilde{\gamma}_{j^{\prime},k^{\prime}}^2(g,r) \leq C_{\mathtt{v},\alpha}^2 2^{j-N} \mathtt{M}_{\mathscr{G}}^2 N^{2 \alpha}$. Putting together
\begin{align*}
\sum_{j^{\prime} < j} (j - j^{\prime})(j - j^{\prime} + 1)2^{j^{\prime} - j} \sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} \widetilde{\gamma}_{j^{\prime},k^{\prime}}^2(g,r)
& \lesssim C_{\mathtt{v},\alpha}^2 S^2 \mathtt{M}_{\mathscr{G}}^2 + C_{\mathtt{v},\alpha}^2 \mathtt{M}_{\mathscr{G}}^2 N^{2 \alpha}.
\end{align*}
\paragraph*{Case 2: $l < N$} Hence for any $0 \leq j^{\prime} \leq j$, we have
\begin{align*}
\sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} |\widetilde{\gamma}_{j^{\prime},k^{\prime}}(g,r)|
&= |\mathbbm{E}[g(\mathbf{x}_i)|\mathbf{x}_i \in \mathcal{X}_{0,l}]| \sum_{m^{\prime}: \mathcal{Y}_{l,j^{\prime},m^{\prime}} \subseteq \mathcal{Y}_{l,j,m}} |\mathbbm{E}[r(y_i)|\mathbf{x}_i \in \mathcal{X}_{0,l}, y_i \in \mathcal{Y}_{l,j-1,2m}] \\
& - \mathbbm{E}[r(y_i)|\mathbf{x}_i \in \mathcal{X}_{0,l}, y_i \in \mathcal{Y}_{l,j-1,2m+1}]| \\
&\leq C_{\mathtt{v},\alpha} |\mathbbm{E}[g(\mathbf{x}_i)|\mathbf{x}_i \in \mathcal{X}_{0,l}]| N^{\alpha} \leq C_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}} N^{\alpha}.
\end{align*}
It follows that
\begin{align*}
\sum_{j^{\prime} < j} (j-j^{\prime})(j-j^{\prime}+1) 2^{j^{\prime}-j} \sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} |\widetilde{\gamma}_{j^{\prime},k^{\prime}}(g,r)| \leq C_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}} N^{\alpha}.
\end{align*}
It follows that
\begin{align*}
\mathtt{C}_{\mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}), M+N} & = \sup_{h \in \mathscr{H}}\min\left\{\sup_{(j,k)} \left[\sum_{l < j} (j-l)(j-l+1) 2^{l-j} \sum_{m: \mathcal{C}_{l,m} \subseteq \mathcal{C}_{j,k}} \widetilde{\gamma}_{l,m}^2(h) \right], \mathtt{M}_{\mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R})}^2 (M + N) \right\}\\
& \leq C_{\mathtt{v},\alpha}^2 \mathtt{M}_{\mathscr{G}}^2 N^{2\alpha} \min\{M+N, S^2+1\}.
\end{align*}
By the characterization of projections in Equation~\ref{eq: M- and R- processes -- projection characterization}, we know the mis-specification error is zero, that is, $\mathtt{\Pi}_{1} G_n(g,r) = \mathtt{\Pi}_{0} G_n(g,r)$ and $\mathtt{\Pi}_{1} Z_n^G(g,r) = \mathtt{\Pi}_{0} G_n(g,r)$. Since $g$ is already piecewise-constant on $\mathcal{X}_{0,l}$'s, the $L_2$-projection error is solely contributed from $r$. Consider $\mathcal{B} = \sigma \left(\left\{\mathbbm{1}_{\mathcal{C}_{0,k}}: 0 \leq k < 2^{M + N + 1}\right\} \right)$. Denote $r_{\tau} = r|_{[-\tau^{1/\alpha}, \tau^{1/\alpha}]}$. Then
\begin{align*}
\left|\mathbbm{E} \left[g(\mathbf{x}_i) r_{\tau}(y_i) \middle| \mathcal{B} \right] - g(\mathbf{x}_i)r_{\tau}(y_i)\right| \leq \mathtt{M}_{\mathscr{G}} \left|r_{\tau}(y_i) - \mathbbm{E}[r_{\tau}(y_i)|\mathcal{B}] \right|.
\end{align*}
Then by the same argument as in the proof for Lemma~\ref{sa-lem: M-Process -- L2 Error Moment} and the argument for truncation error in the proof for Lemma~\ref{sa-lem: M- process -- projection error}, for all $t > N$,
\begin{align}\label{sa-eq: M-process -- proj error}
\mathbbm{P} \left(\lVert G_n - \mathtt{\Pi}_{1} G_n \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}} + \lVert Z_n^G - \mathtt{\Pi}_{1} Z_n^G \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}}\geq N \sqrt{2^{-N}\mathtt{M}_{\mathscr{G}}^2}t^{\alpha + \frac{1}{2}}+ \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}}t^{\alpha + 1}\right) \leq 4 \mathtt{N}_{\mathscr{G} \times \mathscr{R}}(\delta, \mathtt{M}_{\mathscr{G}} M_{\mathscr{R}}) n e^{-t}.
\end{align}
Then apply Lemma~\ref{sa-lem: M- process -- SA error quasi-dyadic}, we get there exists a mean-zero Gaussian process $Z_n^G$ with the same covariance structure as $G_n$ such that with probability at least $1 - 2 \exp(-t) - 2^{M+1} \exp(-C_{\rho} n 2^{-M-1})$,
\begin{align*}
\lVert \mathtt{\Pi}_{1} G_n - \mathtt{\Pi}_{1} Z_n^G \rVert_{\mathscr{G} \times \mathscr{R}} & \leq C_{\rho} \min_{\delta \in (0,1)} \bigg\{\sqrt{\frac{2^{M+2} \mathtt{M}_{\mathscr{G}} \mathtt{E}_{\mathscr{G}}}{n}}(t + \log \mathtt{N}_{\mathscr{G} \times \mathscr{R}}(\delta, \mathtt{M}_{\mathscr{G}} M_{\mathscr{R}}))^{\alpha + \frac{1}{2}} \\
& \qquad + \sqrt{\frac{\mathtt{C}_{\mathtt{\Pi}_{1}(\mathscr{G} \times \mathscr{R}), M+N}}{n}} (t + \log \mathtt{N}_{\mathscr{G} \times \mathscr{R}}(\delta, \mathtt{M}_{\mathscr{G}} M_{\mathscr{R}}))^{\alpha + 1}
+ F_n(t, \delta)\bigg\},
\end{align*}
where $C_{\rho} > 0$ is a constant that only depends on $\rho$.
\end{myproof}
The following theorem presents a generalization of Theorem 4 in the paper. To simplify notation, the parameters of \(\mathscr{G}\) and \(\mathscr{G} \cdot \mathscr{V}_{\mathscr{R}}\) (Definitions 4 to 12, \ref{sa-defn: smooth tv}, \ref{sa-defn: smooth ktv}) are taken with \(\mathcal{C} = \mathcal{Q}_{\mathscr{G}}\), and the index \(\mathcal{Q}_{\mathscr{G}}\) is omitted where there is no ambiguity; the parameters of \(\mathscr{R}\) (Definitions 4 to 12) are taken with \(\mathcal{C} = \mathcal{Y}\), and the index \(\mathcal{Y}\) is omitted where there is no ambiguity; and the parameters of \(\mathscr{G} \times \mathscr{R}\) (Definitions 4 to 12, \ref{sa-defn: uniform covering number for product space}, \ref{sa-defn: uniform entropy integral for product space}) are taken with \(\mathcal{C} = \mathcal{Q}_{\mathscr{G}} \times \mathcal{Y}\), and the index \(\mathcal{Q}_{\mathscr{G}} \times \mathcal{Y}\) is omitted where there is no ambiguity.
\begin{thm}\label{sa-thm: R-Process -- Approximate Dyadic}
Suppose $(\mathbf{z}_i=(\mathbf{x}_i, y_i): 1 \leq i \leq n)$ are i.i.d. random vectors taking values in $(\mathbb{R}^{d+1}, \mathcal{B}(\mathbb{R}^{d+1}))$, where $\mathbf{x}_i$ has distribution $\mathbbm{P}_X$ supported on $\mathcal{X}\subseteq\mathbb{R}^d$, $y_i$ has distribution $\mathbbm{P}_Y$ supported on $\mathcal{Y}\subseteq\mathbb{R}$, and the following conditions hold.
\begin{enumerate}[label=(\roman*)]
\item $\mathscr{G} \subseteq \operatorname{Span}\{\mathbbm{1}_{\Delta_l}: 0 \leq l < L\}$ is a class of Haar functions on $(\mathbb{R}^d, \mathcal{B}(\mathbb{R}^d),\mathbbm{P}_X)$.
\item There exists a surrogate measure $\mathbb{Q}_\mathscr{G}$ for $\mathbbm{P}_X$ with respect to $\mathscr{G}$ such that $\{\Delta_l: 0 \leq l < L\}$ forms a \textit{quasi-uniform partition} of $\mathcal{Q}_\mathscr{G}$ with respect to $\mathbb{Q}_\mathscr{G}$:
\begin{align*}
\mathcal{Q}_\mathscr{G} \subseteq \sqcup_{0\leq l < L} \Delta_l \qquad\text{and}\qquad
\frac{\max_{0 \leq l < L}\mathbb{Q}_\mathscr{G}(\Delta_l)}{\min_{0 \leq l < L}\mathbb{Q}_\mathscr{G}(\Delta_l)} \leq \rho < \infty.
\end{align*}
\item $\mathscr{G}$ is a VC-type class with envelope function $\mathtt{M}_{\mathscr{G}}$ over $\mathcal{Q}_\mathscr{G}$ with $\mathtt{c}_{\mathscr{G}} \geq e$ and $\mathtt{d}_{\mathscr{G}} \geq 1$.
\item $\mathscr{R}$ is a real-valued pointwise measurable class of functions on $(\mathbb{R}, \mathcal{B}(\mathbb{R}),\mathbbm{P}_Y)$.
\item $\mathscr{R}$ is a VC-type class with envelope $M_{\mathscr{R},\mathcal{Y}}$ over $\mathcal{Y}$ with $\mathtt{c}_{\mathscr{R},\mathcal{Y}}\geq e$ and $\mathtt{d}_{\mathscr{R},\mathcal{Y}}\geq 1$, where $M_{\mathscr{R},\mathcal{Y}}(y) + \mathtt{pTV}_{\mathscr{R},(-|y|,|y|)} \leq \mathtt{v} (1 + |y|^{\alpha})$ for all $y \in \mathcal{Y}$, for some $\mathtt{v}>0$, and for some $\alpha\geq0$. Furthermore, if $\alpha>0$, then $\sup_{\mathbf{x} \in \mathcal{X}}\mathbbm{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] \leq 2$.
\end{enumerate}
Then, on a possibly enlarged probability space, there exists mean-zero Gaussian processes $(Z_n^R(g,r): g \in \mathscr{G}, r \in \mathscr{R})$ with almost sure continuous trajectory such that:
\begin{itemize}
\item $\mathbbm{E}[R_n(g_1, r_1) R_n(g_2, r_2)] = \mathbbm{E}[Z^R_n(g_1, r_1) Z^R_n(g_2, r_2)]$ for all $(g_1, r_1), (g_2, r_2) \in \mathscr{G} \times \mathscr{R}$, and
\item $\mathbbm{P}[\lVert R_n - Z_n^R \rVert_{\mathscr{G} \times \mathscr{R}} > C_1 C_{\mathtt{v},\alpha} C_{\rho} \min_{\delta \in (0,1)}(\mathsf{H}^R_n(t,\delta) + \mathsf{F}^R_n(t,\delta)) +\mathsf{W}_n(t))] \leq C_2 e^{-t} + L e^{-C_{\rho}n/L}$ for all $t > 0$,
\end{itemize}
where $C_1$ and $C_2$ are universal constants, $C_{\mathtt{v},\alpha} = \mathtt{v} \max\{1 + (2 \alpha)^{\frac{\alpha}{2}}, 1 + (4 \alpha)^{\alpha}\}$, $C_{\rho}$ is a constant that only depends on $\rho$,
\begin{align*}
\mathsf{H}^R_n(t,\delta) & = \sqrt{\frac{L \mathtt{M}_{\mathscr{G}} \mathtt{E}_{\mathscr{G}}}{n}}\left(t + \log \mathtt{N}_{\mathscr{G}}(\delta/2) + \log \mathtt{N}_{\mathscr{R}}(\delta/2) + \log_2 N^{\ast}\right)^{\alpha + \frac{1}{2}} \\
&\qquad + \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} (\log n)^{\alpha}\left(t + \log \mathtt{N}_{\mathscr{G}}(\delta/2) + \log \mathtt{N}_{\mathscr{R}}(\delta/2) + \log_2 N^{\ast}\right)^{\alpha + 1},\\
\mathsf{W}_n(t) & = \mathbbm{1}(|\mathscr{R}|>1)\sqrt{\mathtt{M}_{\mathscr{G}} \mathtt{E}_{\mathscr{G}}} \Big(\max_{0 \leq l < L} \lVert \Delta_l \rVert_{\infty}\Big) \mathtt{L}_{\mathscr{V}_{\mathscr{R}}} \sqrt{t + \log \mathtt{N}_{\mathscr{G}}(\delta/2) + \log \mathtt{N}_{\mathscr{R}}(\delta/2) + \log_2 N^{\ast}}.
\end{align*}
with $\mathscr{V}_{\mathscr{R}} = \{\theta(\cdot,r): \mathbf{x} \mapsto \mathbbm{E}[r(y_i)|\mathbf{x}_i = \mathbf{x}], \mathbf{x} \in \mathcal{X}, r \in \mathscr{R}\}$ and $N^{\ast}=\lceil \log_2 (\frac{n \mathtt{M}_{\mathscr{G}}}{2^L \mathtt{E}_{\mathscr{G}}}) \rceil$.
\end{thm}
\begin{myproof}{Theorem~\ref{sa-thm: R-Process -- Approximate Dyadic}}
By the same reduction through surrogate measure, we can w.l.o.g. assume $\mathbb{Q}_{\mathscr{G}} = \mathbbm{P}_X$. Suppose $2^M \leq J <2^{M+1}$. By the same cell divisions in the proof for Theorem~\ref{sa-thm: M-Process -- Approximate Dyadic Thm}, there exists a quasi-dyadic expansion $\mathscr{C}_{M+1,N}$ such that
\begin{align*}
\operatorname{Span}\left(\left\{\mathbbm{1}(\Delta_j): 0 \leq j < J \right\}\right) \subseteq \operatorname{Span}\left(\left\{\mathbbm{1}(\mathcal{X}_{0,l}): 0 \leq l < 2^{M+1}\right\}\right).
\end{align*}
By definition, the projection error can be decomposed as
\begin{align*}
R_n(g,r) - \mathtt{\Pi}_2 R_n(g,r) & = G_n(g,r) - \mathtt{\Pi}_{1} G_n(g,r) + X_n(g \, \theta(\cdot,r)) - \mathtt{\Pi}_{0} X_n(g \, \theta(\cdot,r)),
\end{align*}
where $\mathtt{\Pi}_{0}$ denotes the $L_2$-projection from $L_2(\mathbb{R}^d)$ to $\operatorname{Span}(\{\mathbbm{1}(\mathcal{X}_{0,l}): 0 \leq l < 2^{M + 1}\})$. For any $g \in \mathscr{G}$, since $g \in \operatorname{Span}(\{\mathbbm{1}(\mathcal{X}_{0,l}): 0 \leq l < 2^{M + 1}\})$,
\begin{align*}
\mathbbm{E} \left[(X_n(g \, \theta(\cdot,r)) - \mathtt{\Pi}_{0} X_n(g \, \theta(\cdot,r)))^2\right] & = \sum_{0 \leq j < J}\mathbbm{P}_X(\Delta_j)g^2|_{\Delta_j} \mathbbm{E} \left[ (\theta(\mathbf{x}_i,\mathbf{x}) - \mathtt{\Pi}_{0} \theta(\mathbf{x}_i,\mathbf{x}))^2|\mathbf{x}_i \in \Delta_j\right] \\
& \leq \mathbbm{E}[g(\mathbf{x}_i)^2] \max_{0 \leq j < J}\lVert \Delta_j \rVert_{\infty}^2 \mathtt{L}_{\mathscr{V}_{\mathscr{R}}}^2 \\
& \leq \mathtt{M}_{\mathscr{G}} \mathtt{E}_{\mathscr{G}} \max_{0 \leq j < J}\lVert \Delta_j \rVert_{\infty}^2 \mathtt{L}_{\mathscr{V}_{\mathscr{R}}}^2.
\end{align*}
Then $X_n(g \, \theta(\cdot,r)) - \mathtt{\Pi}_{0} X_n(g \, \theta(\cdot,r))$ is bounded through Bernstein inequality and union bound, for all $t > 0$,
\begin{align*}
\mathbbm{P} \left(\lVert X_n(g \, \theta(\cdot,r)) - \mathtt{\Pi}_{0} X_n(g \, \theta(\cdot,r)) \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}} \geq \frac{4}{3} \sqrt{\mathtt{M}_{\mathscr{G}} \mathtt{E}_{\mathscr{G}}} \max_{0 \leq j < J} \lVert \Delta_j \rVert_{\infty} \mathtt{L}_{\mathscr{V}_{\mathscr{R}}}\sqrt{t} + 2 C_{\mathtt{v},\alpha} \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} t
\right) \leq 2 \exp(-t).
\end{align*}
Combining Lemma~\ref{sa-lem: M- process -- SA error quasi-dyadic} and Equation~\eqref{sa-eq: M-process -- proj error}, and the same calculation as in the proof for Theorem~\ref{sa-thm: R-process -- main theorem} to get $\mathtt{C}_{\mathtt{\Pi}_2(\mathscr{G}, \mathscr{R})} \lesssim (C_{\mathtt{v},\alpha} \mathtt{M}_{\mathscr{G}} N^{\alpha})^2$, for all $t > N_{\ast}$, with probability at least $1 - 2 \mathtt{N}_{\mathscr{G} \times \mathscr{R}}(\delta, \mathtt{M}_{\mathscr{G}} M_{\mathscr{R}})e^{-t} - 2^M \exp(-C_{\rho}n 2^{-M})$,
\begin{align*}
\lVert R_n - Z_n^R \rVert_{(\mathscr{G} \times \mathscr{R})_{\delta}} \leq \frac{4}{3} \sqrt{\mathtt{M}_{\mathscr{G}} \mathtt{E}_{\mathscr{G}}} \max_{0 \leq j < J} \lVert \Delta_j \rVert_{\infty} \mathtt{L}_{\mathscr{V}_{\mathscr{R}}}\sqrt{t} + C_{\mathtt{v},\alpha} N_{\ast}^{\alpha + \frac{1}{2}} \sqrt{\frac{J \mathtt{E}_{\mathscr{G}} \mathtt{M}_{\mathscr{G}}}{n}} \sqrt{t} + C_{\mathtt{v},\alpha} \frac{\mathtt{M}_{\mathscr{G}}}{\sqrt{n}} t^{\alpha + 1},
\end{align*}
The rest follows from the error for fluctuation off the $\delta$-net given in Lemma~\ref{sa-lem: M- Process -- Fluctuation off the net}. The ``bias'' term $\sqrt{\mathtt{M}_{\mathscr{G}} \mathtt{E}_{\mathscr{G}}} \max_{0 \leq j < J} \lVert \Delta_j \rVert_{\infty} \mathtt{L}_{\mathscr{V}_{\mathscr{R}}}\sqrt{t}$ comes from $X_n(g \, \theta(\cdot,r)) - \mathtt{\Pi}_{0} X_n(g \, \theta(\cdot,r))$ in the decomposition.
In the special case that we have a singleton $\mathscr{R} = \{r\}$, we can get rid of the "bias" term by redefining $\varepsilon_i = \operatorname{sign}(r(y_i)-\mathbbm{E}[r(y_i)|\mathbf{x}_i])|r(y_i) - \mathbbm{E}[r(y_i)|\mathbf{x}_i]|^{1/\alpha}$. Take $\widetilde{r}(u) = \operatorname{sign}(u)|u|^{\alpha}, u \in \mathbb{R}$. In particular, $\mathbbm{E}[\widetilde{r}(\varepsilon_i)|\mathbf{x}_i] = 0$ almost surely. Either $r$ is bounded and we can take $\alpha = 0$, which makes $\widetilde{r}$ also bounded; or $\alpha > 0$ and $\sup_{\mathbf{x} \in \mathcal{X}}\mathbbm{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] \leq 2$ and $|r(u)| \lesssim 1 + |u|^{\alpha}$, which implies $\sup_{\mathbf{x} \in \mathcal{X}}\mathbbm{E}[\exp(|\varepsilon_i|
)|\mathbf{x}_i = \mathbf{x}] \lesssim 2$ and $\widetilde{r}$ has polynomial growth. Then for any $g \in \mathscr{G}$,
\begin{align*}
R_n(g,r) & = \frac{1}{\sqrt{n}}\sum_{i =1}^n g(\mathbf{x}_i) \widetilde{r}(\varepsilon_i) - \mathbbm{E}[g(\mathbf{x}_i) \widetilde{r}(\varepsilon_i)] = G_n^{\prime}(g,\widetilde{r}),
\end{align*}
where $G_n^{\prime}$ denotes the empirical process based on random sample $((\mathbf{x}_i,\varepsilon_i): 1 \leq i \leq n)$. The result then follows from Theorem~\ref{sa-thm: M-Process -- Approximate Dyadic Thm}. By similar arguments as in the proof of Theorem~\ref{sa-thm: R-Process -- Approximate Dyadic},
\begin{align*}
\mathtt{C}_{\mathtt{\Pi}_{1}(\mathscr{G}, \{\widetilde{r}\})} = \sup_{f \in \mathtt{\Pi}_{1}(\mathscr{G}, \{\widetilde{r}\})}\min\left\{\sup_{(j,k)} \left[\sum_{j^{\prime} < j} (j-j^{\prime})(j-j^{\prime}+1) 2^{j^{\prime}-j} \sum_{k^{\prime}: \mathcal{C}_{j^{\prime},k^{\prime}} \subseteq \mathcal{C}_{j,k}} \widetilde{\beta}_{j^{\prime},k^{\prime}}^2(f) \right], \lVert f \rVert_{\infty}^2(M + N)\right\},
\end{align*}
but $\widetilde{\beta}_{j,k}(f)$ vanishes for all $j > N$ and we obtain similarly $\mathtt{C}_{\mathtt{\Pi}_{1}(\mathscr{G}, \{\widetilde{r}\})} \lesssim (C_{\mathtt{v},\alpha}\mathtt{M}_{\mathscr{G}}N^{\alpha})^2$.
\end{myproof}
\subsubsection{Proof of Theorem 4}
By standard empirical process arguments, $\mathtt{N}_{\mathscr{G}}(\delta) \leq \mathtt{c}_{\mathscr{G}} \delta^{-\mathtt{d}_{\mathscr{G}}}$ and $\mathtt{N}_{\mathscr{R}}(\delta) \leq \mathtt{c}_{\mathscr{R}} \delta^{-\mathtt{d}_{\mathscr{R}}}$ for $\delta \in (0,1]$, and the result follows by Lemma~\ref{sa-thm: R-Process -- Approximate Dyadic}.\qed
\subsubsection{Proof of Corollary 6}
Take $t = C\log n$ with $C>1$ in Theorem 4.\qed
\subsection{Example: Haar Partitioning-based Regression}
The following lemma gives precise regularity conditions for the example in Section 5.3 of the paper.
\begin{lemma}[Haar Basis Regression Estimators]\label{sa-lem: haar basis estimator}
Consider the setup in Example 3, and assume in addition that $\sup_{r \in \mathscr{R}_{\ell}} \sup_{\mathbf{x},\mathbf{y} \in \mathcal{X}}|\theta(\mathbf{x},r) - \theta(\mathbf{y},r)|/\lVert \mathbf{x} - \mathbf{y} \rVert_{\infty} < \infty$ for $\ell = 1,2$.
If $\log (nL) L/n \to 0$, then
\begin{align*}
\sup_{r \in \mathscr{R}_{2}} \sup_{\mathbf{w} \in \mathcal{W}} \big|\mathbf{p}(\mathbf{w})^{\top}(\widehat{\mathbf{Q}}^{-1} - \mathbf{Q}^{-1})\mathbf{T}_r\big|
&= O(\log(nL)L/n) \qquad \text{a.s.}, \quad and\\
\sup_{r \in \mathscr{R}_{\ell}} \sup_{\mathbf{w} \in \mathcal{W}} \big|\mathbbm{E}[\check{\theta}(\mathbf{w},r)|\mathbf{x}_1,\cdots,\mathbf{x}_n] - \theta(\mathbf{w},r)\big|
&= O\big(\max_{0 \leq l < L} \lVert \Delta_l \rVert_{\infty}\big) \qquad \text{a.s.}, \quad l=1,2.
\end{align*}
If, in addition, $\sup_{\mathbf{x} \in \mathcal{X}}\mathbbm{E}[\exp(|y_i|)|\mathbf{x}_i = \mathbf{x}] \leq 2$, then
\begin{align*}
\sup_{r \in \mathscr{R}_{2}} \sup_{\mathbf{w} \in \mathcal{W}} \big|\mathbf{p}(\mathbf{w})^{\top}(\widehat{\mathbf{Q}}^{-1} - \mathbf{Q}^{-1})\mathbf{T}_r\big|
&= O(\log(nL)L/n + (\log n)(\log(nL)L/n)^{3/2}) \qquad \text{a.s.}
\end{align*}
\end{lemma}
\begin{myproof}{Lemma~\ref{sa-lem: haar basis estimator}}
We use the notation $\mathbbm{P}_X(\Delta_l) = \mathbbm{P}(\mathbf{x}_i \in \Delta_l)$, and $\widehat{\mathbbm{P}}_X(\Delta_l) = n^{-1} \sum_{i = 1}^n \mathbbm{1}(\mathbf{x}_i \in \Delta_l)$, $0 \leq l < L$.
\paragraph*{Non-linearity Errors:} For $\ell = 1,2$, $\mathbf{w} \in \mathcal{W}, r \in \mathscr{R}_{\ell}$, we have
\begin{align*}
\mathbf{p}(\mathbf{w})^{\top}(\widehat{\mathbf{J}}^{-1} - \mathbf{J}^{-1}) \mathbf{T}_r & = \sum_{0 \leq l < L} \mathbbm{1}(\mathbf{w} \in \Delta_l) (L^{-1} \widehat{\mathbbm{P}}_X(\Delta_l)^{-1} - L^{-1} \mathbbm{P}_X(\Delta_l)^{-1}) \frac{1}{n}\sum_{i = 1}^n \frac{\mathbbm{1}(\mathbf{x}_i \in \Delta_l)}{L^{-1}} \epsilon_i(r),
\end{align*}
where $\epsilon_i(r) = r(y_i) - \mathbbm{E}[r(y_i)|\mathbf{x}_i]$. By maximal inequality for sub-Gaussian random variables \citep[Lemma 2.2.2]{wellner2013weak-SA}, $\max_{0 \leq l < L} |L\widehat{\mathbbm{P}}_X(\Delta_l) - L\mathbbm{P}_X(\Delta_l)| = O(\sqrt{\frac{\log (n L)}{n/L}})$ a.s.. Since $\{\Delta_l: 0 \leq l < L\}$ is a quasi-uniform partition of $\mathcal{X}$ with respect to $\mathbbm{P}_X$, $\min_{0 \leq l < L} L \mathbbm{P}_X(\Delta_l) = \Omega(1)$. Hence
\begin{align}\label{sa-eq: proof of haar 1}
\max_{0 \leq l < L} |L^{-1} \widehat{\mathbbm{P}}_X(\Delta_l)^{-1} - L^{-1} \mathbbm{P}_X(\Delta_l)^{-1}| = O(\sqrt{(n/L)^{-1} \log (n L)}), \quad a.s..
\end{align}
Take $\mathscr{H}_{\ell} = \{(\mathbf{w},y) \mapsto L \mathbbm{1}(\mathbf{w} \in \Delta_l) (r(y) - \theta(\mathbf{w},r)): 0 \leq l < L, r \in \mathscr{R}_{\ell}\}$, for $\ell = 1,2$. In particular, if we take $\mathscr{G} = \{L\mathbbm{1}(\cdot \in \Delta_l): 0 \leq l < L\}$, then $\mathscr{G}$ is a VC-type class w.r.p. constant envelope $L$ with constant $\mathtt{c}_{\mathscr{G}} = L$ and exponent $\mathtt{d}_{\mathscr{G}} = 1$. In the main text, we explained that both $\mathscr{R}_1$ and $\mathscr{R}_2$ are VC-type class with $\mathtt{c}_{\mathscr{R}_1} = 1$, $\mathtt{d}_{\mathscr{R}_1} = 1$ and $\mathtt{c}_{\mathscr{R}_2}$ some universal constant, $\mathtt{d}_{\mathscr{R}_2} = 2$. By standard empirical process arguments, both $\mathscr{H}_{\ell}$'s are VC-type class with $\mathtt{c}_{\mathscr{H}_1} = L$, $\mathtt{d}_{\mathscr{H}_1} = 1$, $\mathtt{c}_{\mathscr{H}_2} = O(L)$, $\mathtt{d}_{\mathscr{H}_2} = 2$. Since $\sup_{r \in \mathscr{R}_{\ell}} \max_{0 \leq l < L} |\frac{1}{n} \sum_{i = 1}^n L \mathbbm{1}(\mathbf{x}_i \in \Delta_l) \epsilon_i(r)| = \sup_{h \in \mathscr{H}_{\ell}} |\mathbbm{E}_n[h(\mathbf{x}_i, y_i)] - \mathbbm{E}[h(\mathbf{x}_i,y_i)]|$ is the suprema of empirical process, by Corollary 5.1 in \cite{chernozhukov2014gaussian-SA},
\begin{equation}\label{sa-eq: proof of haar 2}
\begin{aligned}
\sup_{r \in \mathscr{R}_1} \max_{0 \leq l < L} \Big|\frac{1}{n} \sum_{i = 1}^n L \mathbbm{1}(\mathbf{x}_i \in \Delta_l) \epsilon_i(r) \Big|
& = O\bigg(\sqrt{\frac{\log (n L)}{n/L}} + \log (n) \frac{\log(n L)}{n/L}\bigg) \quad a.s., \\
\sup_{r \in \mathscr{R}_2} \max_{0 \leq l < L} \Big|\frac{1}{n} \sum_{i = 1}^n L \mathbbm{1}(\mathbf{x}_i \in \Delta_l) \epsilon_i(r) \Big|
& = O\bigg(\sqrt{\frac{\log (n L)}{n/L}}\bigg) \quad a.s..
\end{aligned}
\end{equation}
Putting together Equations~\eqref{sa-eq: proof of haar 1}, \eqref{sa-eq: proof of haar 2}, we have
\begin{align*}
\sup_{\mathbf{w} \in \mathcal{W}} \sup_{r \in \mathscr{R}_{\ell}} \Big|\mathbf{p}(\mathbf{w})^{\top}(\widehat{\mathbf{J}}^{-1} - \mathbf{J}^{-1}) \mathbf{T}_r\Big| = O \bigg(\frac{\log(nL)}{n/L}\bigg) + \mathbbm{1}(\ell = 1) O \bigg( \log(n)\bigg(\frac{\log(nL)}{n/L}\bigg)^{3/2} \bigg).
\end{align*}
\paragraph*{Smoothing Bias: } Since we have assumed that $\sup_{r \in \mathscr{R}_{\ell}} \sup_{\mathbf{x},\mathbf{y} \in \mathcal{X}}|\mu(\mathbf{x},r) - \mu(\mathbf{y},r)|/\lVert \mathbf{x} - \mathbf{y} \rVert_{\infty} < \infty$, $\ell = 1,2$,
\begin{align*}
\sup_{\mathbf{x} \in \mathcal{X}} \sup_{r \in \mathscr{R}_l} |\mathbbm{E}[\widehat{\mu}(\mathbf{x},r)|\mathbf{x}_1, \cdots, \mathbf{x}_n] - \mu(\mathbf{x},r)|
& = \bigg|\sum_{0 \leq l < L} \mathbbm{1}(\mathbf{x} \in \Delta_l) \frac{ \sum_{i = 1}^n \mathbbm{1}(\mathbf{x}_i \in \Delta_l) \mu(\mathbf{x}_i,r)}{\sum_{i = 1}^n \mathbbm{1}(\mathbf{x}_i \in \Delta_l)} - \mu(\mathbf{x},r) \bigg| \\
& = O(\max_{0 \leq l < L}\lVert \Delta_l \rVert_{\infty}).
\end{align*}
\end{myproof}
\clearpage
\bibliographystyle{jasa}
\bibliography{CY_2024_AOS--bib}