EconBase
← Back to paper

Cointegration without Unit Roots

The exact contents of citations.db main_text.text for this paper — one flattened LaTeX string, title through conclusion, appendix excluded, unmodified except for removing email addresses. This is what our citation measures are computed over.

102,602 characters

\RequirePackage{snapshot}
\documentclass[11pt,twoside,english,british]{article}
\usepackage[utf8]{inputenc}
\usepackage[a4paper]{geometry}
\geometry{verbose,tmargin=2.5cm,bmargin=2.5cm,lmargin=2.5cm,rmargin=2.5cm}
\usepackage{fancyhdr}
\pagestyle{fancy}
\usepackage{color}
\usepackage{babel}
\usepackage{array}
\usepackage{verbatim}
\usepackage{refstyle}
\usepackage{booktabs}
\usepackage{mathrsfs}
\usepackage{mathtools}
\usepackage{enumitem}
\usepackage{amsmath}
\usepackage{amsthm}
\usepackage{amssymb}
\usepackage{graphicx}
\usepackage{setspace}
\usepackage[authoryear,longnamesfirst]{natbib}
\onehalfspacing
\usepackage[unicode=true,
 bookmarks=true,bookmarksnumbered=false,bookmarksopen=false,
 breaklinks=true,pdfborder={0 0 0},pdfborderstyle={},backref=false,colorlinks=false]
 {hyperref}

\makeatletter


\AtBeginDocument{}
\AtBeginDocument{}
\AtBeginDocument{}
\AtBeginDocument{}
\AtBeginDocument{}
\AtBeginDocument{}
\AtBeginDocument{}
\AtBeginDocument{}
\AtBeginDocument{}
\AtBeginDocument{}
\AtBeginDocument{}
[(\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\otimes I_{q})-(I_{q}\otimes\Lambda_{{\scriptscriptstyle \textnormal{LU}}})](I_{q}\otimes G^{\mathsf{T}})B+(I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}})
\end{bmatrix}\label{eq:Jacobs}
\end{equation}
for $G^{\mathsf{T}}\coloneqq[0_{q\times r},I_{q}]$, $\beta^{\mathsf{T}}=[I_{r},-A]$,
and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}=\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$, etc.; and
\item \label{enu:deriv:cont}$J(\boldsymbol{\Phi})$ is continuous.
\end{enumerate}
\end{lem}

When $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=I_{q}$, the $pq\times pq$ matrix $J(\boldsymbol{\Phi})$
simplifies as follows.
\begin{lem}
\label{lem:derivatunity}Suppose $\boldsymbol{\Phi}\in\set P$ with $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=I_{q}$.
Then $J(\boldsymbol{\Phi})$ is nonsingular, and
\[
\begin{bmatrix}J_{A}(\boldsymbol{\Phi})\\
J_{\Lambda}(\boldsymbol{\Phi})
\end{bmatrix}=\begin{bmatrix}I_{q}\otimes\beta^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{ST}}}(I_{kp-q}-\Lambda_{{\scriptscriptstyle \textnormal{ST}}})^{-1}L_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\\
I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}
\end{bmatrix}.
\]
\end{lem}

\begin{proof}[Proof of \ref{lem:implicit-maps}]
 We first prove $\set P$ is open. For $F\in\mathbb{R}^{kp\times kp}$,
let $\lambda_{i}(F)$ denote the $i$th eigenvalue of $F$, when these
are placed in descending order of modulus. Let $\set F$ denote the
set of $kp\times kp$ matrices such that
\begin{enumerate}
\item $\smlabs{\lambda_{q+1}(F)}<\smlabs{\lambda_{q}(F)}$; and
\end{enumerate}
there exist $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}\in\mathbb{R}^{q\times q}$ and $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}\in\mathbb{R}^{kp\times q}$
such that
\begin{enumerate}[resume]
\item the eigenvalues of $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}$ are $\{\lambda_{i}(F)\}_{i=1}^{q}$,
$F\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=\largedec R_{{\scriptscriptstyle \textnormal{LU}}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}$; and
\item $\operatorname{rk}\{\largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}\}=q$, where $\largedec G^{\mathsf{T}}\coloneqq[0_{q\times(kp-q)},I_{q}]=[0_{q\times k(p-1)},G^{\mathsf{T}}]$.
\end{enumerate}
In view of \ref{lem:GLR}, $\boldsymbol{\Phi}\in\set P$ if and only if the companion
form matrix $F(\boldsymbol{\Phi})$ is in $\set F$. Since $F(\cdot)$ is trivially
continuous, it suffices to show that $\set F$ is open.

To that end, fix $F_{0}\in\set F$, and let $\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}$ and $\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}$
denote matrices satisfying (ii) and (iii) above. By the continuity
of eigenvalues and simple invariant subspaces (Theorems~IV.1.1 and
V.2.8 in \citealp{SS90}), for every $\epsilon>0$ there exists a
$\delta>0$ such that whenever $\smlnorm{F-F_{0}}<\delta$, $F$ satisfies
requirements (i) and (ii) above, with associated $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}$ such
that $\smlnorm{\largedec R_{{\scriptscriptstyle \textnormal{LU}}}-\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}}<\epsilon$. Since the set of full
rank matrices is open, we may take $\epsilon>0$ sufficiently small
such that (iii) also holds. Thus $F\in\set F$, and so $F_{0}$ is
an interior point of $\set F$; deduce $\set F$ is open.

We turn next to the smoothness of $A(\boldsymbol{\Phi})$ and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$.
For $F_{0}\in\set F$ we have the invariant subspace decomposition
(as per \ref{eq:invsubdecomp} above)
\begin{equation}
F_{0}=\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}+\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\label{eq:Fsubs}
\end{equation}
where $\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}$ and $\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}$ satisfy (ii)--(iii) above.
Since (iii) holds, we may choose $\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}$ such that $\largedec G^{\mathsf{T}}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=I_{q}$;
note that $\largedec L_{0}^{\mathsf{T}}\largedec R_{0}=I_{kp}$ (as per \ref{lem:GLR}\ref{enu:GLR:L})
implies $\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=I_{q}$. Define the maps\begin{subequations}\label{eq:HHast}
\begin{align}
H(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}};F) & \coloneqq\begin{bmatrix}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}-F\largedec R_{{\scriptscriptstyle \textnormal{LU}}}; & \largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}-I_{q}\end{bmatrix}\\
H^{\ast}(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}};F) & \coloneqq\begin{bmatrix}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}-F\largedec R_{{\scriptscriptstyle \textnormal{LU}}}; & \largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}-I_{q}\end{bmatrix},\label{eq:Hast}
\end{align}
\end{subequations}so that $H(\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}},\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}};F_{0})=H^{\ast}(\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}},\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}};F_{0})=0$;
but note that these maps need not otherwise agree, since they impose
distinct normalisations on $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}$. Once we have shown that the
Jacobian of $H^{\ast}$ with respect to $(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}})$
is nonsingular at $(\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}},\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}};F_{0})$, it will follow
by the implicit mapping theorem (\citealp[Thm.~XIV.2.1]{Lang93})
that there exists a neighbourhood $N\subset\set F$ of $F_{0}$ and
smooth functions $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}:N\ensuremath{\rightarrow}\mathbb{R}^{kp\times q}$, $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}:N\ensuremath{\rightarrow}\mathbb{R}^{q\times q}$
such that
\[
H^{\ast}[\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F),\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F);F]=0
\]
for all $F\in N$; by the continuity of $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\cdot)$,
we may choose $N$ such that $\operatorname{rk}\{\largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F)\}=q$
for all $F\in N$. Thus
\begin{align}
\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(F) & \coloneqq\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F)[\largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F)]^{-1}\label{eq:renorm1}\\
\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(F) & \coloneqq[\largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F)]\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F)[\largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F)]^{-1}\label{eq:renorm2}
\end{align}
are well defined for all $F\in N$, and have the property that
\[
H[\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(F),\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(F);F]=0
\]
for all $F\in N$. Since the $(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}})$ satisfying
$H(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}};F)=0$ is unique, repeating this construction
for every $F_{0}\in\set F$ allows the smooth maps $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(F)$
and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(F)$ to be extended to the whole of $\set F$.
The smoothness of $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})\coloneqq\Lambda_{{\scriptscriptstyle \textnormal{LU}}}[F(\boldsymbol{\Phi})]$
and $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})\coloneqq\largedec R_{{\scriptscriptstyle \textnormal{LU}}}[F(\boldsymbol{\Phi})]$ follows immediately,
and that of $A(\boldsymbol{\Phi})$ by noting that it corresponds to rows $(k-1)p+1$
to $(k-1)p+r$ of $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$.

It thus remains to verify that the Jacobian of $H^{\ast}$ with respect
to $(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}})$ is nonsingular at $(\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}},\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}};F_{0})$.
Matrix differentiation gives
\[
\ensuremath{\mathrm{d}} H^{\ast}=\begin{bmatrix}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}(\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}})+(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}-F_{0}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}); & \largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})\end{bmatrix}\eqqcolon\begin{bmatrix}\ensuremath{\mathrm{d}} H_{1}^{\ast}; & \ensuremath{\mathrm{d}} H_{2}^{\ast}\end{bmatrix}
\]
The Jacobian is nonsingular if $\ensuremath{\mathrm{d}} H^{\ast}=0$ implies $\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=0$
and $\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}=0$. To that end, suppose $\ensuremath{\mathrm{d}} H^{\ast}=0$.
Then $0=\ensuremath{\mathrm{d}} H_{2}^{\ast}=\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})$,
and
\[
\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=(\largedec R_{0}\largedec L_{0}^{\mathsf{T}})\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=(\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}+\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=(\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}
\]
and similarly, by \ref{eq:Fsubs} above,
\[
F_{0}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})=(\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}+\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}).
\]
Hence
\begin{align*}
\ensuremath{\mathrm{d}} H_{1}^{\ast} & =\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}(\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}})+\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}[\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}-\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})]\\
 & =\begin{bmatrix}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}} & \largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\end{bmatrix}\begin{bmatrix}\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}\\
\mathcal{T}[\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})]
\end{bmatrix},
\end{align*}
where $\mathcal{T}(M)\coloneqq M\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}-\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}}M.$
Since $\largedec R_{0}$ is nonsingular, $\ensuremath{\mathrm{d}} H_{1}^{\ast}=0$ implies that
$\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}=0$ and $\mathcal{T}[\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})]=0$;
but since $\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}$ and $\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}}$ have no common
eigenvalues, $\mathcal{T}(M)=0$ if and only if $M=0$ (\citealp{SS90},
Thm~V.1.3). Thus $\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})=0$, whence
\[
\begin{bmatrix}\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\\
\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}
\end{bmatrix}\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=0
\]
from which it follows that $\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=0$, since $\largedec L_{0}$ is
nonsingular.
\end{proof}

\begin{proof}[Proof of \ref{lem:derivatives}]
 \textbf{({\romannumeral 1}).} We have
\[
R_{0,{\scriptscriptstyle \textnormal{LU}}}\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{k}-\sum_{i=1}^{k}\Phi_{i}R_{0,{\scriptscriptstyle \textnormal{LU}}}\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{k-i}=\boldsymbol{\Phi}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=_{(1)}\boldsymbol{\Phi}_{0}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=R_{0,{\scriptscriptstyle \textnormal{LU}}}\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{k}-\sum_{i=1}^{k}\Phi_{0,i}R_{0,{\scriptscriptstyle \textnormal{LU}}}\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{k-i}=_{(2)}0
\]
where $=_{(1)}$ is by hypothesis, and $=_{(2)}$ by \ref{lem:GLR}.
Since $\smlabs{\lambda_{q+1}(\boldsymbol{\Phi})}<\smlabs{\lambda_{q}(\boldsymbol{\Phi}_{0})}=\smlabs{\lambda_{q}(\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}})}$
and $\boldsymbol{\Phi}\in\set P$, the result then follows by \ref{lem:GLR-converse}.

\textbf{({\romannumeral 2}).} Analogously to \ref{eq:HHast} above, define
\begin{align*}
H(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}};\boldsymbol{\Phi}) & \coloneqq\begin{bmatrix}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}-F(\boldsymbol{\Phi})\largedec R_{{\scriptscriptstyle \textnormal{LU}}}; & \largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}-I_{q}\end{bmatrix}\\
H^{\ast}(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}};\boldsymbol{\Phi}) & \coloneqq\begin{bmatrix}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}-F(\boldsymbol{\Phi})\largedec R_{{\scriptscriptstyle \textnormal{LU}}}; & \largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}-I_{q}\end{bmatrix}.
\end{align*}
By the argument given in the proof of \ref{lem:implicit-maps}, there
are smooth maps $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$, $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})$, $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$
and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})$ such that $H[\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}),\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi});\boldsymbol{\Phi}]=0$
and $H^{\ast}[\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi}),\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi});\boldsymbol{\Phi}]=0$
for all $\boldsymbol{\Phi}\in\set P$. Since $G^{\mathsf{T}}R_{0,{\scriptscriptstyle \textnormal{LU}}}=I_{q}$ implies
that $\largedec G^{\mathsf{T}}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=I_{q}$, we have $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})=\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}$
and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})=\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}$
when $\boldsymbol{\Phi}=\boldsymbol{\Phi}_{0}$, but otherwise these maps need not agree. Since
the maps $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})$ and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})$
are easier to work with, we first obtain the derivatives of these,
and subsequently those of $A(\boldsymbol{\Phi})$ and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$ via
renormalisation, analogously to \ref{eq:renorm1}--\ref{eq:renorm2}.

Setting the total differential of $H^{\ast}$ to zero gives
\begin{equation}
0=\ensuremath{\mathrm{d}} H^{\ast}=\begin{bmatrix}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}(\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})+(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}-F_{0}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})-F(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}; & \largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})\end{bmatrix}\label{eq:dHastzero}
\end{equation}
where $F_{0}\coloneqq F(\boldsymbol{\Phi})$, whence by similar arguments as were
given in the proof of \ref{lem:implicit-maps},
\begin{equation}
F(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}(\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})+\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}-\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}).\label{eq:totaldiff}
\end{equation}
Vectorising gives
\begin{align}
\operatorname{vec}[F(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}] & =(I_{q}\otimes\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}})\operatorname{vec}(\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})+M\operatorname{vec}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})\label{eq:vecFdP}
\end{align}
for $M\coloneqq(I_{q}\otimes\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}})[(\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\otimes I_{kp-q})-(I_{q}\otimes\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}})](I_{q}\otimes\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})$.
Since $\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=0$ and $\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}=I_{kp-q}$,
setting
\[
M^{\dagger}\coloneqq(I_{q}\otimes\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}})[(\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\otimes I_{kp-q})-(I_{q}\otimes\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}})]^{-1}(I_{q}\otimes\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})
\]
we have $M^{\dagger}(I_{q}\otimes\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}})=0$ and $M^{\dagger}M=I_{q}\otimes\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}$.
Since $\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})=0$ by \ref{eq:dHastzero},
it follows that
\[
\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}=(\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}+\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}=(\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}
\]
whence $M^{\dagger}M\operatorname{vec}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})=\operatorname{vec}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})$,
and so premultiplying \ref{eq:vecFdP} by $M^{\dagger}$ gives
\begin{align*}
\operatorname{vec}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}) & =M^{\dagger}\operatorname{vec}[F(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}^{\ast}].
\end{align*}
By the structure of the companion form matrix, $\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}F(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}$.
Since $R$ is given by the final $p$ rows of $\largedec R$, we have
\begin{align}
\operatorname{vec}(\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}) & =(I_{q}\otimes R_{0,{\scriptscriptstyle \textnormal{ST}}})[(\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\otimes I_{kp-q})-(I_{q}\otimes\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}})]^{-1}(I_{q}\otimes L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})\operatorname{vec}\{(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}\}\nonumber \\
 & =B(\boldsymbol{\Phi}_{0})\operatorname{vec}\{(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}\}.\label{eq:dRast}
\end{align}

To compute the Jacobian of $A(\boldsymbol{\Phi})$, note that by partitioning the
$p\times p$ identity matrix as
\[
\begin{bmatrix}G_{\perp} & G\end{bmatrix}\coloneqq\begin{bmatrix}I_{r} & 0\\
0 & I_{q}
\end{bmatrix}
\]
we have $A(\boldsymbol{\Phi})=G_{\perp}^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=G_{\perp}^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})[G^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})]^{-1}$.
From $R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi}_{0})=R_{0,{\scriptscriptstyle \textnormal{LU}}}$, $G^{\mathsf{T}}R_{0,{\scriptscriptstyle \textnormal{LU}}}=\largedec G^{\mathsf{T}}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=I_{q}$
and $G_{\perp}^{\mathsf{T}}R_{0,{\scriptscriptstyle \textnormal{LU}}}=A_{0}$, it follows that at $\boldsymbol{\Phi}=\boldsymbol{\Phi}_{0}$
\begin{align}
\ensuremath{\mathrm{d}} A & =G_{\perp}^{\mathsf{T}}(\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})-(G_{\perp}^{\mathsf{T}}R_{0,{\scriptscriptstyle \textnormal{LU}}})G^{\mathsf{T}}(\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})=(G_{\perp}^{\mathsf{T}}-A_{0}G^{\mathsf{T}})\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}=\beta_{0}^{\mathsf{T}}\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}\label{eq:dA}
\end{align}
for $\beta_{0}^{\mathsf{T}}=[I_{r},-A_{0}]$. The first part of \ref{eq:Jacobs}
follows immediately from \ref{eq:dRast} and \ref{eq:dA}. For the Jacobian
of $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$, note that (as per \ref{eq:renorm2} above)
\[
\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=[\largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})]\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})[\largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})]^{-1}
\]
whence at $\boldsymbol{\Phi}=\boldsymbol{\Phi}_{0}$,
\begin{align*}
\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}} & =\largedec G^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}+\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}-\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}\largedec G^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}).
\end{align*}
Recognising that $\largedec G^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})=G^{\mathsf{T}}(\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})$
and vectorising, we have
\begin{equation}
\operatorname{vec}(\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}})=\{(\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\otimes I_{q})-(I_{q}\otimes\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}})\}(I_{q}\otimes G^{\mathsf{T}})\operatorname{vec}(\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})+\operatorname{vec}(\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}).\label{eq:dLam}
\end{equation}
$\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}$ is given in \ref{eq:dRast} above. To obtain
$\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}$, note that premultiplying \ref{eq:totaldiff}
by $\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}$ yields
\begin{equation}
\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}=\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}F(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}.\label{eq:dLamast}
\end{equation}
Thus \ref{eq:dRast}, \ref{eq:dLam} and \ref{eq:dLamast} give the second
part of \ref{eq:Jacobs}.

\textbf{({\romannumeral 3}).} Continuity of $J(\boldsymbol{\Phi})$ is immediate from $A(\boldsymbol{\Phi})$
and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$ being smooth.
\end{proof}

\begin{proof}[Proof of \ref{lem:derivatunity}]
 The stated expression for $J(\boldsymbol{\Phi})$ is immediate from \ref{eq:Bdef},
\ref{lem:derivatives}, and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=I_{q}$. That $J(\boldsymbol{\Phi})$
is nonsingular will follow once we have shown that the $(p\times p)$
matrix
\begin{equation}
K\coloneqq\begin{bmatrix}\beta^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{ST}}}(I_{kp-q}-\Lambda_{{\scriptscriptstyle \textnormal{ST}}})^{-1}L_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\\
L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}
\end{bmatrix}\label{eq:K-matrix}
\end{equation}
is nonsingular. We first note the following facts. Since $\boldsymbol{\Phi}\in\set P$
with $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=I_{q}$, it follows from \ref{eq:AlambdaR}
that $\operatorname{rk}\Phi(1)\leq p-q$. Since $\Phi(\cdot)$ has exactly $q$
roots at unity, the reverse inequality holds by Corollary~4.3 of
\citet{Joh95}, whence $\operatorname{rk}\Phi(1)=p-q$. Thus \ref{ass:J} holds:
this implies that $\operatorname{sp}\beta=\operatorname{sp}\Phi(1)^{\mathsf{T}}$ and $\operatorname{rk} L_{{\scriptscriptstyle \textnormal{LU}}}=q$
(see \ref{lem:qcs} and the characterisation of the CS discussed in
\ref{subsec:coint-ur}).

Now let $c\in\mathbb{R}^{p}$ be such that $Kc=0$, so that in particular
$L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}c=0$. Since $\operatorname{rk}\Phi(1)+\operatorname{rk} L_{{\scriptscriptstyle \textnormal{LU}}}=p$, while
\ref{eq:eig-eig} with $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}=I_{q}$ implies $L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Phi(1)=0$,
it follows that $c\in\operatorname{sp}\Phi(1)$, i.e.\ $c=\Phi(1)b$ for some
$b\in\mathbb{R}^{p}$. By \citet[Thm~2.4]{GLR82}, $\Phi(\mu)^{-1}=R(\mu I-\Lambda)^{-1}L^{\mathsf{T}}$
for any $\mu$ not a root of $\Phi(\cdot)$. Since the columns of
the quasi-cointegrating matrix $\beta$ are orthogonal to $R_{{\scriptscriptstyle \textnormal{LU}}}$,
we have
\begin{equation}
\beta^{\mathsf{T}}=\beta^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{ST}}}(\mu I_{kp-q}-\Lambda_{{\scriptscriptstyle \textnormal{ST}}})^{-1}L_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\Phi(\mu)\ensuremath{\rightarrow}\beta^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{ST}}}(I_{kp-q}-\Lambda_{{\scriptscriptstyle \textnormal{ST}}})^{-1}L_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\Phi(1)\label{eq:betastuff}
\end{equation}
by the continuity of the r.h.s., as $\mu\ensuremath{\rightarrow}1$, since $\Lambda_{{\scriptscriptstyle \textnormal{ST}}}$
has no eigenvalues at unity. Hence
\[
0=Kc=\begin{bmatrix}\beta^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{ST}}}(I_{kp-q}-\Lambda_{{\scriptscriptstyle \textnormal{ST}}})^{-1}L_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\Phi(1)b\\
0
\end{bmatrix}=\begin{bmatrix}\beta^{\mathsf{T}}b\\
0
\end{bmatrix}
\]
implying $\beta^{\mathsf{T}}b=0$. But $\operatorname{sp}\beta=\operatorname{sp}\Phi(1)^{\mathsf{T}}$,
so we must have $\Phi(1)b=0$. Thus $c=0$, from which it follows
that $K$ is nonsingular.
\end{proof}


\section{Asymptotics}

\label{app:asymptotics}

The assumptions \ref{ass:DGP} and \ref{ass:LOC} are maintained throughout
this appendix. We first recall some notation. Let $\boldsymbol{\Phi}_{0}\coloneqq\lim_{n\ensuremath{\rightarrow}\infty}\boldsymbol{\Phi}_{n}$,
where $\{\boldsymbol{\Phi}_{n}\}$ is the sequence specified by \ref{ass:LOC}.
Let $R_{n}\coloneqq[R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n}),R_{{\scriptscriptstyle \textnormal{ST}}}]$ and $\Lambda_{n}\coloneqq\operatorname{diag}\{\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{ST}}}\}$
be as in \ref{ass:LOC}. Take $\largedec R_{n}\coloneqq\operatorname{col}\{R_{n}\Lambda_{n}^{k-i}\}_{i=1}^{k}$
and $\largedec L_{n}\coloneqq(\largedec R_{n}^{\mathsf{T}})^{-1}$ as in \ref{lem:GLR}, and
partition these as $\largedec R_{n}=[\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}},\largedec R_{n,{\scriptscriptstyle \textnormal{ST}}}]$\label{app:RnSTdef}
and $\largedec L_{n}=[\largedec L_{n,{\scriptscriptstyle \textnormal{LU}}},\largedec L_{n,{\scriptscriptstyle \textnormal{ST}}}]$ (as per \ref{eq:partition});
note that both these matrices are convergent.

Let $z_{{\scriptscriptstyle \textnormal{LU}},t}\coloneqq\largedec L_{n,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\largedec x_{t}$ and $z_{{\scriptscriptstyle \textnormal{ST}},t}=\largedec L_{n,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\largedec x_{t}$
be as in \ref{lem:repasAR} (for $\boldsymbol{\Phi}=\boldsymbol{\Phi}_{n}$); these follow the
autoregressions given in \ref{eq:zproc}. Recall $E\sim\mathrm{BM}(\Sigma)$
and $Z_{C}(r)\coloneqq\int_{0}^{r}\mathrm{e}^{C(r-s)}L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\ensuremath{\mathrm{d}} E(s)$
from \ref{eq:Zproc}. For $i\in\{{\scriptscriptstyle \textnormal{LU}},{\scriptscriptstyle \textnormal{ST}}\}$, let $\bar{z}_{i,t}$ denote
the residual from an OLS regression of $\{\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}\}_{t=1}^{n}$
onto a constant and linear trend. Recall that $\bar{Z}_{C}$ denotes
the residual from an $L^{2}[0,1]$ projection of each sample path
of $Z_{C}$ onto a constant and linear trend. As in \ref{subsec:likelihood},
let $\hat{\Sigma}_{n}$ denote the unrestricted MLE for $\Sigma$,
i.e.\ the OLS residual variance matrix estimator.

Proofs of the following results appear at the end of this section.
\begin{lem}
\label{lem:wkconv}The following hold jointly:
\begin{enumerate}
\item \label{enu:wkconv:donsker}$n^{-1/2}\sum_{t=1}^{\smlfloor{nr}}\varepsilon_{t}\ensuremath{\rightsquigarrow} E(r)$
\item $n^{-1/2}z_{{\scriptscriptstyle \textnormal{LU}},\smlfloor{nr}}\ensuremath{\rightsquigarrow} Z_{C}(r)$
\item \label{enu:wkconv:Zc}$n^{-1/2}\bar{z}_{{\scriptscriptstyle \textnormal{LU}},\smlfloor{nr}}\ensuremath{\rightsquigarrow}\bar{Z}_{C}(r)$
\end{enumerate}
as weak convergences on the space of right-continuous functions $[0,1]\ensuremath{\rightarrow}\mathbb{R}^{m}$
(with respect to the uniform topology); and
\begin{enumerate}[resume]
\item \label{enu:wkconv:si}$n^{-1}\sum_{t=1}^{n}(\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}\otimes\varepsilon_{t})\ensuremath{\rightsquigarrow}\int_{0}^{1}[\bar{Z}_{C}(r)\otimes\ensuremath{\mathrm{d}} E(r)]\ensuremath{\,\ensuremath{\mathrm{d}}} r$
\item \label{enu:wkconv:mgclt}$n^{-1/2}\sum_{t=1}^{n}(\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}\otimes\varepsilon_{t})\ensuremath{\rightsquigarrow}\xi\sim\mathrm{N}[0,\Omega\otimes\Sigma]$
\item \label{enu:wkconv:varmat}$\hat{\Sigma}_{n}\ensuremath{\overset{p}{\ensuremath{\rightarrow}}}\Sigma$,
\end{enumerate}
where $\Omega\coloneqq\lim_{n\ensuremath{\rightarrow}\infty}\operatorname{var}(z_{{\scriptscriptstyle \textnormal{ST}},n})$ is positive
definite, and $\xi$ is independent of $E$.
\end{lem}
Now define the reparametrisation $\boldsymbol{\Phi}\ensuremath{\mapsto}\varphi$ by
\begin{equation}
\varphi\coloneqq\begin{bmatrix}\varphi_{{\scriptscriptstyle \textnormal{LU}}}\\
\varphi_{{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}=\begin{bmatrix}\operatorname{vec}\{(\boldsymbol{\Phi}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\}\\
\operatorname{vec}\{(\boldsymbol{\Phi}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{ST}}}\}
\end{bmatrix}=\operatorname{vec}\{(\boldsymbol{\Phi}-\boldsymbol{\Phi}_{n})\largedec R_{n}\},\label{eq:reparam}
\end{equation}
which is reversed by setting $\boldsymbol{\Phi}=\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi)\largedec L_{n}^{\mathsf{T}}$,
where $\operatorname{vec}^{-1}(x)$ maps $x\in\mathbb{R}^{kp^{2}}$ to the matrix $X\in\mathbb{R}^{p\times kp}$
for which $\operatorname{vec}(X)=x$. The parameter space for $\varphi$ is the open
set
\begin{equation}
\mathcal{P}_{n}\coloneqq\{\operatorname{vec}[(\boldsymbol{\Phi}-\boldsymbol{\Phi}_{n})\largedec R_{n}]\mid\boldsymbol{\Phi}\in\set P\},\label{eq:Psetn}
\end{equation}
and the true parameters correspond to $\varphi=0$. Although $\mathcal{P}_{n}$
depends on $n$, since $\boldsymbol{\Phi}_{n}\ensuremath{\rightarrow}\boldsymbol{\Phi}_{0}\in\set P$ and $\set P$
is open (\ref{lem:implicit-maps}), there is an $\epsilon>0$ such
that $\mathcal{P}_{n}$ contains a ball of radius $\epsilon$ centred at
the origin, for all $n$ sufficiently large. Let
\[
\mathcal{\ell}^{\ast}_{n}(\varphi)\coloneqq\mathcal{\ell}_{n}[\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi)\largedec L_{n}^{\mathsf{T}},\hat{\Sigma}_{n}].
\]
Define $D_{n}\coloneqq\operatorname{diag}\{nI_{{\scriptscriptstyle \#}{\scriptscriptstyle \textnormal{LU}}},n^{1/2}I_{{\scriptscriptstyle \#}{\scriptscriptstyle \textnormal{ST}}}\}$, where
${\scriptscriptstyle \#}{\scriptscriptstyle \textnormal{LU}}\coloneqq pq$ and ${\scriptscriptstyle \#}{\scriptscriptstyle \textnormal{ST}}\coloneqq p(kp-q)$ correspond to the
dimensions of the vectors $\varphi_{{\scriptscriptstyle \textnormal{LU}}}$ and $\varphi_{{\scriptscriptstyle \textnormal{ST}}}$ respectively.
\begin{lem}
\label{lem:lhoodexp}There exist $S_{n}$ and $H_{n}$ such that for
all $\varphi\in\mathcal{P}_{n}$,
\[
\mathcal{\ell}^{\ast}_{n}(\varphi)-\mathcal{\ell}^{\ast}(0)=S_{n}^{\mathsf{T}}(D_{n}\varphi)-\tfrac{1}{2}(D_{n}\varphi)^{\mathsf{T}}H_{n}(D_{n}\varphi)
\]
where
\begin{gather*}
S_{n}\ensuremath{\rightsquigarrow}\begin{bmatrix}\int_{0}^{1}[\bar{Z}_{C}(r)\otimes\Sigma^{-1}\ensuremath{\mathrm{d}} E(r)]\\
\xi
\end{bmatrix}\eqqcolon\begin{bmatrix}S_{{\scriptscriptstyle \textnormal{LU}}}\\
S_{{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\eqqcolon S\\
H_{n}\ensuremath{\rightsquigarrow}\begin{bmatrix}\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}} & 0\\
0 & \Omega
\end{bmatrix}\otimes\Sigma^{-1}\eqqcolon\begin{bmatrix}H_{{\scriptscriptstyle \textnormal{LU}}} & 0\\
0 & H_{{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\eqqcolon H,
\end{gather*}
for $\xi$ as in \ref{lem:wkconv}.
\end{lem}
Define the constraint maps
\begin{align}
\theta_{n}(\varphi) & \coloneqq\operatorname{vec}\{\Lambda_{{\scriptscriptstyle \textnormal{LU}}}[\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi)\largedec L_{n}^{\mathsf{T}}]-(I_{q}+C/n)\}\label{eq:theta-n}\\
\gamma_{n}(\varphi) & \coloneqq a_{ij}[\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi)\largedec L_{n}^{\mathsf{T}}]-a_{ij}(\boldsymbol{\Phi}_{n}),\nonumber
\end{align}
and the associated restricted parameter spaces
\begin{align*}
\mathcal{P}_{n\mid\theta} & \coloneqq\{\varphi\in\mathcal{P}_{n}\mid\theta_{n}(\varphi)=0\}\\
\mathcal{P}_{n\mid\theta,\gamma} & \coloneqq\{\varphi\in\mathcal{P}_{n}\mid\theta_{n}(\varphi)=0\text{ and }\gamma_{n}(\varphi)=0\}.
\end{align*}
Let $\hat{\varphi}_{n}$, $\hat{\varphi}_{n\mid\theta}$ and $\hat{\varphi}_{n\mid\theta,\gamma}$
denote exact maximisers of $\mathcal{\ell}^{\ast}_{n}(\varphi)$ over the sets $\mathcal{P}_{n}$,
$\mathcal{P}_{n\mid\theta}$ and $\mathcal{P}_{n\mid\theta,\gamma}$ respectively:
which may be shown to exist at least with with probability approaching
one (w.p.a.1), and may be arbitrarily defined otherwise.
\begin{lem}
\label{lem:consistency} Each of $D_{n}\hat{\varphi}_{n}$, $D_{n}\hat{\varphi}_{n\mid\theta}$
and $D_{n}\hat{\varphi}_{n\mid\theta,\gamma}$ are $O_{p}(1)$.
\end{lem}
Let $\nabla_{\varphi}g(\varphi_{0})$ denote the gradient of $g:\mathcal{P}\ensuremath{\rightarrow}\mathbb{R}^{d_{g}}$
at $\varphi=\varphi_{0}$. The derivatives of the maps $\theta_{n}$ and $\gamma_{n}$
can be inferred from \ref{lem:derivatives}. Part~\ref{enu:deriv:values}
of that result gives the derivatives with respect to $\varphi_{{\scriptscriptstyle \textnormal{LU}}}$,
and part~\ref{enu:deriv:zero} implies that when $\varphi_{{\scriptscriptstyle \textnormal{LU}}}=0$, the
first (and higher order) derivatives with respect to $\varphi_{{\scriptscriptstyle \textnormal{ST}}}$ are
identically zero. Now letting $e_{d,i}\in\mathbb{R}^{d}$ denote a vector
with zero everywhere except for a $1$ in the $i$th position, define
\[
\Pi\coloneqq[\begin{matrix}\Theta; & \Gamma\end{matrix}]\coloneqq[\begin{matrix}I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}}}; & e_{q,j}\otimes L_{{\scriptscriptstyle \textnormal{ST}}}(I_{kp-q}-\Lambda_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})^{-1}R_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\beta e_{r,i}\end{matrix}],
\]
which by \ref{lem:derivatunity} has full column rank, and
\begin{align*}
\boldsymbol{\Theta} & \coloneqq\begin{bmatrix}\Theta\\
0_{\#{\scriptscriptstyle \textnormal{ST}}\times q^{2}}
\end{bmatrix} & \boldsymbol{\Pi} & \coloneqq\begin{bmatrix}\Pi\\
0_{\#{\scriptscriptstyle \textnormal{ST}}\times(q^{2}+1)}
\end{bmatrix}.
\end{align*}

\begin{lem}
\label{lem:deriv-limits}~
\begin{enumerate}
\item \label{enu:grad-lim}Let $\{\tilde{\varphi}_{n}\}$ denote a random sequence
in $\mathcal{P}_{n}$ with $\tilde{\varphi}_{n}\ensuremath{\overset{p}{\ensuremath{\rightarrow}}}0$. Then
\begin{align*}
\nabla_{\varphi}\theta_{n}(\tilde{\varphi}_{n}) & \ensuremath{\overset{p}{\ensuremath{\rightarrow}}}\boldsymbol{\Theta} & \nabla_{\varphi}\gamma_{n}(\tilde{\varphi}_{n}) & \ensuremath{\overset{p}{\ensuremath{\rightarrow}}}\boldsymbol{\Gamma}.
\end{align*}
\item \label{enu:proj-lim}Let $\mathcal{Q}_{\boldsymbol{\Theta},\perp}$ and $\mathcal{Q}_{\boldsymbol{\Pi},\perp}$
denote orthogonal projections from $\mathbb{R}^{kp^{2}}$ onto the subspaces
orthogonal to the the columns of $\boldsymbol{\Theta}$ and $\boldsymbol{\Pi}$. Then
\begin{align*}
D_{n}\hat{\varphi}_{n\mid\theta} & =\mathcal{Q}_{\boldsymbol{\Theta},\perp}D_{n}\hat{\varphi}_{n\mid\theta}+o_{p}(1)\\
D_{n}\hat{\varphi}_{n\mid\theta,\gamma} & =\mathcal{Q}_{\boldsymbol{\Pi},\perp}D_{n}\hat{\varphi}_{n\mid\theta,\gamma}+o_{p}(1).
\end{align*}
\end{enumerate}
\end{lem}
Let $\Theta_{\perp}\in\mathbb{R}^{pq\times qr}$ and $\Pi_{\perp}\in\mathbb{R}^{pq\times(qr-1)}$
denote matrices having full column rank, such that $\Theta_{\perp}^{\mathsf{T}}\Theta=0$
and $\Pi_{\perp}^{\mathsf{T}}\Pi=0$. We may take $\Theta_{\perp}=I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}},\perp}$,
for $L_{{\scriptscriptstyle \textnormal{LU}},\perp}$ a $p\times r$ matrix having rank $r$ and for
which $L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}L_{{\scriptscriptstyle \textnormal{LU}}}=0$. Since $\Pi=[\Theta,\Gamma]$
there exists a full column rank matrix $\Xi\in\mathbb{R}^{qr\times(qr-1)}$
for which $\Pi_{\perp}\coloneqq\Theta_{\perp}\Xi$.
\begin{prop}
\label{prop:andrews}~
\begin{enumerate}
\item \label{enu:andrews:unres}$D_{n}\hat{\varphi}_{n}=\begin{bmatrix}n\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}}\\
n^{1/2}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\ensuremath{\rightsquigarrow}\begin{bmatrix}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}S_{{\scriptscriptstyle \textnormal{LU}}}\\
H_{{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}$,
\item \label{enu:andrews:res}$D_{n}\hat{\varphi}_{n\mid\theta}=\begin{bmatrix}n\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}\\
n^{1/2}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta}
\end{bmatrix}\ensuremath{\rightsquigarrow}\begin{bmatrix}\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}\\
H_{{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}$,
\item \label{enu:andrews:lrroot}$2[\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n})-\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n\mid\theta})]\ensuremath{\rightsquigarrow} S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}\Theta(\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}\Theta)^{-1}\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}S_{{\scriptscriptstyle \textnormal{LU}}}$.
\end{enumerate}
Let $H_{\Theta,\perp}\coloneqq\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp}$,
and $\mathcal{Q}\in\mathbb{R}^{qr\times qr}$ denote the orthogonal projection
onto $\operatorname{sp} H_{\Theta,\perp}^{1/2}\Xi$. Then
\begin{enumerate}[resume]
\item \label{enu:andrews:coef}$2[\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n\mid\theta})-\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n\mid\theta,\gamma})]\ensuremath{\rightsquigarrow}(H_{\Theta,\perp}^{-1/2}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}})^{\mathsf{T}}[I_{qr}-\mathcal{Q}](H_{\Theta,\perp}^{-1/2}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}).$
\end{enumerate}
\end{prop}
The preceding gives the limiting distribution of $\hat{\boldsymbol{\Phi}}_{n}$
under the reparametrisation \ref{eq:reparam}; the limiting distributions
of estimators of $A$ and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}$ will then follow by an
application of the delta method, as per
\begin{prop}
\label{prop:deltamethod}Let $\{\boldsymbol{\Phi}_{n}\}$ be as in \ref{ass:LOC},
$\boldsymbol{\Phi}_{0}\coloneqq\lim_{n\ensuremath{\rightarrow}\infty}\boldsymbol{\Phi}_{n}\in\set P$, and $\{\tilde{\boldsymbol{\Phi}}_{n}\}$
a random sequence in $\set P$ with $\tilde{\boldsymbol{\Phi}}_{n}=\boldsymbol{\Phi}_{n}+o_{p}(1)$.
Then
\begin{equation}
\begin{bmatrix}\operatorname{vec}\{A(\tilde{\boldsymbol{\Phi}}_{n})-A(\boldsymbol{\Phi}_{n})\}\\
\operatorname{vec}\{\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\tilde{\boldsymbol{\Phi}}_{n})-\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n})\}
\end{bmatrix}=\left(\begin{bmatrix}J_{A}(\boldsymbol{\Phi}_{0})\\
J_{\Lambda}(\boldsymbol{\Phi}_{0})
\end{bmatrix}+o_{p}(1)\right)\operatorname{vec}\{(\tilde{\boldsymbol{\Phi}}_{n}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\}\label{eq:expansion}
\end{equation}
where $\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\coloneqq\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n})$.
\end{prop}
\begin{proof}[Proof of \ref{lem:wkconv}]
 \ref{enu:wkconv:donsker}--\ref{enu:wkconv:si} follow by Donsker's
theorem for partial sums, Lemma~3.1 in \citet{Phi88Ecta} and the
continuous mapping theorem; \ref{enu:wkconv:mgclt} by the martingale
central limit theorem (\citealp[Thm.~3.2]{HH80}); and \ref{enu:wkconv:varmat}
by arguments similar to those given in Section~3.2.2 of \citet{Lut07}.
\end{proof}

\begin{proof}[Proof of \ref{lem:lhoodexp}]
 Let $\Phi_{i}\coloneqq\boldsymbol{\Phi}\largedec R_{n,i}$ and $\Phi_{n,i}\coloneqq\boldsymbol{\Phi}_{n}\largedec R_{n,i}$
for $i\in\{{\scriptscriptstyle \textnormal{LU}},{\scriptscriptstyle \textnormal{ST}}\}$. Then
\begin{align*}
\mathcal{\ell}_{n}(\boldsymbol{\Phi},\Sigma) & =-\frac{n}{2}\log\det\Sigma-\min_{m,d}\frac{1}{2}\sum_{t=1}^{n}\norm{y_{t}-m-dt-\boldsymbol{\Phi}\largedec y_{t-1}}_{\Sigma^{-1}}^{2}\\
 & =-\frac{n}{2}\log\det\Sigma-\min_{m,d}\frac{1}{2}\sum_{t=1}^{n}\norm{x_{t}-m-dt-\boldsymbol{\Phi}\largedec x_{t-1}}_{\Sigma^{-1}}^{2}\\
 & =-\frac{n}{2}\log\det\Sigma-\min_{m,d}\frac{1}{2}\sum_{t=1}^{n}\norm{x_{t}-m-dt-\Phi_{{\scriptscriptstyle \textnormal{LU}}}z_{{\scriptscriptstyle \textnormal{LU}},t-1}-\Phi_{{\scriptscriptstyle \textnormal{ST}}}z_{{\scriptscriptstyle \textnormal{ST}},t-1}}_{\Sigma^{-1}}^{2}\\
 & =-\frac{n}{2}\log\det\Sigma-\frac{1}{2}\sum_{t=1}^{n}\norm{\bar{x}_{t}-\Phi_{{\scriptscriptstyle \textnormal{LU}}}\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}-\Phi_{{\scriptscriptstyle \textnormal{ST}}}\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}}_{\Sigma^{-1}}^{2}
\end{align*}
Twice differentiating the r.h.s.\ (as in \citealt[Sec.~3.4]{Lut07})
with respect to $\Phi_{{\scriptscriptstyle \textnormal{LU}}}$ and $\Phi_{{\scriptscriptstyle \textnormal{ST}}}$, and noting that $\varphi_{i}=\operatorname{vec}(\Phi_{i}-\Phi_{n,i})$,
we thus have
\begin{align*}
\mathcal{\ell}^{\ast}_{n}(\varphi)-\mathcal{\ell}^{\ast}_{n}(0)=\mathcal{\ell}_{n}(\boldsymbol{\Phi},\hat{\Sigma}_{n})-\mathcal{\ell}_{n}(\boldsymbol{\Phi}_{n},\hat{\Sigma}_{n}) & =S_{n}^{\mathsf{T}}(D_{n}\varphi)-\tfrac{1}{2}(D_{n}\varphi)^{\mathsf{T}}H_{n}(D_{n}\varphi)
\end{align*}
where
\begin{gather*}
S_{n}\coloneqq\begin{bmatrix}n^{-1}\sum_{t=1}^{n}(\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}\otimes\hat{\Sigma}_{n}^{-1}\bar{\varepsilon}_{t})\\
n^{-1/2}\sum_{t=1}^{n}(\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}\otimes\hat{\Sigma}_{n}^{-1}\bar{\varepsilon}_{t})
\end{bmatrix}=_{(1)}\begin{bmatrix}\frac{1}{n}\sum_{t=1}^{n}(\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}\otimes\hat{\Sigma}_{n}^{-1}\varepsilon_{t})\\
\frac{1}{n^{1/2}}\sum_{t=1}^{n}(\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}\otimes\hat{\Sigma}_{n}^{-1}\varepsilon_{t})
\end{bmatrix}\\
H_{n}\coloneqq\begin{bmatrix}n^{-2}\sum_{t=1}^{n}\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}^{\mathsf{T}} & n^{-3/2}\sum_{t=1}^{n}\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}^{\mathsf{T}}\\
n^{-3/2}\sum_{t=1}^{n}\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}^{\mathsf{T}} & n^{-1}\sum_{t=1}^{n}\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}^{\mathsf{T}}
\end{bmatrix}\otimes\hat{\Sigma}_{n}^{-1},
\end{gather*}
and $\bar{\varepsilon}_{t}$ denotes the residual from an OLS regression of
$\{\varepsilon_{t}\}_{t=1}^{n}$ on a constant and a linear trend; $=_{(1)}$
holds because each element of $\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}$ and $\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}$
is orthogonal to a constant and linear trend. The stated convergences
of $S_{n}$ and $H_{n}$ then follow by \ref{lem:wkconv} and the continuous
mapping theorem.
\end{proof}
\begin{proof}[Proof of \ref{lem:consistency}]
 By \ref{lem:lhoodexp}, we have
\begin{align*}
\mathcal{\ell}^{\ast}_{n}(\varphi)-\mathcal{\ell}^{\ast}_{n}(0) & \leq\smlnorm{D_{n}\varphi}[\smlnorm{S_{n}}-\tfrac{1}{2}\lambda_{\min}(H_{n})\smlnorm{D_{n}\varphi}].
\end{align*}
Let $M<\infty$ and $\epsilon>0$. Since $D_{n}=\operatorname{diag}\{nI_{{\scriptscriptstyle \#}{\scriptscriptstyle \textnormal{LU}}},n^{1/2}I_{{\scriptscriptstyle \#}{\scriptscriptstyle \textnormal{ST}}}\}$,
$S_{n}=O_{p}(1)$ and $H_{n}\ensuremath{\rightsquigarrow} H$ is positive definite w.p.a.1,
it is evident that
\begin{align*}
\ensuremath{\mathbb{P}}\left\{ \sup_{\{\varphi\in\mathcal{P}_{n}\mid\smlnorm{D_{n}\varphi}\geq M\}}[\mathcal{\ell}^{\ast}_{n}(\varphi)-\mathcal{\ell}^{\ast}_{n}(0)]<-\epsilon\right\}  & \ge\ensuremath{\mathbb{P}}\left\{ M[\smlnorm{S_{n}}-\tfrac{1}{2}\lambda_{\min}(H_{n})M]<-\epsilon\right\}
\end{align*}
and
\begin{align*}
\limsup_{n\ensuremath{\rightarrow}\infty}\ensuremath{\mathbb{P}}\left\{ M[\smlnorm{S_{n}}-\tfrac{1}{2}\lambda_{\min}(H_{n})M]<-\epsilon\right\}  & \geq\ensuremath{\mathbb{P}}\left\{ M[\smlnorm S-\tfrac{1}{2}\lambda_{\min}(H)M]<-\epsilon\right\} \ensuremath{\rightarrow}1
\end{align*}
as $M\ensuremath{\rightarrow}\infty$. Deduce that $D_{n}\hat{\varphi}_{n}=O_{p}(1)$. Since
$\mathcal{P}_{n\mid\theta,\gamma}\subset\mathcal{P}_{n\mid\theta}\subset\mathcal{P}_{n}$
and $0\in\mathcal{P}_{n\mid\theta,\gamma}$, that $D_{n}\hat{\varphi}_{n\mid\theta}$
and $D_{n}\hat{\varphi}_{n\mid\theta,\gamma}$ are stochastically bounded
follows by the same argument.
\end{proof}
\begin{proof}[Proof of \ref{lem:deriv-limits}]
 \textbf{({\romannumeral 1}).} Since $\boldsymbol{\Phi}_{n}\ensuremath{\rightarrow}\boldsymbol{\Phi}_{0}$, $\largedec L_{n}\ensuremath{\rightarrow}\largedec L_{0}$
and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\cdot)$ is continuously differentiable (\ref{lem:implicit-maps}),
\[
\nabla_{\varphi}\theta_{n}(\tilde{\varphi}_{n})\ensuremath{\overset{p}{\ensuremath{\rightarrow}}}\nabla_{\varphi}\operatorname{vec}\{\Lambda_{{\scriptscriptstyle \textnormal{LU}}}[\boldsymbol{\Phi}_{0}+\operatorname{vec}^{-1}(\varphi)\largedec L_{0}^{\mathsf{T}}]\}|_{\varphi=0}=_{(1)}\begin{bmatrix}I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}}}\\
0_{\#{\scriptscriptstyle \textnormal{ST}}\times q^{2}}
\end{bmatrix}=\boldsymbol{\Theta}
\]
where $=_{(1)}$ follows by Lemmas~\ref{lem:derivatives} and \ref{lem:derivatunity}.
The probability limit of $\nabla_{\varphi}\gamma_{n}(\tilde{\varphi}_{n})$ follows
similarly.

\textbf{({\romannumeral 2}).} By \ref{lem:consistency} and the remarks following
\ref{eq:Psetn}, there exists a ball $B(0,\epsilon)$ of radius $\epsilon>0$,
centred on the origin, such that $B(0,\epsilon)\subset\mathcal{P}_{n}$
for all $n$ sufficiently large, and $\ensuremath{\mathbb{P}}\{\hat{\varphi}_{n\mid\theta}\in B(0,\epsilon)\}\ensuremath{\rightarrow}1$.
We may take $\epsilon$ sufficiently small that $\boldsymbol{\Phi}_{\varphi}\coloneqq\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi)\largedec L_{n}^{\mathsf{T}}$
has $\smlabs{\lambda_{q+1}(\boldsymbol{\Phi}_{\varphi})}<\smlabs{\lambda_{q}(\boldsymbol{\Phi}_{n})}$
for all $n$ sufficiently large, for all $\varphi\in B(0,\epsilon)$. In
particular, suppose $\varphi_{{\scriptscriptstyle \textnormal{LU}}}=0$; then $(\boldsymbol{\Phi}_{\varphi}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}=0$
and we have by \ref{lem:derivatives}\ref{enu:deriv:zero} that $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{\varphi})=\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n})=C/n$.
It follows that $\theta_{n}(0,\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta})=0$ w.p.a.1.,
whence
\begin{multline*}
0=\theta_{n}(\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta},\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta})=\theta_{n}(\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta},\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta})-\theta_{n}(0,\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta})\\
=[\Theta+o_{p}(1)]^{\mathsf{T}}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}=\Theta^{\mathsf{T}}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}+o_{p}(\smlnorm{\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}})
\end{multline*}
by part~(i) of the lemma and a mean value expansion. Hence, letting
$\mathcal{Q}_{\Theta}$ and $\mathcal{Q}_{\Theta,\perp}$ denote the matrices
that orthogonally project from $\mathbb{R}^{\#{\scriptscriptstyle \textnormal{LU}}}$ onto $\operatorname{sp}\Theta$
and $(\operatorname{sp}\Theta)^{\perp}$ respectively, we have
\begin{align*}
D_{n}\hat{\varphi}_{n\mid\theta} & =\begin{bmatrix}nI_{\#{\scriptscriptstyle \textnormal{LU}}} & 0\\
0 & n^{1/2}I_{\#{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\begin{bmatrix}\mathcal{Q}_{\Theta}+\mathcal{Q}_{\Theta,\perp} & 0\\
0 & I_{\#{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\begin{bmatrix}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}\\
\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta}
\end{bmatrix}\\
 & =\begin{bmatrix}\mathcal{Q}_{\Theta,\perp} & 0\\
0 & I_{\#{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\begin{bmatrix}n\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}\\
n^{1/2}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta}
\end{bmatrix}+o_{p}(n\smlnorm{\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}})=\mathcal{Q}_{\boldsymbol{\Theta},\perp}D_{n}\hat{\varphi}_{n\mid\theta}+o_{p}(\smlnorm{D_{n}\hat{\varphi}_{n\mid\theta}}).\qedhere
\end{align*}
\end{proof}

\begin{proof}[Proof of \ref{prop:andrews}]
 \textbf{({\romannumeral 1}).} Immediate from \ref{lem:lhoodexp}.

\textbf{({\romannumeral 2}).} As in the proof of \ref{lem:deriv-limits}\ref{enu:proj-lim},
we may take $\epsilon>0$ such that $B(0,\epsilon)\subset\mathcal{P}_{n}$
for all $n$ sufficiently large, and $\ensuremath{\mathbb{P}}\{\hat{\varphi}_{n\mid\theta}\in B(0,\epsilon)\}\ensuremath{\rightarrow}1$.
Hence w.p.a.1., $\hat{\varphi}_{n\mid\theta}$ satisfies the first-order
conditions for a constrained interior maximum,
\[
\nabla_{\varphi}\mathcal{\ell}_{n}^{\ast}(\hat{\varphi}_{n\mid\theta})=D_{n}S_{n}-D_{n}H_{n}(D_{n}\hat{\varphi}_{n\mid\theta})=\nabla_{\varphi}\theta_{n}(\hat{\varphi}_{n\mid\theta})\mu_{n},
\]
where $\mu_{n}\in\mathbb{R}^{q^{2}}$ is a vector of Lagrange multipliers;
whence
\begin{equation}
S_{n}-H_{n}(D_{n}\hat{\varphi}_{n\mid\theta})=(nD_{n}^{-1})\nabla_{\varphi}\theta_{n}(\hat{\varphi}_{n\mid\theta})(n^{-1}\mu_{n})\eqqcolon\boldsymbol{\Theta}_{n}(n^{-1}\mu_{n})\label{eq:FOCconst}
\end{equation}
w.p.a.1. By a similar argument as given in the proof of \ref{lem:deriv-limits}\ref{enu:proj-lim},
it follows from \ref{lem:derivatives}\ref{enu:deriv:zero} that $\nabla_{\varphi_{{\scriptscriptstyle \textnormal{ST}}}}\theta_{n}(0,\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta})=0$
w.p.a.1, and so by a a mean value expansion and \ref{lem:consistency},
\[
\nabla_{\varphi_{{\scriptscriptstyle \textnormal{ST}}}}\theta_{n}(\hat{\varphi}_{n\mid\theta})=\nabla_{\varphi_{{\scriptscriptstyle \textnormal{ST}}}}\theta_{n}(\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta},\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta})-\nabla_{\varphi_{{\scriptscriptstyle \textnormal{ST}}}}\theta_{n}(0,\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta})=O_{p}(\smlnorm{\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}})=O_{p}(n^{-1}).
\]

Deduce from the preceding and \ref{lem:deriv-limits}\ref{enu:grad-lim}
that
\[
\boldsymbol{\Theta}_{n}=(nD_{n}^{-1})\nabla_{\varphi}\theta_{n}(\hat{\varphi}_{n\mid\theta})=\begin{bmatrix}\nabla_{\varphi_{{\scriptscriptstyle \textnormal{LU}}}}\theta_{n}(\hat{\varphi}_{n\mid\theta})\\
n^{1/2}\nabla_{\varphi_{{\scriptscriptstyle \textnormal{ST}}}}\theta_{n}(\hat{\varphi}_{n\mid\theta})
\end{bmatrix}\ensuremath{\overset{p}{\ensuremath{\rightarrow}}}\boldsymbol{\Theta},
\]
which has full column rank. Let $\boldsymbol{\Theta}_{\perp}\coloneqq\operatorname{diag}\{\Theta_{\perp},I_{\#{\scriptscriptstyle \textnormal{ST}}}\}$,
a full column rank matrix for which $\boldsymbol{\Theta}_{\perp}^{\mathsf{T}}\boldsymbol{\Theta}=0$;
then $\boldsymbol{\Theta}_{n,\perp}\coloneqq[I_{kp^{2}}-\boldsymbol{\Theta}_{n}(\boldsymbol{\Theta}_{n}^{\mathsf{T}}\boldsymbol{\Theta}_{n})^{-1}\boldsymbol{\Theta}_{n}^{\mathsf{T}}]\boldsymbol{\Theta}_{\perp}\ensuremath{\overset{p}{\ensuremath{\rightarrow}}}\boldsymbol{\Theta}_{\perp}$
and $\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}\boldsymbol{\Theta}_{n}=0$ for all $n$. Hence w.p.a.1
\begin{align*}
0 & =_{(1)}\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}S_{n}-\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}H_{n}(D_{n}\hat{\varphi}_{n\mid\theta})\\
 & =_{(2)}\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}S_{n}-\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}H_{n}[\boldsymbol{\Theta}_{\perp}(\boldsymbol{\Theta}_{\perp}^{\mathsf{T}}\boldsymbol{\Theta}_{\perp})^{-1}\boldsymbol{\Theta}_{\perp}^{\mathsf{T}}(D_{n}\hat{\varphi}_{n\mid\theta})+o_{p}(\smlnorm{D_{n}\hat{\varphi}_{n\mid\theta}})]
\end{align*}
where $=_{(1)}$ follows from premultiplying \ref{eq:FOCconst} by
$\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}$, and $=_{(2)}$ from \ref{lem:deriv-limits}\ref{enu:proj-lim}.
A further appeal to that result and rearranging the preceding yields
\[
D_{n}\hat{\varphi}_{n\mid\theta}=\mathcal{Q}_{\boldsymbol{\Theta},\perp}D_{n}\hat{\varphi}_{n\mid\theta}+o_{p}(\smlnorm{D_{n}\hat{\varphi}_{n\mid\theta}})=\boldsymbol{\Theta}_{\perp}(\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}H_{n}\boldsymbol{\Theta}_{\perp})^{-1}\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}S_{n}+o_{p}(1+\smlnorm{D_{n}\hat{\varphi}_{n\mid\theta}}).
\]
The result then follows by Lemmas \ref{lem:lhoodexp} and \ref{lem:consistency}.

\textbf{({\romannumeral 3}).} From parts~(i) and (ii) and \ref{lem:lhoodexp} we
have
\begin{gather}
2[\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n})-\mathcal{\ell}_{n}^{\ast}(0)]\ensuremath{\rightsquigarrow} S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}S_{{\scriptscriptstyle \textnormal{LU}}}+S_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{{\scriptscriptstyle \textnormal{ST}}}\label{eq:lr-uncon}\\
2[\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n\mid\theta})-\mathcal{\ell}_{n}^{\ast}(0)]\ensuremath{\rightsquigarrow} S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}+S_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{{\scriptscriptstyle \textnormal{ST}}}\label{eq:lr-con1}
\end{gather}
whence the result follows by subtracting \ref{eq:lr-con1} from \ref{eq:lr-uncon}
and noting that
\[
H_{{\scriptscriptstyle \textnormal{LU}}}^{-1/2}\Theta(\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}\Theta)^{-1}\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1/2}+H_{{\scriptscriptstyle \textnormal{LU}}}^{1/2}\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1}\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{1/2}=I_{pq}
\]
since the columns of $H_{{\scriptscriptstyle \textnormal{LU}}}^{-1/2}\Theta$ and $H_{{\scriptscriptstyle \textnormal{LU}}}^{1/2}\Theta_{\perp}$
are mutually orthogonal, and collectively span the whole of $\mathbb{R}^{pq}$.

\textbf{({\romannumeral 4}).} The same argument as which yielded \ref{eq:lr-con1}
also gives
\begin{equation}
2[\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n\mid\theta,\gamma})-\mathcal{\ell}_{n}^{\ast}(0)]\ensuremath{\rightsquigarrow} S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Pi_{\perp}(\Pi_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Pi_{\perp})^{-1}\Pi_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}+S_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{{\scriptscriptstyle \textnormal{ST}}}\label{eq:lr-con2}
\end{equation}
so that subtracting \ref{eq:lr-con2} from \ref{eq:lr-con1}, and recalling
$\Pi_{\perp}=\Theta_{\perp}\Xi$, yields
\begin{align*}
2[\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n\mid\theta})-\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n\mid\theta,\gamma})] & \ensuremath{\rightsquigarrow} S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}-S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Pi_{\perp}(\Pi_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Pi_{\perp})^{-1}\Pi_{\perp}^{\mathsf{T}})S_{{\scriptscriptstyle \textnormal{LU}}}\\
 & =(\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}})^{\mathsf{T}}[H_{\Theta,\perp}^{-1}-\Xi(\Xi^{\mathsf{T}}H_{\Theta,\perp}\Xi)^{-1}\Xi^{\mathsf{T}}](\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}})\\
 & =(H_{\Theta,\perp}^{-1/2}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}})^{\mathsf{T}}[I_{qr}-H_{\Theta,\perp}^{1/2}\Xi(\Xi^{\mathsf{T}}H_{\Theta,\perp}\Xi)^{-1}\Xi^{\mathsf{T}}H_{\Theta,\perp}^{1/2}](H_{\Theta,\perp}^{-1/2}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}).
\end{align*}
\end{proof}

\begin{proof}[Proof of \ref{prop:deltamethod}]
 Recall the definitions of $\largedec R_{n}=[\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}},\largedec R_{n,{\scriptscriptstyle \textnormal{ST}}}]$ and
$\largedec L_{n}=[\largedec L_{n,{\scriptscriptstyle \textnormal{LU}}},\largedec L_{n,{\scriptscriptstyle \textnormal{ST}}}]$ given at the beginning of this appendix.
Since $I_{kp}=\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\largedec L_{n,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}+\largedec R_{n,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{n,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}$,
we may write
\[
\tilde{\boldsymbol{\Phi}}_{n}=\boldsymbol{\Phi}_{n}+[(\tilde{\boldsymbol{\Phi}}_{n}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}]\largedec L_{n,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}+[(\tilde{\boldsymbol{\Phi}}_{n}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{ST}}}]\largedec L_{n,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\eqqcolon\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{LU}}}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}}.
\]
Since $\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{LU}}}=o_{p}(1)$ and $\boldsymbol{\Phi}_{n}\ensuremath{\rightarrow}\boldsymbol{\Phi}_{0}$,
we have $\smlabs{\lambda_{q+1}(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}})}<\smlabs{\lambda_{q}(\boldsymbol{\Phi}_{n})}$
w.p.a.1, and so by \ref{lem:derivatives}\ref{enu:deriv:zero}
\begin{align}
A(\tilde{\boldsymbol{\Phi}}_{n})-A(\boldsymbol{\Phi}_{n}) & =A(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{LU}}})-A(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}})\label{eq:Adifference}
\end{align}
w.p.a.1. Since $A(\cdot)$ is smooth, a second-order Taylor series
expansion and \ref{lem:derivatives}\ref{enu:deriv:values} yield
\begin{align}
 & \operatorname{vec}\{A(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{LU}}})-A(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}})\}\nonumber \\
 & \qquad\qquad\qquad\qquad=[J_{A}(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}})+o_{p}(1)]\operatorname{vec}\{\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{LU}}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}})\}\nonumber \\
 & \qquad\qquad\qquad\qquad=[J_{A}(\boldsymbol{\Phi}_{0})+o_{p}(1)]\operatorname{vec}\{\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{LU}}}\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\}\label{eq:Aexp}
\end{align}
where the second equality holds w.p.a.1, and follows from the continuity
of $J_{A}$ (\ref{lem:derivatives}\ref{enu:deriv:cont}), $\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}}=\boldsymbol{\Phi}_{0}+o_{p}(1)$,
and $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}})=\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n})=\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}$
(w.p.a.1, as implied by \ref{lem:derivatives}\ref{enu:deriv:zero}).
Finally, since
\begin{equation}
\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{LU}}}\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}=[(\tilde{\boldsymbol{\Phi}}_{n}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}]\largedec L_{n,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}=(\tilde{\boldsymbol{\Phi}}_{n}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}},\label{eq:DeltaRlu}
\end{equation}
the first part of \ref{eq:expansion} follows from \ref{eq:Adifference}--\ref{eq:DeltaRlu}.
The proof of the second part is analogous.
\end{proof}

\section{Limiting experiments}

The assumptions \ref{ass:DGP} and \ref{ass:LOC} are maintained throughout
this appendix. Recall the re-parametrisation given in \ref{eq:reparm-thm}
above, which in view of \ref{eq:reparam} we can equivalently write
as\begin{subequations}\label{eq:reparm}
\begin{align}
\boldsymbol{\pi} & \coloneqq n\operatorname{vec}\begin{bmatrix}A[\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi)\largedec L_{n}^{\mathsf{T}}]-A(\boldsymbol{\Phi}_{n})\\
\Lambda_{{\scriptscriptstyle \textnormal{LU}}}[\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi)\largedec L_{n}^{\mathsf{T}}]-\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n})
\end{bmatrix}\label{eq:vall}\\
f & \coloneqq n^{1/2}\varphi_{{\scriptscriptstyle \textnormal{ST}}}.\label{eq:fST}
\end{align}
\end{subequations}Under \ref{ass:LOC}, $R_{n,{\scriptscriptstyle \textnormal{ST}}}$ and $\Lambda_{n,{\scriptscriptstyle \textnormal{ST}}}$
associated with $\{\boldsymbol{\Phi}_{n}\}\subset\set P$ are constant (see \ref{enu:LOC:stat}),
so $\largedec R_{n,{\scriptscriptstyle \textnormal{ST}}}=\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}$ for all $n\in\ensuremath{\mathbb{N}}$, so that in particular
$\varphi_{{\scriptscriptstyle \textnormal{ST}}}=\operatorname{vec}\{(\boldsymbol{\Phi}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{ST}}}\}$. Let $\psi_{n}(\varphi)$
denote the smooth mapping $\varphi\ensuremath{\mapsto}(\boldsymbol{\pi},f)$ implied by \ref{eq:reparm},
which has domain $\ensuremath{\mathcal{P}}_{n}$ (defined in \ref{eq:Psetn} above) and
$\psi_{n}(0)=0$ for all $n\in\ensuremath{\mathbb{N}}$.
\begin{lem}
\label{lem:invpsi}~
\begin{enumerate}
\item \label{enu:invpsi:forward}There exists an $n_{0}\in\ensuremath{\mathbb{N}}$ and
an open neighbourhood $N\subset\mathbb{R}^{kp^{2}}$ of the origin, such
that $\psi_{n}$ is a smooth diffeomorphism on $N$, for all $n\geq n_{0}$
\item \label{enu:invpsi:back}Let ${\cal K}\subset\mathbb{R}^{kp^{2}}$ be any
compact neighbourhood of zero. Then there exists an $n_{1}\geq n_{0}$
such that $\psi_{n}^{-1}$ is well defined (and smooth) on ${\cal K}$,
for all $n\geq n_{1}$. Moreover, for any $(\boldsymbol{\pi},f)\in{\cal K}$,
$\varphi_{n}\coloneqq\psi_{n}^{-1}(\boldsymbol{\pi},f)$ is such that $D_{n}\varphi_{n}=O(1)$.
\end{enumerate}
\end{lem}
Thus so long as we restrict attention to $\varphi\in N$, we may equivalently
parametrise the model in terms of $(\boldsymbol{\pi},f)$. For a given $(\boldsymbol{\pi},f)\in\mathbb{R}^{kp^{2}}$,
$\psi_{n}^{-1}$ is well-defined (and smooth) at $(\boldsymbol{\pi},f)$ for
all $n$ sufficiently large, in which case we shall define (with a
slight abuse of notation) $\mathcal{\ell}_{n}(\boldsymbol{\pi},f)\coloneqq\mathcal{\ell}_{n}(\varphi,\Sigma)$,
where $\varphi=\psi_{n}^{-1}(\boldsymbol{\pi},f)$; and set $\mathcal{\ell}_{n}(\boldsymbol{\pi},f)\coloneqq-\infty$
otherwise (to simplify arguments, we treat $\Sigma$ as known here.)
To state our next result, recall the definitions of $S_{\boldsymbol{\pi}}$ and
$H_{\boldsymbol{\pi}}$ given in \ref{eq:SHpi}.
\begin{lem}
\label{lem:limexp}Jointly over any finite collection of $(\boldsymbol{\pi},f)\in\mathbb{R}^{kp^{2}}$,
\[
\mathcal{\ell}_{n}(\boldsymbol{\pi},f)-\mathcal{\ell}_{n}(0,0)\ensuremath{\rightsquigarrow}[S_{\boldsymbol{\pi}}^{\mathsf{T}}\boldsymbol{\pi}-\tfrac{1}{2}\boldsymbol{\pi}^{\mathsf{T}}H_{\boldsymbol{\pi}}\boldsymbol{\pi}]+[S_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}f-\tfrac{1}{2}f^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{ST}}}f].
\]
\end{lem}
We next show that, up to the term depending on $f$, the preceding
is also the limit of the loglikelihood ratio process in a multivariate
predictive regression with a known covariance matrix; recall \ref{ass:PR}
given in \ref{subsec:likelihoodasymp}.
\begin{lem}
\label{lem:predreg}Suppose that $\{y_{{\scriptscriptstyle \textnormal{PR}},t}\}$ and $\{z_{{\scriptscriptstyle \textnormal{PR}} t}\}$
are generated under \ref{ass:PR}, and that $\xi_{t}=[\begin{smallmatrix}\xi_{yt}\\
\xi_{zt}
\end{smallmatrix}]\ensuremath{\sim_{\ensuremath{\textnormal{i.i.d.}}}} N[0,\Omega]$ with $\Omega=K\Sigma K^{\mathsf{T}}$. Then for
\[
\boldsymbol{\pi}=n\operatorname{vec}\begin{bmatrix}A-A(\boldsymbol{\Phi}_{n})\\
\Lambda-\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n})
\end{bmatrix},
\]
we have, jointly over any finite collection of $\boldsymbol{\pi}\in\mathbb{R}^{pq}$
\[
\mathcal{\ell}_{n,{\scriptscriptstyle \textnormal{PR}}}(\boldsymbol{\pi})-\mathcal{\ell}_{n,{\scriptscriptstyle \textnormal{PR}}}(0)\ensuremath{\rightsquigarrow} S_{\boldsymbol{\pi}}^{\mathsf{T}}\boldsymbol{\pi}-\tfrac{1}{2}\boldsymbol{\pi}^{\mathsf{T}}H_{\boldsymbol{\pi}}\boldsymbol{\pi},
\]
where $\mathcal{\ell}_{n,{\scriptscriptstyle \textnormal{PR}}}(\boldsymbol{\pi})$ is the loglikelihood defined in \ref{thm:emw}.
\end{lem}
Finally, we show that when (the entirety of) $\boldsymbol{\Phi}$ is in unknown,
and the model is estimated subject to the constraint \ref{eq:vall},
then the limit of the \emph{concentrated} loglikelihood ratio process
is asymptotically identical to that of the predictive regression,
up to (random) terms that do not depend on $\boldsymbol{\pi}$. Let $\hat{f}_{n\mid\boldsymbol{\pi}}\coloneqq n^{1/2}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\boldsymbol{\pi}}$,
where $\hat{\varphi}_{n\mid\boldsymbol{\pi}}$ denotes the maximiser of $\mathcal{\ell}^{\ast}_{n}(\varphi)$
subject to $\varphi$ satisfying \ref{eq:vall}.
\begin{lem}
\label{lem:like-cons}Jointly over every finite collection of $\boldsymbol{\pi}\in\mathbb{R}^{pq}$,
\[
\mathcal{\ell}_{n}(\boldsymbol{\pi},\hat{f}_{n\mid\boldsymbol{\pi}})-\mathcal{\ell}_{n}(0)\ensuremath{\rightsquigarrow} S_{\boldsymbol{\pi}}^{\mathsf{T}}\boldsymbol{\pi}-\tfrac{1}{2}\boldsymbol{\pi}^{\mathsf{T}}H_{\boldsymbol{\pi}}\boldsymbol{\pi}+S_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{{\scriptscriptstyle \textnormal{ST}}}
\]
\end{lem}
\begin{proof}[Proof of \ref{lem:invpsi}]
 Consider the mapping $\Psi$ and the permutation matrix $M\in\mathbb{R}^{pq\times pq}$
such that
\begin{align}
\Psi(\boldsymbol{\Phi}) & \coloneqq\begin{bmatrix}\operatorname{vec} A(\boldsymbol{\Phi})\\
\operatorname{vec}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})\\
\operatorname{vec}\boldsymbol{\Phi}\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix} & {\cal M}\Psi(\boldsymbol{\Phi})\coloneqq\begin{bmatrix}M & 0\\
0 & I_{\#{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\Psi(\boldsymbol{\Phi}) & =\begin{bmatrix}\operatorname{vec}\begin{bmatrix}A(\boldsymbol{\Phi})\\
\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})
\end{bmatrix}\\
\operatorname{vec}\boldsymbol{\Phi}\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\label{eq:Psi-and-M}
\end{align}
By Lemmas~\ref{lem:implicit-maps} and \ref{lem:derivatives}, $\Psi$
is smooth and at $\boldsymbol{\Phi}=\boldsymbol{\Phi}_{0}$ has first differential
\begin{equation}
\ensuremath{\mathrm{d}}\Psi=\begin{bmatrix}J_{A}(\boldsymbol{\Phi}_{0}) & 0\\
J_{\Lambda}(\boldsymbol{\Phi}_{0}) & 0\\
0 & I_{\#{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\left(\begin{bmatrix}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\\
\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}
\end{bmatrix}\otimes I_{p}\right)\operatorname{vec}(\ensuremath{\mathrm{d}}\boldsymbol{\Phi}).\label{eq:dPsi}
\end{equation}
The Jacobian on the r.h.s.\ is invertible by \ref{lem:GLR}\ref{enu:GLR:L}
and \ref{lem:derivatunity} (for the latter, since $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{0})=I_{q}$).
Thus by the inverse mapping theorem, there is an open neighbourhood
$N_{\set P}\subset\set P$ of $\boldsymbol{\Phi}_{0}$ on which $\Psi$ has a smooth
inverse.

Now let $\tau_{n}(\varphi)\coloneqq\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi)\largedec L_{n}^{\mathsf{T}}$,
which converges (uniformly on compacta) to a linear and invertible
mapping $\tau_{0}(\varphi)$ for which $\tau_{0}(0)=\boldsymbol{\Phi}_{0}$. Hence there
exists an $n_{0}\in\ensuremath{\mathbb{N}}$ and a (fixed) open neighbourhood $N\subset\ensuremath{\mathcal{P}}_{n}$
of zero such that $\tau_{n}(N)\subset N_{\set P}$, for all $n\geq n_{0}$,
with $\tau_{n}$ being invertible on $N$. By composition, the sequence
of maps defined by
\begin{equation}
D_{n}^{-1}\psi_{n}(\varphi)={\cal M}\{\Psi[\tau_{n}(\varphi)]-\Psi(\boldsymbol{\Phi}_{n})\}\label{eq:almost-psi}
\end{equation}
is smooth and invertible on $N$, for all $n\geq0$, and has a smooth
inverse there; hence part~\ref{enu:invpsi:forward} holds. Finally,
since the image of $N$ under the r.h.s.\ must itself be an open
neighbourhood of zero, and
\begin{equation}
\begin{bmatrix}\boldsymbol{\pi}\\
f
\end{bmatrix}=\psi_{n}(\varphi)=D_{n}{\cal M}\{\Psi[\tau_{n}(\varphi)]-\Psi(\boldsymbol{\Phi}_{n})\},\label{eq:repmap}
\end{equation}
we may deduce that for any compact neighbourhood ${\cal K}$ of zero,
there is an $n_{1}\geq n_{0}$ such that the inverse $\psi_{n}^{-1}$
is well-defined and smooth for all $(\boldsymbol{\pi},f)\in{\cal K}$, for all
$n\geq n_{1}$. Finally, to show that the $\varphi_{n}\coloneqq\psi_{n}^{-1}(\boldsymbol{\pi},f)$
has $D_{n}\varphi_{n}=O(1)$, we note that since the r.h.s.\ of \ref{eq:almost-psi}
is (locally to zero) a diffeomorphism, which itself equals zero at
$\varphi=0$, the fact that
\[
{\cal M}\{\Psi[\tau_{n}(\varphi_{n})]-\Psi(\boldsymbol{\Phi}_{n})\}=D_{n}^{-1}\begin{bmatrix}\boldsymbol{\pi}\\
f
\end{bmatrix}\ensuremath{\rightarrow}0
\]
must imply that $\varphi_{n}\ensuremath{\rightarrow}0$. Hence follows from \ref{eq:dPsi}
and \ref{eq:repmap} that, by a Taylor expansion of \ref{eq:repmap}
around $\varphi=0$,
\[
\begin{bmatrix}MJ+o_{p}(1) & 0\\
0 & I_{\#{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\begin{bmatrix}\varphi_{n,{\scriptscriptstyle \textnormal{LU}}}\\
\varphi_{n,{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}=D_{n}^{-1}\begin{bmatrix}\boldsymbol{\pi}\\
f
\end{bmatrix}
\]
whence $D_{n}\varphi_{n}=O(1)$ as claimed. Thus part~\ref{enu:invpsi:back}
holds.
\end{proof}
\begin{proof}[Proof of \ref{lem:limexp}]
In view of \ref{lem:invpsi}, we may take $n$ sufficiently large
such that $\psi_{n}^{-1}$ is well defined at $(\boldsymbol{\pi},f)$. Let $\varphi_{n}^{\ast}\coloneqq\psi_{n}^{-1}(\boldsymbol{\pi},f)=o(1)$,
$\boldsymbol{\Phi}_{n}^{\ast}\coloneqq\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi_{n}^{\ast})\largedec L_{n}^{\mathsf{T}}$,
and $M\in\mathbb{R}^{pq\times pq}$ be as in \ref{eq:Psi-and-M}. Then
by \ref{prop:deltamethod}, for $J\coloneqq J(\boldsymbol{\Phi}_{0})$
\begin{equation}
\boldsymbol{\pi}=n\operatorname{vec}\begin{bmatrix}A(\boldsymbol{\Phi}_{n}^{\ast})-A(\boldsymbol{\Phi}_{n})\\
\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n}^{\ast})-\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n})
\end{bmatrix}=[MJ+o(1)]n\varphi_{n,{\scriptscriptstyle \textnormal{LU}}}^{\ast}\label{eq:pi-deriv}
\end{equation}
where by \ref{lem:derivatunity} and the definition of $M$,
\[
MJ=M\begin{bmatrix}I_{q}\otimes{\cal J}\\
I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}
\end{bmatrix}=I_{q}\otimes\begin{bmatrix}{\cal J}\\
L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}
\end{bmatrix}\eqqcolon I_{q}\otimes K
\]
for ${\cal J}$ and $K$ as defined in \ref{eq:Kdef}. Noting also
that $n^{1/2}\varphi_{n,{\scriptscriptstyle \textnormal{ST}}}^{\ast}=f$, it follows from \ref{lem:lhoodexp}
and \ref{eq:pi-deriv} that
\begin{align*}
\mathcal{\ell}_{n}(\boldsymbol{\pi},f)-\mathcal{\ell}_{n}(0,0) & =\mathcal{\ell}_{n}(\varphi_{n,{\scriptscriptstyle \textnormal{LU}}}^{\ast},\varphi_{n,{\scriptscriptstyle \textnormal{ST}}}^{\ast})-\mathcal{\ell}_{n}(0,0)\\
 & =S_{n}^{\mathsf{T}}(D_{n}\varphi_{n}^{\ast})-\tfrac{1}{2}(D_{n}\varphi_{n}^{\ast})^{\mathsf{T}}H_{n}(D_{n}\varphi_{n}^{\ast})\\
 & \ensuremath{\rightsquigarrow}[S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}(MJ)^{-1}\boldsymbol{\pi}-\tfrac{1}{2}\boldsymbol{\pi}^{\mathsf{T}}[(MJ)^{-1}]^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}(MJ)^{-1}\boldsymbol{\pi}]+[S_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}f-\tfrac{1}{2}f^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{ST}}}f].
\end{align*}
To complete the proof, we note that
\begin{align*}
[(MJ)^{-1}]^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}} & =(I_{q}\otimes K^{-1})^{\mathsf{T}}\int_{0}^{1}[\bar{Z}_{C}(r)\otimes\Sigma^{-1}\ensuremath{\mathrm{d}} E(r)]\\
 & =\int_{0}^{1}[\bar{Z}_{C}(r)\otimes(K\Sigma K^{\mathsf{T}})^{-1/2}\ensuremath{\mathrm{d}} W(r)]=S_{\boldsymbol{\pi}}
\end{align*}
where we have used that $E(s)=\Sigma^{-1/2}W(s)$ and,
\begin{align*}
[(MJ)^{-1}]^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}(MJ)^{-1} & =(I_{q}\otimes K^{-1})^{\mathsf{T}}\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\otimes\Sigma^{-1}\right)(I_{q}\otimes K^{-1})\\
 & =\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\otimes(K\Sigma K^{\mathsf{T}})^{-1}=H_{\boldsymbol{\pi}}.\qedhere
\end{align*}
\end{proof}
\begin{proof}[Proof of \ref{lem:predreg}]
 Letting $\boldsymbol{\Pi}=[\begin{smallmatrix}A_{{\scriptscriptstyle \textnormal{PR}}}\\
\Lambda_{{\scriptscriptstyle \textnormal{PR}}}
\end{smallmatrix}]$ and noting that $\boldsymbol{\pi}=n\operatorname{vec}(\boldsymbol{\Pi}-[\begin{smallmatrix}A(\boldsymbol{\Phi}_{n})\\
\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n})
\end{smallmatrix}])$, we have
\[
\mathcal{\ell}_{n}^{{\scriptscriptstyle \textnormal{PR}}}(\boldsymbol{\pi})=K_{n}-\frac{1}{2}\sum_{t=1}^{n}\smlnorm{x_{t}-\boldsymbol{\Pi} z_{t-1}}_{\Omega^{-1}},
\]
where $K_{n}\coloneqq-\frac{n}{2}\log(2\pi\log\det\Omega)$. It then
follows by exactly the same arguments as were used in the proof of
\ref{lem:lhoodexp} that
\[
\mathcal{\ell}_{n,{\scriptscriptstyle \textnormal{PR}}}(\boldsymbol{\pi})-\mathcal{\ell}_{n,{\scriptscriptstyle \textnormal{PR}}}(0)=S_{n,{\scriptscriptstyle \textnormal{PR}}}^{\mathsf{T}}\boldsymbol{\pi}-\tfrac{1}{2}\boldsymbol{\pi}^{\mathsf{T}}H_{n,{\scriptscriptstyle \textnormal{PR}}}\boldsymbol{\pi}
\]
where
\begin{align*}
S_{n,{\scriptscriptstyle \textnormal{PR}}} & =\frac{1}{n}\sum_{t=1}^{n}(\bar{z}_{t-1}\otimes\Omega^{-1/2}\eta_{t}) & H_{n,{\scriptscriptstyle \textnormal{PR}}} & =\frac{1}{n}\sum_{t=1}^{n}(\bar{z}_{t-1}\bar{z}_{t-1}^{\mathsf{T}}\otimes\Omega^{-1}).
\end{align*}
Under \ref{ass:PR}, it follows by \ref{lem:wkconv}\ref{enu:wkconv:Zc}
that $n^{-1/2}\bar{z}_{\smlfloor{nr}}\ensuremath{\rightsquigarrow}\bar{Z}_{C,{\scriptscriptstyle \textnormal{PR}}}(r)$ on $D[0,1]$,
where $\bar{Z}_{C,{\scriptscriptstyle \textnormal{PR}}}(r)$ denotes the residual from the projection
of
\begin{equation}
Z_{C,{\scriptscriptstyle \textnormal{PR}}}(r)\coloneqq\int_{0}^{r}\mathrm{e}^{C(r-s)}\Omega_{zz}^{1/2}\ensuremath{\mathrm{d}} W(s)\label{eq:Zpr}
\end{equation}
on a constant and a linear trend, and we have partitioned $\Omega=[\begin{smallmatrix}\Omega_{yy} & \Omega_{yz}\\
\Omega_{zy} & \Omega_{zz}
\end{smallmatrix}]$ conformably with $\xi_{t}=[\begin{smallmatrix}\xi_{yt}\\
\xi_{zt}
\end{smallmatrix}]$. Then by the continuous mapping theorem and the same arguments as
used in the proof of \ref{lem:wkconv}\ref{enu:wkconv:si},
\begin{align}
S_{n,{\scriptscriptstyle \textnormal{PR}}} & \ensuremath{\rightsquigarrow}\int_{0}^{1}[\bar{Z}_{C,{\scriptscriptstyle \textnormal{PR}}}(r)\otimes\Omega^{-1/2}W(r)]\ensuremath{\,\ensuremath{\mathrm{d}}} r & H_{n,{\scriptscriptstyle \textnormal{PR}}} & \ensuremath{\rightsquigarrow}\int\bar{Z}_{C,{\scriptscriptstyle \textnormal{PR}}}\bar{Z}_{C,{\scriptscriptstyle \textnormal{PR}}}^{\mathsf{T}}\otimes\Omega^{-1}.\label{eq:SHpr}
\end{align}
Thus we can bring \ref{eq:Zpr} into agreement with \ref{eq:Zproc},
and the limits on the r.h.s.\ of \ref{eq:SHpr} with \ref{eq:SHpi},
by setting
\[
\Omega=\begin{bmatrix}\Omega_{yy} & \Omega_{yz}\\
\Omega_{zy} & \Omega_{zz}
\end{bmatrix}=\begin{bmatrix}{\cal J}\Sigma{\cal J}^{\mathsf{T}} & {\cal J}\Sigma L_{{\scriptscriptstyle \textnormal{LU}}}\\
L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Sigma{\cal J}^{\mathsf{T}} & L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Sigma L_{{\scriptscriptstyle \textnormal{LU}}}
\end{bmatrix}=K\Sigma K^{\mathsf{T}}.\qedhere
\]
\end{proof}
\begin{proof}[Proof of \ref{lem:like-cons}]
 We first show that $D_{n}\hat{\varphi}_{n\mid\boldsymbol{\pi}}=O_{p}(1)$. By \ref{lem:invpsi},
for all $n$ sufficiently large, there exists a (deterministic) sequence
$\varphi_{n\mid\boldsymbol{\pi}}\in\ensuremath{\mathcal{P}}_{n}$ with $D_{n}\varphi_{n\mid\boldsymbol{\pi}}=O(1)$,
such that \ref{eq:vall} holds at $\varphi=\varphi_{n\mid\boldsymbol{\pi}}$. It follows
from \ref{lem:lhoodexp} that for each $\epsilon>0$, there exists
an $N<\infty$ such that
\[
\limsup_{n\ensuremath{\rightarrow}\infty}\ensuremath{\mathbb{P}}\{\mathcal{\ell}_{n}^{\ast}(\varphi_{n\mid\boldsymbol{\pi}})-\mathcal{\ell}_{n}^{\ast}(0)<-N\}<\epsilon/2.
\]
On the other hand, adapting the argument given in the proof of \ref{lem:consistency},
we may also choose $M<\infty$ sufficiently large such that
\begin{align*}
\ensuremath{\mathbb{P}}\left\{ \sup_{\{\varphi\in\mathcal{P}_{n}\mid\smlnorm{D_{n}\varphi}\geq M\}}[\mathcal{\ell}^{\ast}_{n}(\varphi)-\mathcal{\ell}^{\ast}_{n}(0)]<-2N\right\}  & \ge\ensuremath{\mathbb{P}}\left\{ M[\smlnorm{S_{n}}-\tfrac{1}{2}\lambda_{\min}(H_{n})M]\leq-2N\right\} \\
 & >1-\epsilon/2
\end{align*}
for all $n$ sufficiently large. Deduce that with probability at least
$1-\epsilon$, $\mathcal{\ell}_{n}^{\ast}(\varphi_{n\mid\boldsymbol{\pi}})$ must strictly
exceed $\mathcal{\ell}_{n}(\varphi)$ over all $\varphi\in\ensuremath{\mathcal{P}}_{n}$ with $\smlnorm{D_{n}\varphi}\geq M$;
it follows that the constrained maximiser $\hat{\varphi}_{n\mid\boldsymbol{\pi}}$
must have $\smlnorm{D_{n}\hat{\varphi}_{n\mid\boldsymbol{\pi}}}<M$. Deduce that $D_{n}\hat{\varphi}_{n\mid\boldsymbol{\pi}}=O_{p}(1)$
as claimed.

Now it follows from \ref{eq:dPsi} and \ref{eq:repmap} that, at $\varphi=\hat{\varphi}_{n\mid\boldsymbol{\pi}}$,
\[
\begin{bmatrix}\ensuremath{\mathrm{d}}\boldsymbol{\pi}\\
\ensuremath{\mathrm{d}} f
\end{bmatrix}=\begin{bmatrix}MJ+o_{p}(1) & 0\\
0 & I_{\#{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}D_{n}\ensuremath{\mathrm{d}}\varphi
\]
and hence
\[
D_{n}\ensuremath{\mathrm{d}}\varphi=\begin{bmatrix}(MJ)^{-1}+o_{p}(1) & 0\\
0 & I_{\#{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\begin{bmatrix}\ensuremath{\mathrm{d}}\boldsymbol{\pi}\\
\ensuremath{\mathrm{d}} f
\end{bmatrix}
\]
at $(\boldsymbol{\pi},\hat{f}_{n\mid\boldsymbol{\pi}})$. Thus
\begin{equation}
n\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}}=[(MJ)^{-1}+o_{p}(1)]\boldsymbol{\pi},\label{eq:consLUlim}
\end{equation}
and since $\hat{f}_{n\mid\boldsymbol{\pi}}$ must satisfy the first-order conditions
for a maximum, we have from \ref{lem:lhoodexp} that
\begin{align*}
\nabla_{f}\mathcal{\ell}_{n}(\boldsymbol{\pi},\hat{f}_{n\mid\boldsymbol{\pi}}) & =\nabla_{f}[S_{n}^{\mathsf{T}}(D_{n}\varphi)-\tfrac{1}{2}(D_{n}\varphi)^{\mathsf{T}}H_{n}(D_{n}\varphi)]_{f=\hat{f}_{n\mid\boldsymbol{\pi}}}\\
 & =S_{n,{\scriptscriptstyle \textnormal{ST}}}-(n\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\boldsymbol{\pi}})^{\mathsf{T}}H_{n,{\scriptscriptstyle \textnormal{LS}}}-H_{n,{\scriptscriptstyle \textnormal{ST}}}n^{1/2}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\boldsymbol{\pi}}
\end{align*}
whence
\begin{equation}
n^{1/2}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\boldsymbol{\pi}}=H_{n,{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{n,{\scriptscriptstyle \textnormal{ST}}}+o_{p}(1)\ensuremath{\rightsquigarrow} H_{{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{{\scriptscriptstyle \textnormal{ST}}}.\label{eq:consSTlim}
\end{equation}
Thus, in view of \ref{eq:consLUlim} and \ref{eq:consSTlim}, the weak
limit of
\begin{align*}
\mathcal{\ell}_{n}(\boldsymbol{\pi},\hat{f}_{n\mid\boldsymbol{\pi}})-\mathcal{\ell}_{n}(0) & =\mathcal{\ell}_{n}^{\ast}(\hat{\varphi}_{n\mid\boldsymbol{\pi}})-\mathcal{\ell}_{n}^{\ast}(0)=S_{n}^{\mathsf{T}}(D_{n}\hat{\varphi}_{n\mid\pi})-\tfrac{1}{2}(D_{n}\hat{\varphi}_{n\mid\pi})^{\mathsf{T}}H_{n}(D_{n}\hat{\varphi}_{n\mid\pi})
\end{align*}
is as claimed.
\end{proof}

\section{Proofs of theorems}

\label{app:theoremproofs}
\begin{proof}[Proof of \ref{thm:emw}]
 This follows directly from Lemmas~\ref{lem:limexp}--\ref{lem:like-cons},
noting in particular that $\mathcal{\ell}^{\ast}_{n}(\boldsymbol{\pi})=\mathcal{\ell}_{n}(\boldsymbol{\pi},\hat{f}_{n\mid\boldsymbol{\pi}})$,
where the latter is as appears in \ref{lem:like-cons}.
\end{proof}
\begin{proof}[Proof of \ref{thm:estimators}]
 \textbf{({\romannumeral 1}).} In the notation of \ref{app:asymptotics}, $\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}}=\operatorname{vec}\{(\hat{\boldsymbol{\Phi}}_{n}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\}$.
By \ref{prop:andrews}\ref{enu:andrews:unres}
\begin{align*}
n\operatorname{vec}\{(\hat{\boldsymbol{\Phi}}_{n}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\} & \ensuremath{\rightsquigarrow}\left[\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\right)^{-1}\otimes I_{p}\right]\int_{0}^{1}[\bar{Z}_{C}(r)\otimes\ensuremath{\mathrm{d}} E(r)]\\
 & =\operatorname{vec}\left\{ \int(\ensuremath{\mathrm{d}} E)\bar{Z}_{C}^{\mathsf{T}}\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\right)^{-1}\right\} ,
\end{align*}
and so by \ref{prop:deltamethod}
\begin{equation}
\begin{bmatrix}\operatorname{vec}\{A(\hat{\boldsymbol{\Phi}}_{n})-A(\boldsymbol{\Phi}_{n})\}\\
\operatorname{vec}\{\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\hat{\boldsymbol{\Phi}}_{n})-\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n})\}
\end{bmatrix}\ensuremath{\rightsquigarrow}\begin{bmatrix}J_{A}(\boldsymbol{\Phi}_{0})\\
J_{\Lambda}(\boldsymbol{\Phi}_{0})
\end{bmatrix}\operatorname{vec}\left\{ \int(\ensuremath{\mathrm{d}} E)\bar{Z}_{C}^{\mathsf{T}}\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\right)^{-1}\right\} .\label{eq:limdeltameth}
\end{equation}
Since $\boldsymbol{\Phi}_{n}\ensuremath{\rightarrow}\boldsymbol{\Phi}_{0}$ with $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{0})=I_{q}$
under \ref{ass:LOC}, we have by \ref{lem:derivatunity} that
\begin{equation}
\begin{bmatrix}J_{A}(\boldsymbol{\Phi}_{0})\\
J_{\Lambda}(\boldsymbol{\Phi}_{0})
\end{bmatrix}=\begin{bmatrix}I_{q}\otimes\beta^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{ST}}}(I_{kp-q}-\Lambda_{{\scriptscriptstyle \textnormal{ST}}})^{-1}L_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\\
I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}
\end{bmatrix}.\label{eq:limjacob}
\end{equation}
The result then follows from \ref{eq:limdeltameth} and \ref{eq:limjacob},
by reversing the vectorisation.

\textbf{({\romannumeral 2}).} In the notation of \ref{app:asymptotics}, maximising
$\mathcal{\ell}_{n}^{\ast}(\boldsymbol{\Phi})$ subject to $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}}=I_{q}+C/n$
corresponds to maximising $\mathcal{\ell}_{n}(\varphi)$ subject to $\theta_{n}(\varphi)=0$.
Thus $\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}=\operatorname{vec}\{(\hat{\boldsymbol{\Phi}}_{n\mid\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}}}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\}$,
and so by \ref{prop:andrews}\ref{enu:andrews:res}
\[
n\operatorname{vec}\{(\hat{\boldsymbol{\Phi}}_{n\mid\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}}}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\}\ensuremath{\rightsquigarrow}\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}
\]
where $\Theta_{\perp}=I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}},\perp}$. Hence by \ref{prop:deltamethod},
\[
\operatorname{vec}\{A(\hat{\boldsymbol{\Phi}}_{n\mid\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}}})-A(\boldsymbol{\Phi}_{n})\}\ensuremath{\rightsquigarrow} J_{A}(\boldsymbol{\Phi}_{0})\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}.
\]

To determine the distribution of the r.h.s., we note that
\begin{equation}
\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}=\int_{0}^{1}[\bar{Z}_{C}(r)\otimes L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\Sigma^{-1}\ensuremath{\mathrm{d}} E(r)]\eqqcolon\int_{0}^{1}[\bar{Z}_{C}(r)\otimes\ensuremath{\mathrm{d}} U(r)].\label{eq:ThetaSlu}
\end{equation}
Recall that $\bar{Z}_{C}$ is a function only of $Z_{C}$, which from
\ref{eq:Zproc} is given by
\begin{equation}
Z_{C}(r)=\int_{0}^{r}\mathrm{e}^{C(r-s)}L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\ensuremath{\mathrm{d}} E(s)\eqqcolon\int_{0}^{r}\mathrm{e}^{C(r-s)}\ensuremath{\mathrm{d}} V(s).\label{eq:Zprocagain}
\end{equation}
$(U,V)=(L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\Sigma^{-1}E,L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}E)$ is
a pair of vector Brownian motions, with covariance
\[
\ensuremath{\mathbb{E}} U(1)V(1)^{\mathsf{T}}=L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\Sigma^{-1}\ensuremath{\mathbb{E}}[E(1)E(1)^{\mathsf{T}}]L_{{\scriptscriptstyle \textnormal{LU}}}=L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}L_{{\scriptscriptstyle \textnormal{LU}}}=0;
\]
whence $U$ and $V$ are independent. In particular, we have from
\ref{eq:Zprocagain} that $U$ is independent of $\bar{Z}_{C}$. This,
combined with the fact that
\begin{align*}
J_{A}(\boldsymbol{\Phi}_{0})\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1} & =\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\right)^{-1}\otimes\mathcal{J}L_{{\scriptscriptstyle \textnormal{LU}},\perp}(L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\Sigma^{-1}L_{{\scriptscriptstyle \textnormal{LU}},\perp})^{-1}
\end{align*}
depends only on $\bar{Z}_{C}$, implies $J_{A}(\boldsymbol{\Phi}_{0})\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}$
is mixed normal with variance
\[
\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\right)^{-1}\otimes\mathcal{J}L_{{\scriptscriptstyle \textnormal{LU}},\perp}(L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\Sigma^{-1}L_{{\scriptscriptstyle \textnormal{LU}},\perp})^{-1}L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\mathcal{J}^{\mathsf{T}},
\]
which proves \ref{eq:Arstr}.

Finally, note that the preceding holds for any choice of $L_{{\scriptscriptstyle \textnormal{LU}},\perp}\in\mathbb{R}^{p\times r}$
having full column rank and $L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}L_{{\scriptscriptstyle \textnormal{LU}}}=0$. Let
$\alpha\coloneqq\Phi_{0}(1)\beta(\beta^{\mathsf{T}}\beta)^{-1}\in\mathbb{R}^{p\times r}$,
where $\Phi_{0}(1)\coloneqq\lim_{n\ensuremath{\rightarrow}\infty}\Phi_{n}(1)$; then
\[
L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\alpha=L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Phi_{0}(1)\beta(\beta^{\mathsf{T}}\beta)^{-1}=0
\]
by \ref{eq:eig-eig} with $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}=\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{0})=I_{q}$.
Further, $\operatorname{rk}\alpha=r$ since $\operatorname{sp}\Phi_{0}(1)=\operatorname{sp}\beta$, and
thus we may indeed choose $L_{{\scriptscriptstyle \textnormal{LU}},\perp}=\alpha$. In this case,
\begin{align*}
\mathcal{J}L_{{\scriptscriptstyle \textnormal{LU}},\perp} & =\beta^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{ST}}}(I_{kp-q}-\Lambda_{{\scriptscriptstyle \textnormal{ST}}})^{-1}L_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\Phi_{0}(1)\beta(\beta^{\mathsf{T}}\beta)^{-1}=_{(1)}\beta^{\mathsf{T}}\beta(\beta^{\mathsf{T}}\beta)^{-1}=I_{r},
\end{align*}
where $=_{(1)}$ follows from \ref{eq:betastuff} above. Thus \ref{eq:johvar}
is proved.
\end{proof}
\begin{proof}[Proof of \ref{thm:lrstats}]
 We first prove \ref{eq:multDF}. In the notation of \ref{app:asymptotics},
$\mathcal{LR}_{n}(\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}})=2[\mathcal{\ell}_{n}^{\ast}(\hat{\varphi}_{n})-\mathcal{\ell}_{n}^{\ast}(\hat{\varphi}_{n\mid\theta})]$.
By \ref{prop:andrews}\ref{enu:andrews:lrroot},
\[
\mathcal{LR}_{n}(\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}})\ensuremath{\rightsquigarrow} S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}\Theta(\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}\Theta)^{-1}\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}S_{{\scriptscriptstyle \textnormal{LU}}}\eqqcolon\mathcal{LR},
\]
where $\Theta=I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}}}$, $S_{{\scriptscriptstyle \textnormal{LU}}}=\int[\bar{Z}_{C}(r)\otimes\Sigma^{-1}\ensuremath{\mathrm{d}} E]$,
and $H_{{\scriptscriptstyle \textnormal{LU}}}=\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\otimes\Sigma^{-1}$.
To obtain the claimed expression for $\mathcal{LR}$, note that
\[
S_{{\scriptscriptstyle \textnormal{LU}}}=\int[\bar{Z}_{C}(r)\otimes\Sigma^{-1}\ensuremath{\mathrm{d}} E]=\operatorname{vec}\left\{ \Sigma^{-1}\int(\ensuremath{\mathrm{d}} E)\bar{Z}_{C}^{\mathsf{T}}\right\}
\]
and
\[
H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}\Theta(\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}\Theta)^{-1}\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}=\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\right)^{-1}\otimes\Sigma L_{{\scriptscriptstyle \textnormal{LU}}}(L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Sigma L_{{\scriptscriptstyle \textnormal{LU}}})^{-1}L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Sigma
\]
whence, using $\operatorname{vec}(A)^{\mathsf{T}}\operatorname{vec}(B)=\operatorname{tr}(A^{\mathsf{T}}B)$,
\begin{align}
\mathcal{LR} & =\operatorname{tr}\left\{ \Delta^{-1/2}L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\int(\ensuremath{\mathrm{d}} E)\bar{Z}_{C}^{\mathsf{T}}\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\right)^{-1}\int\bar{Z}_{C}(\ensuremath{\mathrm{d}} E)^{\mathsf{T}}L_{{\scriptscriptstyle \textnormal{LU}}}\Delta^{-1/2}\right\} \label{eq:LRmed}
\end{align}
where $\Delta\coloneqq L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Sigma L_{{\scriptscriptstyle \textnormal{LU}}}$. To simplify
this further, note that $L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}E$ is a $q$-dimensional
Brownian motion with variance $\Delta$, and so for $W_{\ast}(r)\coloneqq\Delta^{-1/2}L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}E(r)\sim\mathrm{BM}(I_{q})$,
we have
\begin{multline*}
Z_{C}(r)=\int_{0}^{r}\mathrm{e}^{C(r-s)}L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\ensuremath{\mathrm{d}} E(s)=\int_{0}^{r}\mathrm{e}^{C(r-s)}\Delta^{1/2}\ensuremath{\mathrm{d}} W_{\ast}(s)\\
=_{(1)}\Delta^{1/2}\int_{0}^{r}\mathrm{e}^{C_{\ast}(r-s)}\ensuremath{\mathrm{d}} W_{\ast}(s)\eqqcolon\Delta^{1/2}Z_{C_{\ast}}(r)
\end{multline*}
where $C_{\ast}\coloneqq\Delta^{-1/2}C\Delta^{1/2}$ is as in the statement
of the theorem, and $=_{(1)}$ follows from $\mathrm{e}^{C}D=D\mathrm{e}^{D^{-1}CD}$
for any nonsingular $D$. Hence $\bar{Z}_{C}(r)=\Delta^{1/2}\bar{Z}_{C_{\ast}}(r)$,
whereupon \ref{eq:multDF} follows from \ref{eq:LRmed} and the definition
of $W_{\ast}$.

We next prove \ref{eq:chisqlim}. Maximisation of $\mathcal{\ell}_{n}^{\ast}(\boldsymbol{\Phi})$
subject to $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=I_{q}+C/n$ and $a_{ij}(\boldsymbol{\Phi})=a_{0}$
corresponds, in the notation of \ref{app:asymptotics}, to maximisation
of $\mathcal{\ell}_{n}(\varphi)$ subject to $\theta_{n}(\varphi)=0$ and $\gamma_{n}(\varphi)=0$.
Therefore by \ref{prop:andrews}\ref{enu:andrews:coef},
\begin{align*}
\mathcal{LR}_{n}[a_{ij}(\boldsymbol{\Phi}_{n});\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}}] & =2[\mathcal{\ell}_{n}(\hat{\varphi}_{n\mid\theta})-\mathcal{\ell}_{n}(\hat{\varphi}_{n\mid\theta,\gamma})]\ensuremath{\rightsquigarrow}(H_{\Theta,\perp}^{-1/2}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}})^{\mathsf{T}}[I_{qr}-\mathcal{Q}](H_{\Theta,\perp}^{-1/2}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}).
\end{align*}
Recall from \ref{eq:ThetaSlu} and the subsequent arguments that
\[
\operatorname{vec}\{\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}\}=_{d}\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\otimes L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\Sigma^{-1}L_{{\scriptscriptstyle \textnormal{LU}},\perp}\right)^{1/2}\eta
\]
for $\eta\sim\mathrm{N}[0,I_{qr}]$ independent of $\bar{Z}_{C}$, and
therefore also of
\[
H_{\Theta,\perp}=\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp}=\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\otimes L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\Sigma^{-1}L_{{\scriptscriptstyle \textnormal{LU}},\perp}.
\]
Thus $\operatorname{vec}\{H_{\Theta,\perp}^{-1/2}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}\}\sim\mathrm{N}[0,I_{qr}]$
is independent of $H_{{\scriptscriptstyle \textnormal{LU}}}$, and therefore also of $\mathcal{Q}$.
The result follows by noting that $H_{\Theta,\perp}^{1/2}\Xi$ has
rank $qr-1$ a.s., whence $I_{qr}-\mathcal{Q}$ projects orthogonally
onto a subspace of dimension 1, a.s.
\end{proof}