Extracted main text — title through conclusion, appendix excluded. This is what our citation measures are computed over, published so the extraction can be checked by eye.
Rendered from LaTeX for readability, not typeset faithfully. Citation keys are highlighted; maths is left as source; figures, tables and equation environments are summarised rather than reproduced; unrecognised commands are greyed out so nothing is silently dropped. Email addresses are removed.
\RequirePackage{snapshot}
\geometry{verbose,tmargin=2.5cm,bmargin=2.5cm,lmargin=2.5cm,rmargin=2.5cm}
\pagestyle{fancy}
\onehalfspacing
\makeatletter
\AtBeginDocument
\AtBeginDocument
\AtBeginDocument
\AtBeginDocument
\AtBeginDocument
\AtBeginDocument
\AtBeginDocument
\AtBeginDocument
\AtBeginDocument
\AtBeginDocument
\AtBeginDocument
[(\Lambda_{{\scriptscriptstyle LU}}^{\mathsf{T}}\otimes I_{q})-(I_{q}\otimes\Lambda_{{\scriptscriptstyle LU}})](I_{q}\otimes G^{\mathsf{T}})B+(I_{q}\otimes L_{{\scriptscriptstyle LU}}^{\mathsf{T}})
\end{bmatrix}
\end{equation}
for $G^{\mathsf{T}}\coloneqq[0_{q\times r},I_{q}]$, $\beta^{\mathsf{T}}=[I_{r},-A]$,
and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}=\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$, etc.; and
• $J(\boldsymbol{\Phi})$ is continuous.
\end{enumerate}
\end{lem}
When $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=I_{q}$, the $pq\times pq$ matrix $J(\boldsymbol{\Phi})$
simplifies as follows.
lemSuppose $\boldsymbol{\Phi}\in\set P$ with $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=I_{q}$.
Then $J(\boldsymbol{\Phi})$ is nonsingular, and
\[
\begin{bmatrix}J_{A}(\boldsymbol{\Phi})\\
J_{\Lambda}(\boldsymbol{\Phi})
\end{bmatrix}=\begin{bmatrix}I_{q}\otimes\beta^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{ST}}}(I_{kp-q}-\Lambda_{{\scriptscriptstyle \textnormal{ST}}})^{-1}L_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\\
I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}
\end{bmatrix}.
\]
proof[Proof of (ref)]
We first prove $\set P$ is open. For $F\in\mathbb{R}^{kp\times kp}$,
let $\lambda_{i}(F)$ denote the $i$th eigenvalue of $F$, when these
are placed in descending order of modulus. Let $\set F$ denote the
set of $kp\times kp$ matrices such that
\begin{enumerate}
• $\smlabs{\lambda_{q+1}(F)}<\smlabs{\lambda_{q}(F)}$; and
\end{enumerate}
there exist $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}\in\mathbb{R}^{q\times q}$ and $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}\in\mathbb{R}^{kp\times q}$
such that
\begin{enumerate}[resume]
• the eigenvalues of $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}$ are $\{\lambda_{i}(F)\}_{i=1}^{q}$,
$F\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=\largedec R_{{\scriptscriptstyle \textnormal{LU}}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}$; and
• $\operatorname{rk}\{\largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}\}=q$, where $\largedec G^{\mathsf{T}}\coloneqq[0_{q\times(kp-q)},I_{q}]=[0_{q\times k(p-1)},G^{\mathsf{T}}]$.
\end{enumerate}
In view of (ref), $\boldsymbol{\Phi}\in\set P$ if and only if the companion
form matrix $F(\boldsymbol{\Phi})$ is in $\set F$. Since $F(\cdot)$ is trivially
continuous, it suffices to show that $\set F$ is open.
To that end, fix $F_{0}\in\set F$, and let $\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}$ and $\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}$
denote matrices satisfying (ii) and (iii) above. By the continuity
of eigenvalues and simple invariant subspaces (Theorems IV.1.1 and
V.2.8 in SS90), for every $\epsilon>0$ there exists a
$\delta>0$ such that whenever $\smlnorm{F-F_{0}}<\delta$, $F$ satisfies
requirements (i) and (ii) above, with associated $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}$ such
that $\smlnorm{\largedec R_{{\scriptscriptstyle \textnormal{LU}}}-\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}}<\epsilon$. Since the set of full
rank matrices is open, we may take $\epsilon>0$ sufficiently small
such that (iii) also holds. Thus $F\in\set F$, and so $F_{0}$ is
an interior point of $\set F$; deduce $\set F$ is open.
We turn next to the smoothness of $A(\boldsymbol{\Phi})$ and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$.
For $F_{0}\in\set F$ we have the invariant subspace decomposition
(as per (ref) above)
\begin{equation}
F_{0}=\largedec R_{0,{\scriptscriptstyle LU}}\Lambda_{0,{\scriptscriptstyle LU}}\largedec L_{0,{\scriptscriptstyle LU}}^{\mathsf{T}}+\largedec R_{0,{\scriptscriptstyle ST}}\Lambda_{0,{\scriptscriptstyle ST}}\largedec L_{0,{\scriptscriptstyle ST}}^{\mathsf{T}}
\end{equation}
where $\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}$ and $\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}$ satisfy (ii)--(iii) above.
Since (iii) holds, we may choose $\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}$ such that $\largedec G^{\mathsf{T}}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=I_{q}$;
note that $\largedec L_{0}^{\mathsf{T}}\largedec R_{0}=I_{kp}$ (as per (ref)(ref))
implies $\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=I_{q}$. Define the maps\begin{subequations}
\begin{align}
H(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}};F) & \coloneqq\begin{bmatrix}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}-F\largedec R_{{\scriptscriptstyle \textnormal{LU}}}; & \largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}-I_{q}\end{bmatrix}\\
H^{\ast}(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}};F) & \coloneqq\begin{bmatrix}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}-F\largedec R_{{\scriptscriptstyle \textnormal{LU}}}; & \largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}-I_{q}\end{bmatrix},
\end{align}
\end{subequations}so that $H(\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}},\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}};F_{0})=H^{\ast}(\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}},\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}};F_{0})=0$;
but note that these maps need not otherwise agree, since they impose
distinct normalisations on $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}$. Once we have shown that the
Jacobian of $H^{\ast}$ with respect to $(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}})$
is nonsingular at $(\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}},\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}};F_{0})$, it will follow
by the implicit mapping theorem (Lang93)
that there exists a neighbourhood $N\subset\set F$ of $F_{0}$ and
smooth functions $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}:N\ensuremath{\rightarrow}\mathbb{R}^{kp\times q}$, $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}:N\ensuremath{\rightarrow}\mathbb{R}^{q\times q}$
such that
\[
H^{\ast}[\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F),\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F);F]=0
\]
for all $F\in N$; by the continuity of $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\cdot)$,
we may choose $N$ such that $\operatorname{rk}\{\largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F)\}=q$
for all $F\in N$. Thus
\begin{align}
\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(F) & \coloneqq\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F)[\largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F)]^{-1}\\
\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(F) & \coloneqq[\largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F)]\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F)[\largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(F)]^{-1}
\end{align}
are well defined for all $F\in N$, and have the property that
\[
H[\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(F),\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(F);F]=0
\]
for all $F\in N$. Since the $(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}})$ satisfying
$H(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}};F)=0$ is unique, repeating this construction
for every $F_{0}\in\set F$ allows the smooth maps $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(F)$
and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(F)$ to be extended to the whole of $\set F$.
The smoothness of $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})\coloneqq\Lambda_{{\scriptscriptstyle \textnormal{LU}}}[F(\boldsymbol{\Phi})]$
and $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})\coloneqq\largedec R_{{\scriptscriptstyle \textnormal{LU}}}[F(\boldsymbol{\Phi})]$ follows immediately,
and that of $A(\boldsymbol{\Phi})$ by noting that it corresponds to rows $(k-1)p+1$
to $(k-1)p+r$ of $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$.
It thus remains to verify that the Jacobian of $H^{\ast}$ with respect
to $(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}})$ is nonsingular at $(\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}},\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}};F_{0})$.
Matrix differentiation gives
\[
\ensuremath{\mathrm{d}} H^{\ast}=\begin{bmatrix}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}(\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}})+(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}-F_{0}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}); & \largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})\end{bmatrix}\eqqcolon\begin{bmatrix}\ensuremath{\mathrm{d}} H_{1}^{\ast}; & \ensuremath{\mathrm{d}} H_{2}^{\ast}\end{bmatrix}
\]
The Jacobian is nonsingular if $\ensuremath{\mathrm{d}} H^{\ast}=0$ implies $\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=0$
and $\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}=0$. To that end, suppose $\ensuremath{\mathrm{d}} H^{\ast}=0$.
Then $0=\ensuremath{\mathrm{d}} H_{2}^{\ast}=\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})$,
and
\[
\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=(\largedec R_{0}\largedec L_{0}^{\mathsf{T}})\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=(\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}+\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=(\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}
\]
and similarly, by (ref) above,
\[
F_{0}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})=(\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}+\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}).
\]
Hence
\begin{align*}
\ensuremath{\mathrm{d}} H_{1}^{\ast} & =\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}(\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}})+\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}[\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}-\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})]\\
& =\begin{bmatrix}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}} & \largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\end{bmatrix}\begin{bmatrix}\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}\\
\mathcal{T}[\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})]
\end{bmatrix},
\end{align*}
where $\mathcal{T}(M)\coloneqq M\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}-\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}}M.$
Since $\largedec R_{0}$ is nonsingular, $\ensuremath{\mathrm{d}} H_{1}^{\ast}=0$ implies that
$\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}=0$ and $\mathcal{T}[\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})]=0$;
but since $\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}$ and $\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}}$ have no common
eigenvalues, $\mathcal{T}(M)=0$ if and only if $M=0$ (SS90,
Thm V.1.3). Thus $\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}})=0$, whence
\[
\begin{bmatrix}\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\\
\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}
\end{bmatrix}\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=0
\]
from which it follows that $\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}=0$, since $\largedec L_{0}$ is
nonsingular.
proof[Proof of (ref)]
({\romannumeral 1}). We have
\[
R_{0,{\scriptscriptstyle \textnormal{LU}}}\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{k}-\sum_{i=1}^{k}\Phi_{i}R_{0,{\scriptscriptstyle \textnormal{LU}}}\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{k-i}=\boldsymbol{\Phi}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=_{(1)}\boldsymbol{\Phi}_{0}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=R_{0,{\scriptscriptstyle \textnormal{LU}}}\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{k}-\sum_{i=1}^{k}\Phi_{0,i}R_{0,{\scriptscriptstyle \textnormal{LU}}}\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{k-i}=_{(2)}0
\]
where $=_{(1)}$ is by hypothesis, and $=_{(2)}$ by (ref).
Since $\smlabs{\lambda_{q+1}(\boldsymbol{\Phi})}<\smlabs{\lambda_{q}(\boldsymbol{\Phi}_{0})}=\smlabs{\lambda_{q}(\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}})}$
and $\boldsymbol{\Phi}\in\set P$, the result then follows by (ref).
({\romannumeral 2}). Analogously to (ref) above, define
\begin{align*}
H(\largedec R_{{\scriptscriptstyle LU}},\Lambda_{{\scriptscriptstyle LU}};\boldsymbol{\Phi}) & \coloneqq\begin{bmatrix}\largedec R_{{\scriptscriptstyle LU}}\Lambda_{{\scriptscriptstyle LU}}-F(\boldsymbol{\Phi})\largedec R_{{\scriptscriptstyle \textnormal{LU}}}; & \largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}-I_{q}\end{bmatrix}\\
H^{\ast}(\largedec R_{{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{LU}}};\boldsymbol{\Phi}) & \coloneqq\begin{bmatrix}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}-F(\boldsymbol{\Phi})\largedec R_{{\scriptscriptstyle \textnormal{LU}}}; & \largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}-I_{q}\end{bmatrix}.
\end{align*}
By the argument given in the proof of (ref), there
are smooth maps $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$, $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})$, $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$
and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})$ such that $H[\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}),\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi});\boldsymbol{\Phi}]=0$
and $H^{\ast}[\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi}),\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi});\boldsymbol{\Phi}]=0$
for all $\boldsymbol{\Phi}\in\set P$. Since $G^{\mathsf{T}}R_{0,{\scriptscriptstyle \textnormal{LU}}}=I_{q}$ implies
that $\largedec G^{\mathsf{T}}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=I_{q}$, we have $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})=\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}$
and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})=\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}$
when $\boldsymbol{\Phi}=\boldsymbol{\Phi}_{0}$, but otherwise these maps need not agree. Since
the maps $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})$ and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})$
are easier to work with, we first obtain the derivatives of these,
and subsequently those of $A(\boldsymbol{\Phi})$ and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$ via
renormalisation, analogously to (ref)--(ref).
Setting the total differential of $H^{\ast}$ to zero gives
\begin{equation}
0=\ensuremath{\mathrm{d}} H^{\ast}=\begin{bmatrix}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}(\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})+(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}-F_{0}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})-F(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}; & \largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})\end{bmatrix}
\end{equation}
where $F_{0}\coloneqq F(\boldsymbol{\Phi})$, whence by similar arguments as were
given in the proof of (ref),
\begin{equation}
F(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}(\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})+\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}-\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}).
\end{equation}
Vectorising gives
\begin{align}
\operatorname{vec}[F(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}] & =(I_{q}\otimes\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}})\operatorname{vec}(\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})+M\operatorname{vec}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})
\end{align}
for $M\coloneqq(I_{q}\otimes\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}})[(\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\otimes I_{kp-q})-(I_{q}\otimes\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}})](I_{q}\otimes\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})$.
Since $\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=0$ and $\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}=I_{kp-q}$,
setting
\[
M^{\dagger}\coloneqq(I_{q}\otimes\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}})[(\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\otimes I_{kp-q})-(I_{q}\otimes\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}})]^{-1}(I_{q}\otimes\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})
\]
we have $M^{\dagger}(I_{q}\otimes\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}})=0$ and $M^{\dagger}M=I_{q}\otimes\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}$.
Since $\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})=0$ by (ref),
it follows that
\[
\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}=(\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}+\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}=(\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}
\]
whence $M^{\dagger}M\operatorname{vec}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})=\operatorname{vec}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})$,
and so premultiplying (ref) by $M^{\dagger}$ gives
\begin{align*}
\operatorname{vec}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}) & =M^{\dagger}\operatorname{vec}[F(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}^{\ast}].
\end{align*}
By the structure of the companion form matrix, $\largedec L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}F(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}$.
Since $R$ is given by the final $p$ rows of $\largedec R$, we have
\begin{align}
\operatorname{vec}(\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}) & =(I_{q}\otimes R_{0,{\scriptscriptstyle \textnormal{ST}}})[(\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\otimes I_{kp-q})-(I_{q}\otimes\Lambda_{0,{\scriptscriptstyle \textnormal{ST}}})]^{-1}(I_{q}\otimes L_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}})\operatorname{vec}\{(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}\}\nonumber \\
& =B(\boldsymbol{\Phi}_{0})\operatorname{vec}\{(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}\}.
\end{align}
To compute the Jacobian of $A(\boldsymbol{\Phi})$, note that by partitioning the
$p\times p$ identity matrix as
\[
\begin{bmatrix}G_{\perp} & G\end{bmatrix}\coloneqq\begin{bmatrix}I_{r} & 0\\
0 & I_{q}
\end{bmatrix}
\]
we have $A(\boldsymbol{\Phi})=G_{\perp}^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=G_{\perp}^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})[G^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})]^{-1}$.
From $R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi}_{0})=R_{0,{\scriptscriptstyle \textnormal{LU}}}$, $G^{\mathsf{T}}R_{0,{\scriptscriptstyle \textnormal{LU}}}=\largedec G^{\mathsf{T}}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=I_{q}$
and $G_{\perp}^{\mathsf{T}}R_{0,{\scriptscriptstyle \textnormal{LU}}}=A_{0}$, it follows that at $\boldsymbol{\Phi}=\boldsymbol{\Phi}_{0}$
\begin{align}
\ensuremath{\mathrm{d}} A & =G_{\perp}^{\mathsf{T}}(\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})-(G_{\perp}^{\mathsf{T}}R_{0,{\scriptscriptstyle \textnormal{LU}}})G^{\mathsf{T}}(\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})=(G_{\perp}^{\mathsf{T}}-A_{0}G^{\mathsf{T}})\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}=\beta_{0}^{\mathsf{T}}\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}
\end{align}
for $\beta_{0}^{\mathsf{T}}=[I_{r},-A_{0}]$. The first part of (ref)
follows immediately from (ref) and (ref). For the Jacobian
of $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$, note that (as per (ref) above)
\[
\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=[\largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})]\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})[\largedec G^{\mathsf{T}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}(\boldsymbol{\Phi})]^{-1}
\]
whence at $\boldsymbol{\Phi}=\boldsymbol{\Phi}_{0}$,
\begin{align*}
\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}} & =\largedec G^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}+\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}-\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}\largedec G^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}).
\end{align*}
Recognising that $\largedec G^{\mathsf{T}}(\ensuremath{\mathrm{d}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})=G^{\mathsf{T}}(\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})$
and vectorising, we have
\begin{equation}
\operatorname{vec}(\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}})=\{(\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\otimes I_{q})-(I_{q}\otimes\Lambda_{0,{\scriptscriptstyle \textnormal{LU}}})\}(I_{q}\otimes G^{\mathsf{T}})\operatorname{vec}(\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast})+\operatorname{vec}(\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}).
\end{equation}
$\ensuremath{\mathrm{d}} R_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}$ is given in (ref) above. To obtain
$\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}$, note that premultiplying (ref)
by $\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}$ yields
\begin{equation}
\ensuremath{\mathrm{d}}\Lambda_{{\scriptscriptstyle \textnormal{LU}}}^{\ast}=\largedec L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}F(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}=L_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}(\ensuremath{\mathrm{d}}\boldsymbol{\Phi})\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}.
\end{equation}
Thus (ref), (ref) and (ref) give the second
part of (ref).
\textbf{({\romannumeral 3}).} Continuity of $J(\boldsymbol{\Phi})$ is immediate from $A(\boldsymbol{\Phi})$
and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})$ being smooth.
proof[Proof of (ref)]
The stated expression for $J(\boldsymbol{\Phi})$ is immediate from (ref),
(ref), and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=I_{q}$. That $J(\boldsymbol{\Phi})$
is nonsingular will follow once we have shown that the $(p\times p)$
matrix
\begin{equation}
K\coloneqq\begin{bmatrix}\beta^{\mathsf{T}}R_{{\scriptscriptstyle ST}}(I_{kp-q}-\Lambda_{{\scriptscriptstyle ST}})^{-1}L_{{\scriptscriptstyle ST}}^{\mathsf{T}}\\
L_{{\scriptscriptstyle LU}}^{\mathsf{T}}
\end{bmatrix}
\end{equation}
is nonsingular. We first note the following facts. Since $\boldsymbol{\Phi}\in\set P$
with $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=I_{q}$, it follows from (ref)
that $\operatorname{rk}\Phi(1)\leq p-q$. Since $\Phi(\cdot)$ has exactly $q$
roots at unity, the reverse inequality holds by Corollary 4.3 of
Joh95, whence $\operatorname{rk}\Phi(1)=p-q$. Thus (ref) holds:
this implies that $\operatorname{sp}\beta=\operatorname{sp}\Phi(1)^{\mathsf{T}}$ and $\operatorname{rk} L_{{\scriptscriptstyle \textnormal{LU}}}=q$
(see (ref) and the characterisation of the CS discussed in
(ref)).
Now let $c\in\mathbb{R}^{p}$ be such that $Kc=0$, so that in particular
$L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}c=0$. Since $\operatorname{rk}\Phi(1)+\operatorname{rk} L_{{\scriptscriptstyle \textnormal{LU}}}=p$, while
(ref) with $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}=I_{q}$ implies $L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Phi(1)=0$,
it follows that $c\in\operatorname{sp}\Phi(1)$, i.e.\ $c=\Phi(1)b$ for some
$b\in\mathbb{R}^{p}$. By GLR82, $\Phi(\mu)^{-1}=R(\mu I-\Lambda)^{-1}L^{\mathsf{T}}$
for any $\mu$ not a root of $\Phi(\cdot)$. Since the columns of
the quasi-cointegrating matrix $\beta$ are orthogonal to $R_{{\scriptscriptstyle \textnormal{LU}}}$,
we have
\begin{equation}
\beta^{\mathsf{T}}=\beta^{\mathsf{T}}R_{{\scriptscriptstyle ST}}(\mu I_{kp-q}-\Lambda_{{\scriptscriptstyle ST}})^{-1}L_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\Phi(\mu)\ensuremath{\rightarrow}\beta^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{ST}}}(I_{kp-q}-\Lambda_{{\scriptscriptstyle \textnormal{ST}}})^{-1}L_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\Phi(1)
\end{equation}
by the continuity of the r.h.s., as $\mu\ensuremath{\rightarrow}1$, since $\Lambda_{{\scriptscriptstyle \textnormal{ST}}}$
has no eigenvalues at unity. Hence
\[
0=Kc=\begin{bmatrix}\beta^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{ST}}}(I_{kp-q}-\Lambda_{{\scriptscriptstyle \textnormal{ST}}})^{-1}L_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\Phi(1)b\\
0
\end{bmatrix}=\begin{bmatrix}\beta^{\mathsf{T}}b\\
0
\end{bmatrix}
\]
implying $\beta^{\mathsf{T}}b=0$. But $\operatorname{sp}\beta=\operatorname{sp}\Phi(1)^{\mathsf{T}}$,
so we must have $\Phi(1)b=0$. Thus $c=0$, from which it follows
that $K$ is nonsingular.
Asymptotics
The assumptions (ref) and (ref) are maintained throughout
this appendix. We first recall some notation. Let $\boldsymbol{\Phi}_{0}\coloneqq\lim_{n\ensuremath{\rightarrow}\infty}\boldsymbol{\Phi}_{n}$,
where $\{\boldsymbol{\Phi}_{n}\}$ is the sequence specified by (ref).
Let $R_{n}\coloneqq[R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n}),R_{{\scriptscriptstyle \textnormal{ST}}}]$ and $\Lambda_{n}\coloneqq\operatorname{diag}\{\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}},\Lambda_{{\scriptscriptstyle \textnormal{ST}}}\}$
be as in (ref). Take $\largedec R_{n}\coloneqq\operatorname{col}\{R_{n}\Lambda_{n}^{k-i}\}_{i=1}^{k}$
and $\largedec L_{n}\coloneqq(\largedec R_{n}^{\mathsf{T}})^{-1}$ as in (ref), and
partition these as $\largedec R_{n}=[\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}},\largedec R_{n,{\scriptscriptstyle \textnormal{ST}}}]$
and $\largedec L_{n}=[\largedec L_{n,{\scriptscriptstyle \textnormal{LU}}},\largedec L_{n,{\scriptscriptstyle \textnormal{ST}}}]$ (as per (ref));
note that both these matrices are convergent.
Let $z_{{\scriptscriptstyle \textnormal{LU}},t}\coloneqq\largedec L_{n,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\largedec x_{t}$ and $z_{{\scriptscriptstyle \textnormal{ST}},t}=\largedec L_{n,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\largedec x_{t}$
be as in (ref) (for $\boldsymbol{\Phi}=\boldsymbol{\Phi}_{n}$); these follow the
autoregressions given in (ref). Recall $E\sim\mathrm{BM}(\Sigma)$
and $Z_{C}(r)\coloneqq\int_{0}^{r}\mathrm{e}^{C(r-s)}L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\ensuremath{\mathrm{d}} E(s)$
from (ref). For $i\in\{{\scriptscriptstyle \textnormal{LU}},{\scriptscriptstyle \textnormal{ST}}\}$, let $\bar{z}_{i,t}$ denote
the residual from an OLS regression of $\{\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}\}_{t=1}^{n}$
onto a constant and linear trend. Recall that $\bar{Z}_{C}$ denotes
the residual from an $L^{2}[0,1]$ projection of each sample path
of $Z_{C}$ onto a constant and linear trend. As in (ref),
let $\hat{\Sigma}_{n}$ denote the unrestricted MLE for $\Sigma$,
i.e.\ the OLS residual variance matrix estimator.
Proofs of the following results appear at the end of this section.
lemThe following hold jointly:
\begin{enumerate}
• $n^{-1/2}\sum_{t=1}^{\smlfloor{nr}}\varepsilon_{t}\ensuremath{\rightsquigarrow} E(r)$
• $n^{-1/2}z_{{\scriptscriptstyle \textnormal{LU}},\smlfloor{nr}}\ensuremath{\rightsquigarrow} Z_{C}(r)$
• $n^{-1/2}\bar{z}_{{\scriptscriptstyle \textnormal{LU}},\smlfloor{nr}}\ensuremath{\rightsquigarrow}\bar{Z}_{C}(r)$
\end{enumerate}
as weak convergences on the space of right-continuous functions $[0,1]\ensuremath{\rightarrow}\mathbb{R}^{m}$
(with respect to the uniform topology); and
\begin{enumerate}[resume]
• $n^{-1}\sum_{t=1}^{n}(\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}\otimes\varepsilon_{t})\ensuremath{\rightsquigarrow}\int_{0}^{1}[\bar{Z}_{C}(r)\otimes\ensuremath{\mathrm{d}} E(r)]\ensuremath{\,\ensuremath{\mathrm{d}}} r$
• $n^{-1/2}\sum_{t=1}^{n}(\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}\otimes\varepsilon_{t})\ensuremath{\rightsquigarrow}\xi\sim\mathrm{N}[0,\Omega\otimes\Sigma]$
• $\hat{\Sigma}_{n}\ensuremath{\overset{p}{\ensuremath{\rightarrow}}}\Sigma$,
\end{enumerate}
where $\Omega\coloneqq\lim_{n\ensuremath{\rightarrow}\infty}\operatorname{var}(z_{{\scriptscriptstyle \textnormal{ST}},n})$ is positive
definite, and $\xi$ is independent of $E$.
Now define the reparametrisation $\boldsymbol{\Phi}\ensuremath{\mapsto}\varphi$ by
equation[equation omitted — 517 chars of source]
which is reversed by setting $\boldsymbol{\Phi}=\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi)\largedec L_{n}^{\mathsf{T}}$,
where $\operatorname{vec}^{-1}(x)$ maps $x\in\mathbb{R}^{kp^{2}}$ to the matrix $X\in\mathbb{R}^{p\times kp}$
for which $\operatorname{vec}(X)=x$. The parameter space for $\varphi$ is the open
set
equation[equation omitted — 167 chars of source]
and the true parameters correspond to $\varphi=0$. Although $\mathcal{P}_{n}$
depends on $n$, since $\boldsymbol{\Phi}_{n}\ensuremath{\rightarrow}\boldsymbol{\Phi}_{0}\in\set P$ and $\set P$
is open ((ref)), there is an $\epsilon>0$ such
that $\mathcal{P}_{n}$ contains a ball of radius $\epsilon$ centred at
the origin, for all $n$ sufficiently large. Let
\[
\mathcal{\ell}^{\ast}_{n}(\varphi)\coloneqq\mathcal{\ell}_{n}[\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi)\largedec L_{n}^{\mathsf{T}},\hat{\Sigma}_{n}].
\]
Define $D_{n}\coloneqq\operatorname{diag}\{nI_{{\scriptscriptstyle \#}{\scriptscriptstyle \textnormal{LU}}},n^{1/2}I_{{\scriptscriptstyle \#}{\scriptscriptstyle \textnormal{ST}}}\}$, where
${\scriptscriptstyle \#}{\scriptscriptstyle \textnormal{LU}}\coloneqq pq$ and ${\scriptscriptstyle \#}{\scriptscriptstyle \textnormal{ST}}\coloneqq p(kp-q)$ correspond to the
dimensions of the vectors $\varphi_{{\scriptscriptstyle \textnormal{LU}}}$ and $\varphi_{{\scriptscriptstyle \textnormal{ST}}}$ respectively.
lemThere exist $S_{n}$ and $H_{n}$ such that for
all $\varphi\in\mathcal{P}_{n}$,
\[
\mathcal{\ell}^{\ast}_{n}(\varphi)-\mathcal{\ell}^{\ast}(0)=S_{n}^{\mathsf{T}}(D_{n}\varphi)-\tfrac{1}{2}(D_{n}\varphi)^{\mathsf{T}}H_{n}(D_{n}\varphi)
\]
where
\begin{gather*}
S_{n}\ensuremath{\rightsquigarrow}\begin{bmatrix}\int_{0}^{1}[\bar{Z}_{C}(r)\otimes\Sigma^{-1}\ensuremath{\mathrm{d}} E(r)]\\
\xi
\end{bmatrix}\eqqcolon\begin{bmatrix}S_{{\scriptscriptstyle LU}}\\
S_{{\scriptscriptstyle ST}}
\end{bmatrix}\eqqcolon S\\
H_{n}\ensuremath{\rightsquigarrow}\begin{bmatrix}\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}} & 0\\
0 & \Omega
\end{bmatrix}\otimes\Sigma^{-1}\eqqcolon\begin{bmatrix}H_{{\scriptscriptstyle LU}} & 0\\
0 & H_{{\scriptscriptstyle ST}}
\end{bmatrix}\eqqcolon H,
\end{gather*}
for $\xi$ as in (ref).
Define the constraint maps
align[align omitted — 391 chars of source]
and the associated restricted parameter spaces
align*[align* omitted — 240 chars of source]
Let $\hat{\varphi}_{n}$, $\hat{\varphi}_{n\mid\theta}$ and $\hat{\varphi}_{n\mid\theta,\gamma}$
denote exact maximisers of $\mathcal{\ell}^{\ast}_{n}(\varphi)$ over the sets $\mathcal{P}_{n}$,
$\mathcal{P}_{n\mid\theta}$ and $\mathcal{P}_{n\mid\theta,\gamma}$ respectively:
which may be shown to exist at least with with probability approaching
one (w.p.a.1), and may be arbitrarily defined otherwise.
lemEach of $D_{n}\hat{\varphi}_{n}$, $D_{n}\hat{\varphi}_{n\mid\theta}$
and $D_{n}\hat{\varphi}_{n\mid\theta,\gamma}$ are $O_{p}(1)$.
Let $\nabla_{\varphi}g(\varphi_{0})$ denote the gradient of $g:\mathcal{P}\ensuremath{\rightarrow}\mathbb{R}^{d_{g}}$
at $\varphi=\varphi_{0}$. The derivatives of the maps $\theta_{n}$ and $\gamma_{n}$
can be inferred from (ref). Part (ref)
of that result gives the derivatives with respect to $\varphi_{{\scriptscriptstyle \textnormal{LU}}}$,
and part (ref) implies that when $\varphi_{{\scriptscriptstyle \textnormal{LU}}}=0$, the
first (and higher order) derivatives with respect to $\varphi_{{\scriptscriptstyle \textnormal{ST}}}$ are
identically zero. Now letting $e_{d,i}\in\mathbb{R}^{d}$ denote a vector
with zero everywhere except for a $1$ in the $i$th position, define
\[
\Pi\coloneqq[
matrix[matrix omitted — 28 chars of source]
]\coloneqq[
matrix[matrix omitted — 265 chars of source]
],
\]
which by (ref) has full column rank, and
align*[align* omitted — 261 chars of source]
lem\begin{enumerate}
• Let $\{\tilde{\varphi}_{n}\}$ denote a random sequence
in $\mathcal{P}_{n}$ with $\tilde{\varphi}_{n}\ensuremath{\overset{p}{\ensuremath{\rightarrow}}}0$. Then
\begin{align*}
\nabla_{\varphi}\theta_{n}(\tilde{\varphi}_{n}) & \ensuremath{\overset{p}{\ensuremath{\rightarrow}}}\boldsymbol{\Theta} & \nabla_{\varphi}\gamma_{n}(\tilde{\varphi}_{n}) & \ensuremath{\overset{p}{\ensuremath{\rightarrow}}}\boldsymbol{\Gamma}.
\end{align*}
• Let $\mathcal{Q}_{\boldsymbol{\Theta},\perp}$ and $\mathcal{Q}_{\boldsymbol{\Pi},\perp}$
denote orthogonal projections from $\mathbb{R}^{kp^{2}}$ onto the subspaces
orthogonal to the the columns of $\boldsymbol{\Theta}$ and $\boldsymbol{\Pi}$. Then
\begin{align*}
D_{n}\hat{\varphi}_{n\mid\theta} & =\mathcal{Q}_{\boldsymbol{\Theta},\perp}D_{n}\hat{\varphi}_{n\mid\theta}+o_{p}(1)\\
D_{n}\hat{\varphi}_{n\mid\theta,\gamma} & =\mathcal{Q}_{\boldsymbol{\Pi},\perp}D_{n}\hat{\varphi}_{n\mid\theta,\gamma}+o_{p}(1).
\end{align*}
\end{enumerate}
Let $\Theta_{\perp}\in\mathbb{R}^{pq\times qr}$ and $\Pi_{\perp}\in\mathbb{R}^{pq\times(qr-1)}$
denote matrices having full column rank, such that $\Theta_{\perp}^{\mathsf{T}}\Theta=0$
and $\Pi_{\perp}^{\mathsf{T}}\Pi=0$. We may take $\Theta_{\perp}=I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}},\perp}$,
for $L_{{\scriptscriptstyle \textnormal{LU}},\perp}$ a $p\times r$ matrix having rank $r$ and for
which $L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}L_{{\scriptscriptstyle \textnormal{LU}}}=0$. Since $\Pi=[\Theta,\Gamma]$
there exists a full column rank matrix $\Xi\in\mathbb{R}^{qr\times(qr-1)}$
for which $\Pi_{\perp}\coloneqq\Theta_{\perp}\Xi$.
prop\begin{enumerate}
• $D_{n}\hat{\varphi}_{n}=\begin{bmatrix}n\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}}\\
n^{1/2}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\ensuremath{\rightsquigarrow}\begin{bmatrix}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}S_{{\scriptscriptstyle \textnormal{LU}}}\\
H_{{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}$,
• $D_{n}\hat{\varphi}_{n\mid\theta}=\begin{bmatrix}n\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}\\
n^{1/2}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta}
\end{bmatrix}\ensuremath{\rightsquigarrow}\begin{bmatrix}\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}\\
H_{{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}$,
• $2[\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n})-\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n\mid\theta})]\ensuremath{\rightsquigarrow} S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}\Theta(\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}\Theta)^{-1}\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}S_{{\scriptscriptstyle \textnormal{LU}}}$.
\end{enumerate}
Let $H_{\Theta,\perp}\coloneqq\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp}$,
and $\mathcal{Q}\in\mathbb{R}^{qr\times qr}$ denote the orthogonal projection
onto $\operatorname{sp} H_{\Theta,\perp}^{1/2}\Xi$. Then
\begin{enumerate}[resume]
• $2[\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n\mid\theta})-\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n\mid\theta,\gamma})]\ensuremath{\rightsquigarrow}(H_{\Theta,\perp}^{-1/2}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}})^{\mathsf{T}}[I_{qr}-\mathcal{Q}](H_{\Theta,\perp}^{-1/2}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}).$
\end{enumerate}
The preceding gives the limiting distribution of $\hat{\boldsymbol{\Phi}}_{n}$
under the reparametrisation (ref); the limiting distributions
of estimators of $A$ and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}$ will then follow by an
application of the delta method, as per
propLet $\{\boldsymbol{\Phi}_{n}\}$ be as in (ref),
$\boldsymbol{\Phi}_{0}\coloneqq\lim_{n\ensuremath{\rightarrow}\infty}\boldsymbol{\Phi}_{n}\in\set P$, and $\{\tilde{\boldsymbol{\Phi}}_{n}\}$
a random sequence in $\set P$ with $\tilde{\boldsymbol{\Phi}}_{n}=\boldsymbol{\Phi}_{n}+o_{p}(1)$.
Then
\begin{equation}
\begin{bmatrix}\operatorname{vec}\{A(\tilde{\boldsymbol{\Phi}}_{n})-A(\boldsymbol{\Phi}_{n})\}\\
\operatorname{vec}\{\Lambda_{{\scriptscriptstyle LU}}(\tilde{\boldsymbol{\Phi}}_{n})-\Lambda_{{\scriptscriptstyle LU}}(\boldsymbol{\Phi}_{n})\}
\end{bmatrix}=\left(\begin{bmatrix}J_{A}(\boldsymbol{\Phi}_{0})\\
J_{\Lambda}(\boldsymbol{\Phi}_{0})
\end{bmatrix}+o_{p}(1)\right)\operatorname{vec}\{(\tilde{\boldsymbol{\Phi}}_{n}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle LU}}\}
\end{equation}
where $\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\coloneqq\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n})$.
proof[Proof of (ref)]
(ref)--(ref) follow by Donsker's
theorem for partial sums, Lemma 3.1 in Phi88Ecta and the
continuous mapping theorem; (ref) by the martingale
central limit theorem (HH80); and (ref)
by arguments similar to those given in Section 3.2.2 of Lut07.
proof[Proof of (ref)]
Let $\Phi_{i}\coloneqq\boldsymbol{\Phi}\largedec R_{n,i}$ and $\Phi_{n,i}\coloneqq\boldsymbol{\Phi}_{n}\largedec R_{n,i}$
for $i\in\{{\scriptscriptstyle \textnormal{LU}},{\scriptscriptstyle \textnormal{ST}}\}$. Then
\begin{align*}
\mathcal{\ell}_{n}(\boldsymbol{\Phi},\Sigma) & =-\frac{n}{2}\log\det\Sigma-\min_{m,d}\frac{1}{2}\sum_{t=1}^{n}\norm{y_{t}-m-dt-\boldsymbol{\Phi}\largedec y_{t-1}}_{\Sigma^{-1}}^{2}\\
& =-\frac{n}{2}\log\det\Sigma-\min_{m,d}\frac{1}{2}\sum_{t=1}^{n}\norm{x_{t}-m-dt-\boldsymbol{\Phi}\largedec x_{t-1}}_{\Sigma^{-1}}^{2}\\
& =-\frac{n}{2}\log\det\Sigma-\min_{m,d}\frac{1}{2}\sum_{t=1}^{n}\norm{x_{t}-m-dt-\Phi_{{\scriptscriptstyle LU}}z_{{\scriptscriptstyle LU},t-1}-\Phi_{{\scriptscriptstyle ST}}z_{{\scriptscriptstyle ST},t-1}}_{\Sigma^{-1}}^{2}\\
& =-\frac{n}{2}\log\det\Sigma-\frac{1}{2}\sum_{t=1}^{n}\norm{\bar{x}_{t}-\Phi_{{\scriptscriptstyle LU}}\bar{z}_{{\scriptscriptstyle LU},t-1}-\Phi_{{\scriptscriptstyle \textnormal{ST}}}\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}}_{\Sigma^{-1}}^{2}
\end{align*}
Twice differentiating the r.h.s.\ (as in Lut07)
with respect to $\Phi_{{\scriptscriptstyle \textnormal{LU}}}$ and $\Phi_{{\scriptscriptstyle \textnormal{ST}}}$, and noting that $\varphi_{i}=\operatorname{vec}(\Phi_{i}-\Phi_{n,i})$,
we thus have
\begin{align*}
\mathcal{\ell}^{\ast}_{n}(\varphi)-\mathcal{\ell}^{\ast}_{n}(0)=\mathcal{\ell}_{n}(\boldsymbol{\Phi},\hat{\Sigma}_{n})-\mathcal{\ell}_{n}(\boldsymbol{\Phi}_{n},\hat{\Sigma}_{n}) & =S_{n}^{\mathsf{T}}(D_{n}\varphi)-\tfrac{1}{2}(D_{n}\varphi)^{\mathsf{T}}H_{n}(D_{n}\varphi)
\end{align*}
where
\begin{gather*}
S_{n}\coloneqq\begin{bmatrix}n^{-1}\sum_{t=1}^{n}(\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}\otimes\hat{\Sigma}_{n}^{-1}\bar{\varepsilon}_{t})\\
n^{-1/2}\sum_{t=1}^{n}(\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}\otimes\hat{\Sigma}_{n}^{-1}\bar{\varepsilon}_{t})
\end{bmatrix}=_{(1)}\begin{bmatrix}\frac{1}{n}\sum_{t=1}^{n}(\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}\otimes\hat{\Sigma}_{n}^{-1}\varepsilon_{t})\\
\frac{1}{n^{1/2}}\sum_{t=1}^{n}(\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}\otimes\hat{\Sigma}_{n}^{-1}\varepsilon_{t})
\end{bmatrix}\\
H_{n}\coloneqq\begin{bmatrix}n^{-2}\sum_{t=1}^{n}\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}^{\mathsf{T}} & n^{-3/2}\sum_{t=1}^{n}\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}^{\mathsf{T}}\\
n^{-3/2}\sum_{t=1}^{n}\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}^{\mathsf{T}} & n^{-1}\sum_{t=1}^{n}\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}^{\mathsf{T}}
\end{bmatrix}\otimes\hat{\Sigma}_{n}^{-1},
\end{gather*}
and $\bar{\varepsilon}_{t}$ denotes the residual from an OLS regression of
$\{\varepsilon_{t}\}_{t=1}^{n}$ on a constant and a linear trend; $=_{(1)}$
holds because each element of $\bar{z}_{{\scriptscriptstyle \textnormal{LU}},t-1}$ and $\bar{z}_{{\scriptscriptstyle \textnormal{ST}},t-1}$
is orthogonal to a constant and linear trend. The stated convergences
of $S_{n}$ and $H_{n}$ then follow by (ref) and the continuous
mapping theorem.
proof[Proof of (ref)]
By (ref), we have
\begin{align*}
\mathcal{\ell}^{\ast}_{n}(\varphi)-\mathcal{\ell}^{\ast}_{n}(0) & \leq\smlnorm{D_{n}\varphi}[\smlnorm{S_{n}}-\tfrac{1}{2}\lambda_{\min}(H_{n})\smlnorm{D_{n}\varphi}].
\end{align*}
Let $M<\infty$ and $\epsilon>0$. Since $D_{n}=\operatorname{diag}\{nI_{{\scriptscriptstyle \#}{\scriptscriptstyle \textnormal{LU}}},n^{1/2}I_{{\scriptscriptstyle \#}{\scriptscriptstyle \textnormal{ST}}}\}$,
$S_{n}=O_{p}(1)$ and $H_{n}\ensuremath{\rightsquigarrow} H$ is positive definite w.p.a.1,
it is evident that
\begin{align*}
\ensuremath{\mathbb{P}}\left\{ \sup_{\{\varphi\in\mathcal{P}_{n}\mid\smlnorm{D_{n}\varphi}\geq M\}}[\mathcal{\ell}^{\ast}_{n}(\varphi)-\mathcal{\ell}^{\ast}_{n}(0)]<-\epsilon\right\} & \ge\ensuremath{\mathbb{P}}\left\{ M[\smlnorm{S_{n}}-\tfrac{1}{2}\lambda_{\min}(H_{n})M]<-\epsilon\right\}
\end{align*}
and
\begin{align*}
\limsup_{n\ensuremath{\rightarrow}\infty}\ensuremath{\mathbb{P}}\left\{ M[\smlnorm{S_{n}}-\tfrac{1}{2}\lambda_{\min}(H_{n})M]<-\epsilon\right\} & \geq\ensuremath{\mathbb{P}}\left\{ M[\smlnorm S-\tfrac{1}{2}\lambda_{\min}(H)M]<-\epsilon\right\} \ensuremath{\rightarrow}1
\end{align*}
as $M\ensuremath{\rightarrow}\infty$. Deduce that $D_{n}\hat{\varphi}_{n}=O_{p}(1)$. Since
$\mathcal{P}_{n\mid\theta,\gamma}\subset\mathcal{P}_{n\mid\theta}\subset\mathcal{P}_{n}$
and $0\in\mathcal{P}_{n\mid\theta,\gamma}$, that $D_{n}\hat{\varphi}_{n\mid\theta}$
and $D_{n}\hat{\varphi}_{n\mid\theta,\gamma}$ are stochastically bounded
follows by the same argument.
proof[Proof of (ref)]
({\romannumeral 1}). Since $\boldsymbol{\Phi}_{n}\ensuremath{\rightarrow}\boldsymbol{\Phi}_{0}$, $\largedec L_{n}\ensuremath{\rightarrow}\largedec L_{0}$
and $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\cdot)$ is continuously differentiable ((ref)),
\[
\nabla_{\varphi}\theta_{n}(\tilde{\varphi}_{n})\ensuremath{\overset{p}{\ensuremath{\rightarrow}}}\nabla_{\varphi}\operatorname{vec}\{\Lambda_{{\scriptscriptstyle \textnormal{LU}}}[\boldsymbol{\Phi}_{0}+\operatorname{vec}^{-1}(\varphi)\largedec L_{0}^{\mathsf{T}}]\}|_{\varphi=0}=_{(1)}\begin{bmatrix}I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}}}\\
0_{\#{\scriptscriptstyle \textnormal{ST}}\times q^{2}}
\end{bmatrix}=\boldsymbol{\Theta}
\]
where $=_{(1)}$ follows by Lemmas (ref) and (ref).
The probability limit of $\nabla_{\varphi}\gamma_{n}(\tilde{\varphi}_{n})$ follows
similarly.
({\romannumeral 2}). By (ref) and the remarks following
(ref), there exists a ball $B(0,\epsilon)$ of radius $\epsilon>0$,
centred on the origin, such that $B(0,\epsilon)\subset\mathcal{P}_{n}$
for all $n$ sufficiently large, and $\ensuremath{\mathbb{P}}\{\hat{\varphi}_{n\mid\theta}\in B(0,\epsilon)\}\ensuremath{\rightarrow}1$.
We may take $\epsilon$ sufficiently small that $\boldsymbol{\Phi}_{\varphi}\coloneqq\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi)\largedec L_{n}^{\mathsf{T}}$
has $\smlabs{\lambda_{q+1}(\boldsymbol{\Phi}_{\varphi})}<\smlabs{\lambda_{q}(\boldsymbol{\Phi}_{n})}$
for all $n$ sufficiently large, for all $\varphi\in B(0,\epsilon)$. In
particular, suppose $\varphi_{{\scriptscriptstyle \textnormal{LU}}}=0$; then $(\boldsymbol{\Phi}_{\varphi}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}=0$
and we have by (ref)(ref) that $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{\varphi})=\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n})=C/n$.
It follows that $\theta_{n}(0,\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta})=0$ w.p.a.1.,
whence
\begin{multline*}
0=\theta_{n}(\hat{\varphi}_{n,{\scriptscriptstyle LU}\mid\theta},\hat{\varphi}_{n,{\scriptscriptstyle ST}\mid\theta})=\theta_{n}(\hat{\varphi}_{n,{\scriptscriptstyle LU}\mid\theta},\hat{\varphi}_{n,{\scriptscriptstyle ST}\mid\theta})-\theta_{n}(0,\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta})\\
=[\Theta+o_{p}(1)]^{\mathsf{T}}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}=\Theta^{\mathsf{T}}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}+o_{p}(\smlnorm{\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}})
\end{multline*}
by part (i) of the lemma and a mean value expansion. Hence, letting
$\mathcal{Q}_{\Theta}$ and $\mathcal{Q}_{\Theta,\perp}$ denote the matrices
that orthogonally project from $\mathbb{R}^{\#{\scriptscriptstyle \textnormal{LU}}}$ onto $\operatorname{sp}\Theta$
and $(\operatorname{sp}\Theta)^{\perp}$ respectively, we have
\begin{align*}
D_{n}\hat{\varphi}_{n\mid\theta} & =\begin{bmatrix}nI_{\#{\scriptscriptstyle \textnormal{LU}}} & 0\\
0 & n^{1/2}I_{\#{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\begin{bmatrix}\mathcal{Q}_{\Theta}+\mathcal{Q}_{\Theta,\perp} & 0\\
0 & I_{\#{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\begin{bmatrix}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}\\
\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta}
\end{bmatrix}\\
& =\begin{bmatrix}\mathcal{Q}_{\Theta,\perp} & 0\\
0 & I_{\#{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\begin{bmatrix}n\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}\\
n^{1/2}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta}
\end{bmatrix}+o_{p}(n\smlnorm{\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}})=\mathcal{Q}_{\boldsymbol{\Theta},\perp}D_{n}\hat{\varphi}_{n\mid\theta}+o_{p}(\smlnorm{D_{n}\hat{\varphi}_{n\mid\theta}}).\qedhere
\end{align*}
proof[Proof of (ref)]
({\romannumeral 1}). Immediate from (ref).
({\romannumeral 2}). As in the proof of (ref)(ref),
we may take $\epsilon>0$ such that $B(0,\epsilon)\subset\mathcal{P}_{n}$
for all $n$ sufficiently large, and $\ensuremath{\mathbb{P}}\{\hat{\varphi}_{n\mid\theta}\in B(0,\epsilon)\}\ensuremath{\rightarrow}1$.
Hence w.p.a.1., $\hat{\varphi}_{n\mid\theta}$ satisfies the first-order
conditions for a constrained interior maximum,
\[
\nabla_{\varphi}\mathcal{\ell}_{n}^{\ast}(\hat{\varphi}_{n\mid\theta})=D_{n}S_{n}-D_{n}H_{n}(D_{n}\hat{\varphi}_{n\mid\theta})=\nabla_{\varphi}\theta_{n}(\hat{\varphi}_{n\mid\theta})\mu_{n},
\]
where $\mu_{n}\in\mathbb{R}^{q^{2}}$ is a vector of Lagrange multipliers;
whence
\begin{equation}
S_{n}-H_{n}(D_{n}\hat{\varphi}_{n\mid\theta})=(nD_{n}^{-1})\nabla_{\varphi}\theta_{n}(\hat{\varphi}_{n\mid\theta})(n^{-1}\mu_{n})\eqqcolon\boldsymbol{\Theta}_{n}(n^{-1}\mu_{n})
\end{equation}
w.p.a.1. By a similar argument as given in the proof of (ref)(ref),
it follows from (ref)(ref) that $\nabla_{\varphi_{{\scriptscriptstyle \textnormal{ST}}}}\theta_{n}(0,\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta})=0$
w.p.a.1, and so by a a mean value expansion and (ref),
\[
\nabla_{\varphi_{{\scriptscriptstyle \textnormal{ST}}}}\theta_{n}(\hat{\varphi}_{n\mid\theta})=\nabla_{\varphi_{{\scriptscriptstyle \textnormal{ST}}}}\theta_{n}(\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta},\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta})-\nabla_{\varphi_{{\scriptscriptstyle \textnormal{ST}}}}\theta_{n}(0,\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\theta})=O_{p}(\smlnorm{\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}})=O_{p}(n^{-1}).
\]
Deduce from the preceding and (ref)(ref)
that
\[
\boldsymbol{\Theta}_{n}=(nD_{n}^{-1})\nabla_{\varphi}\theta_{n}(\hat{\varphi}_{n\mid\theta})=\begin{bmatrix}\nabla_{\varphi_{{\scriptscriptstyle \textnormal{LU}}}}\theta_{n}(\hat{\varphi}_{n\mid\theta})\\
n^{1/2}\nabla_{\varphi_{{\scriptscriptstyle \textnormal{ST}}}}\theta_{n}(\hat{\varphi}_{n\mid\theta})
\end{bmatrix}\ensuremath{\overset{p}{\ensuremath{\rightarrow}}}\boldsymbol{\Theta},
\]
which has full column rank. Let $\boldsymbol{\Theta}_{\perp}\coloneqq\operatorname{diag}\{\Theta_{\perp},I_{\#{\scriptscriptstyle \textnormal{ST}}}\}$,
a full column rank matrix for which $\boldsymbol{\Theta}_{\perp}^{\mathsf{T}}\boldsymbol{\Theta}=0$;
then $\boldsymbol{\Theta}_{n,\perp}\coloneqq[I_{kp^{2}}-\boldsymbol{\Theta}_{n}(\boldsymbol{\Theta}_{n}^{\mathsf{T}}\boldsymbol{\Theta}_{n})^{-1}\boldsymbol{\Theta}_{n}^{\mathsf{T}}]\boldsymbol{\Theta}_{\perp}\ensuremath{\overset{p}{\ensuremath{\rightarrow}}}\boldsymbol{\Theta}_{\perp}$
and $\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}\boldsymbol{\Theta}_{n}=0$ for all $n$. Hence w.p.a.1
\begin{align*}
0 & =_{(1)}\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}S_{n}-\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}H_{n}(D_{n}\hat{\varphi}_{n\mid\theta})\\
& =_{(2)}\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}S_{n}-\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}H_{n}[\boldsymbol{\Theta}_{\perp}(\boldsymbol{\Theta}_{\perp}^{\mathsf{T}}\boldsymbol{\Theta}_{\perp})^{-1}\boldsymbol{\Theta}_{\perp}^{\mathsf{T}}(D_{n}\hat{\varphi}_{n\mid\theta})+o_{p}(\smlnorm{D_{n}\hat{\varphi}_{n\mid\theta}})]
\end{align*}
where $=_{(1)}$ follows from premultiplying (ref) by
$\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}$, and $=_{(2)}$ from (ref)(ref).
A further appeal to that result and rearranging the preceding yields
\[
D_{n}\hat{\varphi}_{n\mid\theta}=\mathcal{Q}_{\boldsymbol{\Theta},\perp}D_{n}\hat{\varphi}_{n\mid\theta}+o_{p}(\smlnorm{D_{n}\hat{\varphi}_{n\mid\theta}})=\boldsymbol{\Theta}_{\perp}(\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}H_{n}\boldsymbol{\Theta}_{\perp})^{-1}\boldsymbol{\Theta}_{n,\perp}^{\mathsf{T}}S_{n}+o_{p}(1+\smlnorm{D_{n}\hat{\varphi}_{n\mid\theta}}).
\]
The result then follows by Lemmas (ref) and (ref).
({\romannumeral 3}). From parts (i) and (ii) and (ref) we
have
\begin{gather}
2[\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n})-\mathcal{\ell}_{n}^{\ast}(0)]\ensuremath{\rightsquigarrow} S_{{\scriptscriptstyle LU}}^{\mathsf{T}}H_{{\scriptscriptstyle LU}}^{-1}S_{{\scriptscriptstyle LU}}+S_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{{\scriptscriptstyle \textnormal{ST}}}\\
2[\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n\mid\theta})-\mathcal{\ell}_{n}^{\ast}(0)]\ensuremath{\rightsquigarrow} S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}+S_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{{\scriptscriptstyle \textnormal{ST}}}
\end{gather}
whence the result follows by subtracting (ref) from (ref)
and noting that
\[
H_{{\scriptscriptstyle \textnormal{LU}}}^{-1/2}\Theta(\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}\Theta)^{-1}\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1/2}+H_{{\scriptscriptstyle \textnormal{LU}}}^{1/2}\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1}\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{1/2}=I_{pq}
\]
since the columns of $H_{{\scriptscriptstyle \textnormal{LU}}}^{-1/2}\Theta$ and $H_{{\scriptscriptstyle \textnormal{LU}}}^{1/2}\Theta_{\perp}$
are mutually orthogonal, and collectively span the whole of $\mathbb{R}^{pq}$.
\textbf{({\romannumeral 4}).} The same argument as which yielded (ref)
also gives
\begin{equation}
2[\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n\mid\theta,\gamma})-\mathcal{\ell}_{n}^{\ast}(0)]\ensuremath{\rightsquigarrow} S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Pi_{\perp}(\Pi_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Pi_{\perp})^{-1}\Pi_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}+S_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{{\scriptscriptstyle \textnormal{ST}}}
\end{equation}
so that subtracting (ref) from (ref), and recalling
$\Pi_{\perp}=\Theta_{\perp}\Xi$, yields
\begin{align*}
2[\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n\mid\theta})-\mathcal{\ell}^{\ast}_{n}(\hat{\varphi}_{n\mid\theta,\gamma})] & \ensuremath{\rightsquigarrow} S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}-S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Pi_{\perp}(\Pi_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Pi_{\perp})^{-1}\Pi_{\perp}^{\mathsf{T}})S_{{\scriptscriptstyle \textnormal{LU}}}\\
& =(\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}})^{\mathsf{T}}[H_{\Theta,\perp}^{-1}-\Xi(\Xi^{\mathsf{T}}H_{\Theta,\perp}\Xi)^{-1}\Xi^{\mathsf{T}}](\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}})\\
& =(H_{\Theta,\perp}^{-1/2}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}})^{\mathsf{T}}[I_{qr}-H_{\Theta,\perp}^{1/2}\Xi(\Xi^{\mathsf{T}}H_{\Theta,\perp}\Xi)^{-1}\Xi^{\mathsf{T}}H_{\Theta,\perp}^{1/2}](H_{\Theta,\perp}^{-1/2}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}).
\end{align*}
proof[Proof of (ref)]
Recall the definitions of $\largedec R_{n}=[\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}},\largedec R_{n,{\scriptscriptstyle \textnormal{ST}}}]$ and
$\largedec L_{n}=[\largedec L_{n,{\scriptscriptstyle \textnormal{LU}}},\largedec L_{n,{\scriptscriptstyle \textnormal{ST}}}]$ given at the beginning of this appendix.
Since $I_{kp}=\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\largedec L_{n,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}+\largedec R_{n,{\scriptscriptstyle \textnormal{ST}}}\largedec L_{n,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}$,
we may write
\[
\tilde{\boldsymbol{\Phi}}_{n}=\boldsymbol{\Phi}_{n}+[(\tilde{\boldsymbol{\Phi}}_{n}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}]\largedec L_{n,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}+[(\tilde{\boldsymbol{\Phi}}_{n}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{ST}}}]\largedec L_{n,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\eqqcolon\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{LU}}}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}}.
\]
Since $\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{LU}}}=o_{p}(1)$ and $\boldsymbol{\Phi}_{n}\ensuremath{\rightarrow}\boldsymbol{\Phi}_{0}$,
we have $\smlabs{\lambda_{q+1}(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}})}<\smlabs{\lambda_{q}(\boldsymbol{\Phi}_{n})}$
w.p.a.1, and so by (ref)(ref)
\begin{align}
A(\tilde{\boldsymbol{\Phi}}_{n})-A(\boldsymbol{\Phi}_{n}) & =A(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle ST}}+\tilde{\Delta}_{n,{\scriptscriptstyle LU}})-A(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle ST}})
\end{align}
w.p.a.1. Since $A(\cdot)$ is smooth, a second-order Taylor series
expansion and (ref)(ref) yield
\begin{align}
& \operatorname{vec}\{A(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle ST}}+\tilde{\Delta}_{n,{\scriptscriptstyle LU}})-A(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle ST}})\}\nonumber \\
& \qquad\qquad\qquad\qquad=[J_{A}(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}})+o_{p}(1)]\operatorname{vec}\{\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{LU}}}\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}})\}\nonumber \\
& \qquad\qquad\qquad\qquad=[J_{A}(\boldsymbol{\Phi}_{0})+o_{p}(1)]\operatorname{vec}\{\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{LU}}}\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\}
\end{align}
where the second equality holds w.p.a.1, and follows from the continuity
of $J_{A}$ ((ref)(ref)), $\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}}=\boldsymbol{\Phi}_{0}+o_{p}(1)$,
and $\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n}+\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{ST}}})=\largedec R_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n})=\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}$
(w.p.a.1, as implied by (ref)(ref)).
Finally, since
\begin{equation}
\tilde{\Delta}_{n,{\scriptscriptstyle \textnormal{LU}}}\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}=[(\tilde{\boldsymbol{\Phi}}_{n}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}]\largedec L_{n,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}=(\tilde{\boldsymbol{\Phi}}_{n}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}},
\end{equation}
the first part of (ref) follows from (ref)--(ref).
The proof of the second part is analogous.
Limiting experiments
The assumptions (ref) and (ref) are maintained throughout
this appendix. Recall the re-parametrisation given in (ref)
above, which in view of (ref) we can equivalently write
as
subequations\begin{align}
\boldsymbol{\pi} & \coloneqq n\operatorname{vec}\begin{bmatrix}A[\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi)\largedec L_{n}^{\mathsf{T}}]-A(\boldsymbol{\Phi}_{n})\\
\Lambda_{{\scriptscriptstyle LU}}[\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi)\largedec L_{n}^{\mathsf{T}}]-\Lambda_{{\scriptscriptstyle LU}}(\boldsymbol{\Phi}_{n})
\end{bmatrix}\\
f & \coloneqq n^{1/2}\varphi_{{\scriptscriptstyle ST}}.
\end{align}
Under (ref), $R_{n,{\scriptscriptstyle \textnormal{ST}}}$ and $\Lambda_{n,{\scriptscriptstyle \textnormal{ST}}}$
associated with $\{\boldsymbol{\Phi}_{n}\}\subset\set P$ are constant (see (ref)),
so $\largedec R_{n,{\scriptscriptstyle \textnormal{ST}}}=\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}$ for all $n\in\ensuremath{\mathbb{N}}$, so that in particular
$\varphi_{{\scriptscriptstyle \textnormal{ST}}}=\operatorname{vec}\{(\boldsymbol{\Phi}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{ST}}}\}$. Let $\psi_{n}(\varphi)$
denote the smooth mapping $\varphi\ensuremath{\mapsto}(\boldsymbol{\pi},f)$ implied by (ref),
which has domain $\ensuremath{\mathcal{P}}_{n}$ (defined in (ref) above) and
$\psi_{n}(0)=0$ for all $n\in\ensuremath{\mathbb{N}}$.
lem\begin{enumerate}
• There exists an $n_{0}\in\ensuremath{\mathbb{N}}$ and
an open neighbourhood $N\subset\mathbb{R}^{kp^{2}}$ of the origin, such
that $\psi_{n}$ is a smooth diffeomorphism on $N$, for all $n\geq n_{0}$
• Let ${\cal K}\subset\mathbb{R}^{kp^{2}}$ be any
compact neighbourhood of zero. Then there exists an $n_{1}\geq n_{0}$
such that $\psi_{n}^{-1}$ is well defined (and smooth) on ${\cal K}$,
for all $n\geq n_{1}$. Moreover, for any $(\boldsymbol{\pi},f)\in{\cal K}$,
$\varphi_{n}\coloneqq\psi_{n}^{-1}(\boldsymbol{\pi},f)$ is such that $D_{n}\varphi_{n}=O(1)$.
\end{enumerate}
Thus so long as we restrict attention to $\varphi\in N$, we may equivalently
parametrise the model in terms of $(\boldsymbol{\pi},f)$. For a given $(\boldsymbol{\pi},f)\in\mathbb{R}^{kp^{2}}$,
$\psi_{n}^{-1}$ is well-defined (and smooth) at $(\boldsymbol{\pi},f)$ for
all $n$ sufficiently large, in which case we shall define (with a
slight abuse of notation) $\mathcal{\ell}_{n}(\boldsymbol{\pi},f)\coloneqq\mathcal{\ell}_{n}(\varphi,\Sigma)$,
where $\varphi=\psi_{n}^{-1}(\boldsymbol{\pi},f)$; and set $\mathcal{\ell}_{n}(\boldsymbol{\pi},f)\coloneqq-\infty$
otherwise (to simplify arguments, we treat $\Sigma$ as known here.)
To state our next result, recall the definitions of $S_{\boldsymbol{\pi}}$ and
$H_{\boldsymbol{\pi}}$ given in (ref).
lemJointly over any finite collection of $(\boldsymbol{\pi},f)\in\mathbb{R}^{kp^{2}}$,
\[
\mathcal{\ell}_{n}(\boldsymbol{\pi},f)-\mathcal{\ell}_{n}(0,0)\ensuremath{\rightsquigarrow}[S_{\boldsymbol{\pi}}^{\mathsf{T}}\boldsymbol{\pi}-\tfrac{1}{2}\boldsymbol{\pi}^{\mathsf{T}}H_{\boldsymbol{\pi}}\boldsymbol{\pi}]+[S_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}f-\tfrac{1}{2}f^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{ST}}}f].
\]
We next show that, up to the term depending on $f$, the preceding
is also the limit of the loglikelihood ratio process in a multivariate
predictive regression with a known covariance matrix; recall (ref)
given in (ref).
lemSuppose that $\{y_{{\scriptscriptstyle \textnormal{PR}},t}\}$ and $\{z_{{\scriptscriptstyle \textnormal{PR}} t}\}$
are generated under (ref), and that $\xi_{t}=[\begin{smallmatrix}\xi_{yt}\\
\xi_{zt}
\end{smallmatrix}]\ensuremath{\sim_{\ensuremath{\textnormal{i.i.d.}}}} N[0,\Omega]$ with $\Omega=K\Sigma K^{\mathsf{T}}$. Then for
\[
\boldsymbol{\pi}=n\operatorname{vec}\begin{bmatrix}A-A(\boldsymbol{\Phi}_{n})\\
\Lambda-\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n})
\end{bmatrix},
\]
we have, jointly over any finite collection of $\boldsymbol{\pi}\in\mathbb{R}^{pq}$
\[
\mathcal{\ell}_{n,{\scriptscriptstyle \textnormal{PR}}}(\boldsymbol{\pi})-\mathcal{\ell}_{n,{\scriptscriptstyle \textnormal{PR}}}(0)\ensuremath{\rightsquigarrow} S_{\boldsymbol{\pi}}^{\mathsf{T}}\boldsymbol{\pi}-\tfrac{1}{2}\boldsymbol{\pi}^{\mathsf{T}}H_{\boldsymbol{\pi}}\boldsymbol{\pi},
\]
where $\mathcal{\ell}_{n,{\scriptscriptstyle \textnormal{PR}}}(\boldsymbol{\pi})$ is the loglikelihood defined in (ref).
Finally, we show that when (the entirety of) $\boldsymbol{\Phi}$ is in unknown,
and the model is estimated subject to the constraint (ref),
then the limit of the concentrated loglikelihood ratio process
is asymptotically identical to that of the predictive regression,
up to (random) terms that do not depend on $\boldsymbol{\pi}$. Let $\hat{f}_{n\mid\boldsymbol{\pi}}\coloneqq n^{1/2}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\boldsymbol{\pi}}$,
where $\hat{\varphi}_{n\mid\boldsymbol{\pi}}$ denotes the maximiser of $\mathcal{\ell}^{\ast}_{n}(\varphi)$
subject to $\varphi$ satisfying (ref).
lemJointly over every finite collection of $\boldsymbol{\pi}\in\mathbb{R}^{pq}$,
\[
\mathcal{\ell}_{n}(\boldsymbol{\pi},\hat{f}_{n\mid\boldsymbol{\pi}})-\mathcal{\ell}_{n}(0)\ensuremath{\rightsquigarrow} S_{\boldsymbol{\pi}}^{\mathsf{T}}\boldsymbol{\pi}-\tfrac{1}{2}\boldsymbol{\pi}^{\mathsf{T}}H_{\boldsymbol{\pi}}\boldsymbol{\pi}+S_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{{\scriptscriptstyle \textnormal{ST}}}
\]
proof[Proof of (ref)]
Consider the mapping $\Psi$ and the permutation matrix $M\in\mathbb{R}^{pq\times pq}$
such that
\begin{align}
\Psi(\boldsymbol{\Phi}) & \coloneqq\begin{bmatrix}\operatorname{vec} A(\boldsymbol{\Phi})\\
\operatorname{vec}\Lambda_{{\scriptscriptstyle LU}}(\boldsymbol{\Phi})\\
\operatorname{vec}\boldsymbol{\Phi}\largedec R_{0,{\scriptscriptstyle ST}}
\end{bmatrix} & {\cal M}\Psi(\boldsymbol{\Phi})\coloneqq\begin{bmatrix}M & 0\\
0 & I_{\#{\scriptscriptstyle ST}}
\end{bmatrix}\Psi(\boldsymbol{\Phi}) & =\begin{bmatrix}\operatorname{vec}\begin{bmatrix}A(\boldsymbol{\Phi})\\
\Lambda_{{\scriptscriptstyle LU}}(\boldsymbol{\Phi})
\end{bmatrix}\\
\operatorname{vec}\boldsymbol{\Phi}\largedec R_{0,{\scriptscriptstyle ST}}
\end{bmatrix}
\end{align}
By Lemmas (ref) and (ref), $\Psi$
is smooth and at $\boldsymbol{\Phi}=\boldsymbol{\Phi}_{0}$ has first differential
\begin{equation}
\ensuremath{\mathrm{d}}\Psi=\begin{bmatrix}J_{A}(\boldsymbol{\Phi}_{0}) & 0\\
J_{\Lambda}(\boldsymbol{\Phi}_{0}) & 0\\
0 & I_{\#{\scriptscriptstyle ST}}
\end{bmatrix}\left(\begin{bmatrix}\largedec R_{0,{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\\
\largedec R_{0,{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}
\end{bmatrix}\otimes I_{p}\right)\operatorname{vec}(\ensuremath{\mathrm{d}}\boldsymbol{\Phi}).
\end{equation}
The Jacobian on the r.h.s.\ is invertible by (ref)(ref)
and (ref) (for the latter, since $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{0})=I_{q}$).
Thus by the inverse mapping theorem, there is an open neighbourhood
$N_{\set P}\subset\set P$ of $\boldsymbol{\Phi}_{0}$ on which $\Psi$ has a smooth
inverse.
Now let $\tau_{n}(\varphi)\coloneqq\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi)\largedec L_{n}^{\mathsf{T}}$,
which converges (uniformly on compacta) to a linear and invertible
mapping $\tau_{0}(\varphi)$ for which $\tau_{0}(0)=\boldsymbol{\Phi}_{0}$. Hence there
exists an $n_{0}\in\ensuremath{\mathbb{N}}$ and a (fixed) open neighbourhood $N\subset\ensuremath{\mathcal{P}}_{n}$
of zero such that $\tau_{n}(N)\subset N_{\set P}$, for all $n\geq n_{0}$,
with $\tau_{n}$ being invertible on $N$. By composition, the sequence
of maps defined by
\begin{equation}
D_{n}^{-1}\psi_{n}(\varphi)={\cal M}\{\Psi[\tau_{n}(\varphi)]-\Psi(\boldsymbol{\Phi}_{n})\}
\end{equation}
is smooth and invertible on $N$, for all $n\geq0$, and has a smooth
inverse there; hence part (ref) holds. Finally,
since the image of $N$ under the r.h.s.\ must itself be an open
neighbourhood of zero, and
\begin{equation}
\begin{bmatrix}\boldsymbol{\pi}\\
f
\end{bmatrix}=\psi_{n}(\varphi)=D_{n}{\cal M}\{\Psi[\tau_{n}(\varphi)]-\Psi(\boldsymbol{\Phi}_{n})\},
\end{equation}
we may deduce that for any compact neighbourhood ${\cal K}$ of zero,
there is an $n_{1}\geq n_{0}$ such that the inverse $\psi_{n}^{-1}$
is well-defined and smooth for all $(\boldsymbol{\pi},f)\in{\cal K}$, for all
$n\geq n_{1}$. Finally, to show that the $\varphi_{n}\coloneqq\psi_{n}^{-1}(\boldsymbol{\pi},f)$
has $D_{n}\varphi_{n}=O(1)$, we note that since the r.h.s.\ of (ref)
is (locally to zero) a diffeomorphism, which itself equals zero at
$\varphi=0$, the fact that
\[
{\cal M}\{\Psi[\tau_{n}(\varphi_{n})]-\Psi(\boldsymbol{\Phi}_{n})\}=D_{n}^{-1}\begin{bmatrix}\boldsymbol{\pi}\\
f
\end{bmatrix}\ensuremath{\rightarrow}0
\]
must imply that $\varphi_{n}\ensuremath{\rightarrow}0$. Hence follows from (ref)
and (ref) that, by a Taylor expansion of (ref)
around $\varphi=0$,
\[
\begin{bmatrix}MJ+o_{p}(1) & 0\\
0 & I_{\#{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\begin{bmatrix}\varphi_{n,{\scriptscriptstyle \textnormal{LU}}}\\
\varphi_{n,{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}=D_{n}^{-1}\begin{bmatrix}\boldsymbol{\pi}\\
f
\end{bmatrix}
\]
whence $D_{n}\varphi_{n}=O(1)$ as claimed. Thus part (ref)
holds.
proof[Proof of (ref)]
In view of (ref), we may take $n$ sufficiently large
such that $\psi_{n}^{-1}$ is well defined at $(\boldsymbol{\pi},f)$. Let $\varphi_{n}^{\ast}\coloneqq\psi_{n}^{-1}(\boldsymbol{\pi},f)=o(1)$,
$\boldsymbol{\Phi}_{n}^{\ast}\coloneqq\boldsymbol{\Phi}_{n}+\operatorname{vec}^{-1}(\varphi_{n}^{\ast})\largedec L_{n}^{\mathsf{T}}$,
and $M\in\mathbb{R}^{pq\times pq}$ be as in (ref). Then
by (ref), for $J\coloneqq J(\boldsymbol{\Phi}_{0})$
\begin{equation}
\boldsymbol{\pi}=n\operatorname{vec}\begin{bmatrix}A(\boldsymbol{\Phi}_{n}^{\ast})-A(\boldsymbol{\Phi}_{n})\\
\Lambda_{{\scriptscriptstyle LU}}(\boldsymbol{\Phi}_{n}^{\ast})-\Lambda_{{\scriptscriptstyle LU}}(\boldsymbol{\Phi}_{n})
\end{bmatrix}=[MJ+o(1)]n\varphi_{n,{\scriptscriptstyle LU}}^{\ast}
\end{equation}
where by (ref) and the definition of $M$,
\[
MJ=M\begin{bmatrix}I_{q}\otimes{\cal J}\\
I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}
\end{bmatrix}=I_{q}\otimes\begin{bmatrix}{\cal J}\\
L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}
\end{bmatrix}\eqqcolon I_{q}\otimes K
\]
for ${\cal J}$ and $K$ as defined in (ref). Noting also
that $n^{1/2}\varphi_{n,{\scriptscriptstyle \textnormal{ST}}}^{\ast}=f$, it follows from (ref)
and (ref) that
\begin{align*}
\mathcal{\ell}_{n}(\boldsymbol{\pi},f)-\mathcal{\ell}_{n}(0,0) & =\mathcal{\ell}_{n}(\varphi_{n,{\scriptscriptstyle LU}}^{\ast},\varphi_{n,{\scriptscriptstyle ST}}^{\ast})-\mathcal{\ell}_{n}(0,0)\\
& =S_{n}^{\mathsf{T}}(D_{n}\varphi_{n}^{\ast})-\tfrac{1}{2}(D_{n}\varphi_{n}^{\ast})^{\mathsf{T}}H_{n}(D_{n}\varphi_{n}^{\ast})\\
& \ensuremath{\rightsquigarrow}[S_{{\scriptscriptstyle LU}}^{\mathsf{T}}(MJ)^{-1}\boldsymbol{\pi}-\tfrac{1}{2}\boldsymbol{\pi}^{\mathsf{T}}[(MJ)^{-1}]^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}(MJ)^{-1}\boldsymbol{\pi}]+[S_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}f-\tfrac{1}{2}f^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{ST}}}f].
\end{align*}
To complete the proof, we note that
\begin{align*}
[(MJ)^{-1}]^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}} & =(I_{q}\otimes K^{-1})^{\mathsf{T}}\int_{0}^{1}[\bar{Z}_{C}(r)\otimes\Sigma^{-1}\ensuremath{\mathrm{d}} E(r)]\\
& =\int_{0}^{1}[\bar{Z}_{C}(r)\otimes(K\Sigma K^{\mathsf{T}})^{-1/2}\ensuremath{\mathrm{d}} W(r)]=S_{\boldsymbol{\pi}}
\end{align*}
where we have used that $E(s)=\Sigma^{-1/2}W(s)$ and,
\begin{align*}
[(MJ)^{-1}]^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}(MJ)^{-1} & =(I_{q}\otimes K^{-1})^{\mathsf{T}}\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\otimes\Sigma^{-1}\right)(I_{q}\otimes K^{-1})\\
& =\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\otimes(K\Sigma K^{\mathsf{T}})^{-1}=H_{\boldsymbol{\pi}}.\qedhere
\end{align*}
proof[Proof of (ref)]
Letting $\boldsymbol{\Pi}=[\begin{smallmatrix}A_{{\scriptscriptstyle \textnormal{PR}}}\\
\Lambda_{{\scriptscriptstyle \textnormal{PR}}}
\end{smallmatrix}]$ and noting that $\boldsymbol{\pi}=n\operatorname{vec}(\boldsymbol{\Pi}-[\begin{smallmatrix}A(\boldsymbol{\Phi}_{n})\\
\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{n})
\end{smallmatrix}])$, we have
\[
\mathcal{\ell}_{n}^{{\scriptscriptstyle \textnormal{PR}}}(\boldsymbol{\pi})=K_{n}-\frac{1}{2}\sum_{t=1}^{n}\smlnorm{x_{t}-\boldsymbol{\Pi} z_{t-1}}_{\Omega^{-1}},
\]
where $K_{n}\coloneqq-\frac{n}{2}\log(2\pi\log\det\Omega)$. It then
follows by exactly the same arguments as were used in the proof of
(ref) that
\[
\mathcal{\ell}_{n,{\scriptscriptstyle \textnormal{PR}}}(\boldsymbol{\pi})-\mathcal{\ell}_{n,{\scriptscriptstyle \textnormal{PR}}}(0)=S_{n,{\scriptscriptstyle \textnormal{PR}}}^{\mathsf{T}}\boldsymbol{\pi}-\tfrac{1}{2}\boldsymbol{\pi}^{\mathsf{T}}H_{n,{\scriptscriptstyle \textnormal{PR}}}\boldsymbol{\pi}
\]
where
\begin{align*}
S_{n,{\scriptscriptstyle PR}} & =\frac{1}{n}\sum_{t=1}^{n}(\bar{z}_{t-1}\otimes\Omega^{-1/2}\eta_{t}) & H_{n,{\scriptscriptstyle PR}} & =\frac{1}{n}\sum_{t=1}^{n}(\bar{z}_{t-1}\bar{z}_{t-1}^{\mathsf{T}}\otimes\Omega^{-1}).
\end{align*}
Under (ref), it follows by (ref)(ref)
that $n^{-1/2}\bar{z}_{\smlfloor{nr}}\ensuremath{\rightsquigarrow}\bar{Z}_{C,{\scriptscriptstyle \textnormal{PR}}}(r)$ on $D[0,1]$,
where $\bar{Z}_{C,{\scriptscriptstyle \textnormal{PR}}}(r)$ denotes the residual from the projection
of
\begin{equation}
Z_{C,{\scriptscriptstyle PR}}(r)\coloneqq\int_{0}^{r}\mathrm{e}^{C(r-s)}\Omega_{zz}^{1/2}\ensuremath{\mathrm{d}} W(s)
\end{equation}
on a constant and a linear trend, and we have partitioned $\Omega=[\begin{smallmatrix}\Omega_{yy} & \Omega_{yz}\\
\Omega_{zy} & \Omega_{zz}
\end{smallmatrix}]$ conformably with $\xi_{t}=[\begin{smallmatrix}\xi_{yt}\\
\xi_{zt}
\end{smallmatrix}]$. Then by the continuous mapping theorem and the same arguments as
used in the proof of (ref)(ref),
\begin{align}
S_{n,{\scriptscriptstyle PR}} & \ensuremath{\rightsquigarrow}\int_{0}^{1}[\bar{Z}_{C,{\scriptscriptstyle PR}}(r)\otimes\Omega^{-1/2}W(r)]\ensuremath{\,\ensuremath{\mathrm{d}}} r & H_{n,{\scriptscriptstyle PR}} & \ensuremath{\rightsquigarrow}\int\bar{Z}_{C,{\scriptscriptstyle \textnormal{PR}}}\bar{Z}_{C,{\scriptscriptstyle \textnormal{PR}}}^{\mathsf{T}}\otimes\Omega^{-1}.
\end{align}
Thus we can bring (ref) into agreement with (ref),
and the limits on the r.h.s.\ of (ref) with (ref),
by setting
\[
\Omega=\begin{bmatrix}\Omega_{yy} & \Omega_{yz}\\
\Omega_{zy} & \Omega_{zz}
\end{bmatrix}=\begin{bmatrix}{\cal J}\Sigma{\cal J}^{\mathsf{T}} & {\cal J}\Sigma L_{{\scriptscriptstyle \textnormal{LU}}}\\
L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Sigma{\cal J}^{\mathsf{T}} & L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Sigma L_{{\scriptscriptstyle \textnormal{LU}}}
\end{bmatrix}=K\Sigma K^{\mathsf{T}}.\qedhere
\]
proof[Proof of (ref)]
We first show that $D_{n}\hat{\varphi}_{n\mid\boldsymbol{\pi}}=O_{p}(1)$. By (ref),
for all $n$ sufficiently large, there exists a (deterministic) sequence
$\varphi_{n\mid\boldsymbol{\pi}}\in\ensuremath{\mathcal{P}}_{n}$ with $D_{n}\varphi_{n\mid\boldsymbol{\pi}}=O(1)$,
such that (ref) holds at $\varphi=\varphi_{n\mid\boldsymbol{\pi}}$. It follows
from (ref) that for each $\epsilon>0$, there exists
an $N<\infty$ such that
\[
\limsup_{n\ensuremath{\rightarrow}\infty}\ensuremath{\mathbb{P}}\{\mathcal{\ell}_{n}^{\ast}(\varphi_{n\mid\boldsymbol{\pi}})-\mathcal{\ell}_{n}^{\ast}(0)<-N\}<\epsilon/2.
\]
On the other hand, adapting the argument given in the proof of (ref),
we may also choose $M<\infty$ sufficiently large such that
\begin{align*}
\ensuremath{\mathbb{P}}\left\{ \sup_{\{\varphi\in\mathcal{P}_{n}\mid\smlnorm{D_{n}\varphi}\geq M\}}[\mathcal{\ell}^{\ast}_{n}(\varphi)-\mathcal{\ell}^{\ast}_{n}(0)]<-2N\right\} & \ge\ensuremath{\mathbb{P}}\left\{ M[\smlnorm{S_{n}}-\tfrac{1}{2}\lambda_{\min}(H_{n})M]\leq-2N\right\} \\
& >1-\epsilon/2
\end{align*}
for all $n$ sufficiently large. Deduce that with probability at least
$1-\epsilon$, $\mathcal{\ell}_{n}^{\ast}(\varphi_{n\mid\boldsymbol{\pi}})$ must strictly
exceed $\mathcal{\ell}_{n}(\varphi)$ over all $\varphi\in\ensuremath{\mathcal{P}}_{n}$ with $\smlnorm{D_{n}\varphi}\geq M$;
it follows that the constrained maximiser $\hat{\varphi}_{n\mid\boldsymbol{\pi}}$
must have $\smlnorm{D_{n}\hat{\varphi}_{n\mid\boldsymbol{\pi}}}<M$. Deduce that $D_{n}\hat{\varphi}_{n\mid\boldsymbol{\pi}}=O_{p}(1)$
as claimed.
Now it follows from (ref) and (ref) that, at $\varphi=\hat{\varphi}_{n\mid\boldsymbol{\pi}}$,
\[
\begin{bmatrix}\ensuremath{\mathrm{d}}\boldsymbol{\pi}\\
\ensuremath{\mathrm{d}} f
\end{bmatrix}=\begin{bmatrix}MJ+o_{p}(1) & 0\\
0 & I_{\#{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}D_{n}\ensuremath{\mathrm{d}}\varphi
\]
and hence
\[
D_{n}\ensuremath{\mathrm{d}}\varphi=\begin{bmatrix}(MJ)^{-1}+o_{p}(1) & 0\\
0 & I_{\#{\scriptscriptstyle \textnormal{ST}}}
\end{bmatrix}\begin{bmatrix}\ensuremath{\mathrm{d}}\boldsymbol{\pi}\\
\ensuremath{\mathrm{d}} f
\end{bmatrix}
\]
at $(\boldsymbol{\pi},\hat{f}_{n\mid\boldsymbol{\pi}})$. Thus
\begin{equation}
n\hat{\varphi}_{n,{\scriptscriptstyle LU}}=[(MJ)^{-1}+o_{p}(1)]\boldsymbol{\pi},
\end{equation}
and since $\hat{f}_{n\mid\boldsymbol{\pi}}$ must satisfy the first-order conditions
for a maximum, we have from (ref) that
\begin{align*}
\nabla_{f}\mathcal{\ell}_{n}(\boldsymbol{\pi},\hat{f}_{n\mid\boldsymbol{\pi}}) & =\nabla_{f}[S_{n}^{\mathsf{T}}(D_{n}\varphi)-\tfrac{1}{2}(D_{n}\varphi)^{\mathsf{T}}H_{n}(D_{n}\varphi)]_{f=\hat{f}_{n\mid\boldsymbol{\pi}}}\\
& =S_{n,{\scriptscriptstyle ST}}-(n\hat{\varphi}_{n,{\scriptscriptstyle LU}\mid\boldsymbol{\pi}})^{\mathsf{T}}H_{n,{\scriptscriptstyle LS}}-H_{n,{\scriptscriptstyle ST}}n^{1/2}\hat{\varphi}_{n,{\scriptscriptstyle ST}\mid\boldsymbol{\pi}}
\end{align*}
whence
\begin{equation}
n^{1/2}\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{ST}}\mid\boldsymbol{\pi}}=H_{n,{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{n,{\scriptscriptstyle \textnormal{ST}}}+o_{p}(1)\ensuremath{\rightsquigarrow} H_{{\scriptscriptstyle \textnormal{ST}}}^{-1}S_{{\scriptscriptstyle \textnormal{ST}}}.
\end{equation}
Thus, in view of (ref) and (ref), the weak
limit of
\begin{align*}
\mathcal{\ell}_{n}(\boldsymbol{\pi},\hat{f}_{n\mid\boldsymbol{\pi}})-\mathcal{\ell}_{n}(0) & =\mathcal{\ell}_{n}^{\ast}(\hat{\varphi}_{n\mid\boldsymbol{\pi}})-\mathcal{\ell}_{n}^{\ast}(0)=S_{n}^{\mathsf{T}}(D_{n}\hat{\varphi}_{n\mid\pi})-\tfrac{1}{2}(D_{n}\hat{\varphi}_{n\mid\pi})^{\mathsf{T}}H_{n}(D_{n}\hat{\varphi}_{n\mid\pi})
\end{align*}
is as claimed.
Proofs of theorems
proof[Proof of (ref)]
This follows directly from Lemmas (ref)--(ref),
noting in particular that $\mathcal{\ell}^{\ast}_{n}(\boldsymbol{\pi})=\mathcal{\ell}_{n}(\boldsymbol{\pi},\hat{f}_{n\mid\boldsymbol{\pi}})$,
where the latter is as appears in (ref).
proof[Proof of (ref)]
({\romannumeral 1}). In the notation of (ref), $\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}}=\operatorname{vec}\{(\hat{\boldsymbol{\Phi}}_{n}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\}$.
By (ref)(ref)
\begin{align*}
n\operatorname{vec}\{(\hat{\boldsymbol{\Phi}}_{n}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle LU}}\} & \ensuremath{\rightsquigarrow}\left[\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\right)^{-1}\otimes I_{p}\right]\int_{0}^{1}[\bar{Z}_{C}(r)\otimes\ensuremath{\mathrm{d}} E(r)]\\
& =\operatorname{vec}\left\{ \int(\ensuremath{\mathrm{d}} E)\bar{Z}_{C}^{\mathsf{T}}\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\right)^{-1}\right\} ,
\end{align*}
and so by (ref)
\begin{equation}
\begin{bmatrix}\operatorname{vec}\{A(\hat{\boldsymbol{\Phi}}_{n})-A(\boldsymbol{\Phi}_{n})\}\\
\operatorname{vec}\{\Lambda_{{\scriptscriptstyle LU}}(\hat{\boldsymbol{\Phi}}_{n})-\Lambda_{{\scriptscriptstyle LU}}(\boldsymbol{\Phi}_{n})\}
\end{bmatrix}\ensuremath{\rightsquigarrow}\begin{bmatrix}J_{A}(\boldsymbol{\Phi}_{0})\\
J_{\Lambda}(\boldsymbol{\Phi}_{0})
\end{bmatrix}\operatorname{vec}\left\{ \int(\ensuremath{\mathrm{d}} E)\bar{Z}_{C}^{\mathsf{T}}\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\right)^{-1}\right\} .
\end{equation}
Since $\boldsymbol{\Phi}_{n}\ensuremath{\rightarrow}\boldsymbol{\Phi}_{0}$ with $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{0})=I_{q}$
under (ref), we have by (ref) that
\begin{equation}
\begin{bmatrix}J_{A}(\boldsymbol{\Phi}_{0})\\
J_{\Lambda}(\boldsymbol{\Phi}_{0})
\end{bmatrix}=\begin{bmatrix}I_{q}\otimes\beta^{\mathsf{T}}R_{{\scriptscriptstyle ST}}(I_{kp-q}-\Lambda_{{\scriptscriptstyle ST}})^{-1}L_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\\
I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}
\end{bmatrix}.
\end{equation}
The result then follows from (ref) and (ref),
by reversing the vectorisation.
\textbf{({\romannumeral 2}).} In the notation of (ref), maximising
$\mathcal{\ell}_{n}^{\ast}(\boldsymbol{\Phi})$ subject to $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}}=I_{q}+C/n$
corresponds to maximising $\mathcal{\ell}_{n}(\varphi)$ subject to $\theta_{n}(\varphi)=0$.
Thus $\hat{\varphi}_{n,{\scriptscriptstyle \textnormal{LU}}\mid\theta}=\operatorname{vec}\{(\hat{\boldsymbol{\Phi}}_{n\mid\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}}}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\}$,
and so by (ref)(ref)
\[
n\operatorname{vec}\{(\hat{\boldsymbol{\Phi}}_{n\mid\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}}}-\boldsymbol{\Phi}_{n})\largedec R_{n,{\scriptscriptstyle \textnormal{LU}}}\}\ensuremath{\rightsquigarrow}\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}
\]
where $\Theta_{\perp}=I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}},\perp}$. Hence by (ref),
\[
\operatorname{vec}\{A(\hat{\boldsymbol{\Phi}}_{n\mid\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}}})-A(\boldsymbol{\Phi}_{n})\}\ensuremath{\rightsquigarrow} J_{A}(\boldsymbol{\Phi}_{0})\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}.
\]
To determine the distribution of the r.h.s., we note that
\begin{equation}
\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}=\int_{0}^{1}[\bar{Z}_{C}(r)\otimes L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\Sigma^{-1}\ensuremath{\mathrm{d}} E(r)]\eqqcolon\int_{0}^{1}[\bar{Z}_{C}(r)\otimes\ensuremath{\mathrm{d}} U(r)].
\end{equation}
Recall that $\bar{Z}_{C}$ is a function only of $Z_{C}$, which from
(ref) is given by
\begin{equation}
Z_{C}(r)=\int_{0}^{r}\mathrm{e}^{C(r-s)}L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\ensuremath{\mathrm{d}} E(s)\eqqcolon\int_{0}^{r}\mathrm{e}^{C(r-s)}\ensuremath{\mathrm{d}} V(s).
\end{equation}
$(U,V)=(L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\Sigma^{-1}E,L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}E)$ is
a pair of vector Brownian motions, with covariance
\[
\ensuremath{\mathbb{E}} U(1)V(1)^{\mathsf{T}}=L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\Sigma^{-1}\ensuremath{\mathbb{E}}[E(1)E(1)^{\mathsf{T}}]L_{{\scriptscriptstyle \textnormal{LU}}}=L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}L_{{\scriptscriptstyle \textnormal{LU}}}=0;
\]
whence $U$ and $V$ are independent. In particular, we have from
(ref) that $U$ is independent of $\bar{Z}_{C}$. This,
combined with the fact that
\begin{align*}
J_{A}(\boldsymbol{\Phi}_{0})\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1} & =\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\right)^{-1}\otimes\mathcal{J}L_{{\scriptscriptstyle \textnormal{LU}},\perp}(L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\Sigma^{-1}L_{{\scriptscriptstyle \textnormal{LU}},\perp})^{-1}
\end{align*}
depends only on $\bar{Z}_{C}$, implies $J_{A}(\boldsymbol{\Phi}_{0})\Theta_{\perp}(\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp})^{-1}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}$
is mixed normal with variance
\[
\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\right)^{-1}\otimes\mathcal{J}L_{{\scriptscriptstyle \textnormal{LU}},\perp}(L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\Sigma^{-1}L_{{\scriptscriptstyle \textnormal{LU}},\perp})^{-1}L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\mathcal{J}^{\mathsf{T}},
\]
which proves (ref).
Finally, note that the preceding holds for any choice of $L_{{\scriptscriptstyle \textnormal{LU}},\perp}\in\mathbb{R}^{p\times r}$
having full column rank and $L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}L_{{\scriptscriptstyle \textnormal{LU}}}=0$. Let
$\alpha\coloneqq\Phi_{0}(1)\beta(\beta^{\mathsf{T}}\beta)^{-1}\in\mathbb{R}^{p\times r}$,
where $\Phi_{0}(1)\coloneqq\lim_{n\ensuremath{\rightarrow}\infty}\Phi_{n}(1)$; then
\[
L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\alpha=L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Phi_{0}(1)\beta(\beta^{\mathsf{T}}\beta)^{-1}=0
\]
by (ref) with $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}=\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi}_{0})=I_{q}$.
Further, $\operatorname{rk}\alpha=r$ since $\operatorname{sp}\Phi_{0}(1)=\operatorname{sp}\beta$, and
thus we may indeed choose $L_{{\scriptscriptstyle \textnormal{LU}},\perp}=\alpha$. In this case,
\begin{align*}
\mathcal{J}L_{{\scriptscriptstyle \textnormal{LU}},\perp} & =\beta^{\mathsf{T}}R_{{\scriptscriptstyle \textnormal{ST}}}(I_{kp-q}-\Lambda_{{\scriptscriptstyle \textnormal{ST}}})^{-1}L_{{\scriptscriptstyle \textnormal{ST}}}^{\mathsf{T}}\Phi_{0}(1)\beta(\beta^{\mathsf{T}}\beta)^{-1}=_{(1)}\beta^{\mathsf{T}}\beta(\beta^{\mathsf{T}}\beta)^{-1}=I_{r},
\end{align*}
where $=_{(1)}$ follows from (ref) above. Thus (ref)
is proved.
proof[Proof of (ref)]
We first prove (ref). In the notation of (ref),
$\mathcal{LR}_{n}(\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}})=2[\mathcal{\ell}_{n}^{\ast}(\hat{\varphi}_{n})-\mathcal{\ell}_{n}^{\ast}(\hat{\varphi}_{n\mid\theta})]$.
By (ref)(ref),
\[
\mathcal{LR}_{n}(\Lambda_{n,{\scriptscriptstyle \textnormal{LU}}})\ensuremath{\rightsquigarrow} S_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}\Theta(\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}\Theta)^{-1}\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}S_{{\scriptscriptstyle \textnormal{LU}}}\eqqcolon\mathcal{LR},
\]
where $\Theta=I_{q}\otimes L_{{\scriptscriptstyle \textnormal{LU}}}$, $S_{{\scriptscriptstyle \textnormal{LU}}}=\int[\bar{Z}_{C}(r)\otimes\Sigma^{-1}\ensuremath{\mathrm{d}} E]$,
and $H_{{\scriptscriptstyle \textnormal{LU}}}=\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\otimes\Sigma^{-1}$.
To obtain the claimed expression for $\mathcal{LR}$, note that
\[
S_{{\scriptscriptstyle \textnormal{LU}}}=\int[\bar{Z}_{C}(r)\otimes\Sigma^{-1}\ensuremath{\mathrm{d}} E]=\operatorname{vec}\left\{ \Sigma^{-1}\int(\ensuremath{\mathrm{d}} E)\bar{Z}_{C}^{\mathsf{T}}\right\}
\]
and
\[
H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}\Theta(\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}\Theta)^{-1}\Theta^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}^{-1}=\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\right)^{-1}\otimes\Sigma L_{{\scriptscriptstyle \textnormal{LU}}}(L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Sigma L_{{\scriptscriptstyle \textnormal{LU}}})^{-1}L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Sigma
\]
whence, using $\operatorname{vec}(A)^{\mathsf{T}}\operatorname{vec}(B)=\operatorname{tr}(A^{\mathsf{T}}B)$,
\begin{align}
\mathcal{LR} & =\operatorname{tr}\left\{ \Delta^{-1/2}L_{{\scriptscriptstyle LU}}^{\mathsf{T}}\int(\ensuremath{\mathrm{d}} E)\bar{Z}_{C}^{\mathsf{T}}\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\right)^{-1}\int\bar{Z}_{C}(\ensuremath{\mathrm{d}} E)^{\mathsf{T}}L_{{\scriptscriptstyle LU}}\Delta^{-1/2}\right\}
\end{align}
where $\Delta\coloneqq L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}\Sigma L_{{\scriptscriptstyle \textnormal{LU}}}$. To simplify
this further, note that $L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}E$ is a $q$-dimensional
Brownian motion with variance $\Delta$, and so for $W_{\ast}(r)\coloneqq\Delta^{-1/2}L_{{\scriptscriptstyle \textnormal{LU}}}^{\mathsf{T}}E(r)\sim\mathrm{BM}(I_{q})$,
we have
\begin{multline*}
Z_{C}(r)=\int_{0}^{r}\mathrm{e}^{C(r-s)}L_{{\scriptscriptstyle LU}}^{\mathsf{T}}\ensuremath{\mathrm{d}} E(s)=\int_{0}^{r}\mathrm{e}^{C(r-s)}\Delta^{1/2}\ensuremath{\mathrm{d}} W_{\ast}(s)\\
=_{(1)}\Delta^{1/2}\int_{0}^{r}\mathrm{e}^{C_{\ast}(r-s)}\ensuremath{\mathrm{d}} W_{\ast}(s)\eqqcolon\Delta^{1/2}Z_{C_{\ast}}(r)
\end{multline*}
where $C_{\ast}\coloneqq\Delta^{-1/2}C\Delta^{1/2}$ is as in the statement
of the theorem, and $=_{(1)}$ follows from $\mathrm{e}^{C}D=D\mathrm{e}^{D^{-1}CD}$
for any nonsingular $D$. Hence $\bar{Z}_{C}(r)=\Delta^{1/2}\bar{Z}_{C_{\ast}}(r)$,
whereupon (ref) follows from (ref) and the definition
of $W_{\ast}$.
We next prove (ref). Maximisation of $\mathcal{\ell}_{n}^{\ast}(\boldsymbol{\Phi})$
subject to $\Lambda_{{\scriptscriptstyle \textnormal{LU}}}(\boldsymbol{\Phi})=I_{q}+C/n$ and $a_{ij}(\boldsymbol{\Phi})=a_{0}$
corresponds, in the notation of (ref), to maximisation
of $\mathcal{\ell}_{n}(\varphi)$ subject to $\theta_{n}(\varphi)=0$ and $\gamma_{n}(\varphi)=0$.
Therefore by (ref)(ref),
\begin{align*}
\mathcal{LR}_{n}[a_{ij}(\boldsymbol{\Phi}_{n});\Lambda_{n,{\scriptscriptstyle LU}}] & =2[\mathcal{\ell}_{n}(\hat{\varphi}_{n\mid\theta})-\mathcal{\ell}_{n}(\hat{\varphi}_{n\mid\theta,\gamma})]\ensuremath{\rightsquigarrow}(H_{\Theta,\perp}^{-1/2}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle LU}})^{\mathsf{T}}[I_{qr}-\mathcal{Q}](H_{\Theta,\perp}^{-1/2}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle LU}}).
\end{align*}
Recall from (ref) and the subsequent arguments that
\[
\operatorname{vec}\{\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}\}=_{d}\left(\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\otimes L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\Sigma^{-1}L_{{\scriptscriptstyle \textnormal{LU}},\perp}\right)^{1/2}\eta
\]
for $\eta\sim\mathrm{N}[0,I_{qr}]$ independent of $\bar{Z}_{C}$, and
therefore also of
\[
H_{\Theta,\perp}=\Theta_{\perp}^{\mathsf{T}}H_{{\scriptscriptstyle \textnormal{LU}}}\Theta_{\perp}=\int\bar{Z}_{C}\bar{Z}_{C}^{\mathsf{T}}\otimes L_{{\scriptscriptstyle \textnormal{LU}},\perp}^{\mathsf{T}}\Sigma^{-1}L_{{\scriptscriptstyle \textnormal{LU}},\perp}.
\]
Thus $\operatorname{vec}\{H_{\Theta,\perp}^{-1/2}\Theta_{\perp}^{\mathsf{T}}S_{{\scriptscriptstyle \textnormal{LU}}}\}\sim\mathrm{N}[0,I_{qr}]$
is independent of $H_{{\scriptscriptstyle \textnormal{LU}}}$, and therefore also of $\mathcal{Q}$.
The result follows by noting that $H_{\Theta,\perp}^{1/2}\Xi$ has
rank $qr-1$ a.s., whence $I_{qr}-\mathcal{Q}$ projects orthogonally
onto a subspace of dimension 1, a.s.