The exact contents of citations.db main_text.text for this paper — one flattened LaTeX string, title through conclusion, appendix excluded, unmodified except for removing email addresses. This is what our citation measures are computed over.
90,494 characters
Common Trends and Long-Run Identification in Nonlinear Structural VARs
\global\long\def\uwrite#1#2{\underset{#2}{\underbrace{#1}} }
\global\long\def\blw#1{\ensuremath{\underline{#1}}}
\global\long\def\abv#1{\ensuremath{\overline{#1}}}
\global\long\def\vect#1{\mathbf{#1}}
\global\long\def\smlseq#1{\{#1\} }
\global\long\def\seq#1{\left\{ #1\right\} }
\global\long\def\smlsetof#1#2{\{#1\mid#2\} }
\global\long\def\setof#1#2{\left\{ #1\mid#2\right\} }
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long\def\Ellp#1{\ensuremath{\mathcal{L}^{#1}}}
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long\def\abs#1{\ensuremath{\left|#1\right|}}
\global\long\def\smlabs#1{\ensuremath{\lvert#1\rvert}}
\global\long\def\bigabs#1{\ensuremath{\bigl|#1\bigr|}}
\global\long\def\Bigabs#1{\ensuremath{\Bigl|#1\Bigr|}}
\global\long\def\biggabs#1{\ensuremath{\biggl|#1\biggr|}}
\global\long\def\norm#1{\ensuremath{\left\Vert #1\right\Vert }}
\global\long\def\smlnorm#1{\ensuremath{\lVert#1\rVert}}
\global\long\def\bignorm#1{\ensuremath{\bigl\|#1\bigr\|}}
\global\long\def\Bignorm#1{\ensuremath{\Bigl\|#1\Bigr\|}}
\global\long\def\biggnorm#1{\ensuremath{\biggl\|#1\biggr\|}}
\global\long\def\floor#1{\left\lfloor #1\right\rfloor }
\global\long\def\smlfloor#1{\lfloor#1\rfloor}
\global\long\def\ceil#1{\left\lceil #1\right\rceil }
\global\long\def\smlceil#1{\lceil#1\rceil}
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long\def\clsr#1{\ensuremath{\overline{#1}}}
\global\long
\global\long
\global\long
\global\long
\global\long\def\smlinprd#1#2{\ensuremath{\langle#1,#2\rangle}}
\global\long\def\inprd#1#2{\ensuremath{\left\langle #1,#2\right\rangle }}
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long\def\sigf#1{\mathcal{#1}}
\global\long\global\long
\global\long\def\flt#1{\mathcal{#1}}
\global\long\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\def\independenT#1#2{\mathrel{\rlap{$#1#2$}\mkern2mu{#1#2}}}
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long\def\inprobu#1{\ensuremath{\overset{#1}{\ensuremath{\rightarrow}}}}
\global\long
\global\long
\global\long\def\inLp#1{\ensuremath{\overset{\Ellp{#1}}{\ensuremath{\rightarrow}}}}
\global\long
\global\long
\global\long
\global\long\def\wkcu#1{\overset{#1}{\ensuremath{\rightsquigarrow}}}
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long\def\cv#1{\left\langle #1\right\rangle }
\global\long\def\smlcv#1{\langle#1\rangle}
\global\long\def\qv#1{\left[#1\right]}
\global\long\def\smlqv#1{[#1]}
\global\long
\global\long
\global\long
\global\long
\global\long\global\long
\global\long\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\begin{comment}
\global\long\def\objlabel#1#2{\{\}}
\global\long\def\objref#1{\{\}}
\end{comment}
\global\long
\global\long\def\mset#1{\mathcal{#1}}
\global\long\def\largedec#1{\mathbf{#1}}
\global\long
\newcommandx\Ican[1][usedefault, addprefix=\global, 1=]{I_{#1}^{\ast}}
\global\long
\global\long
\global\long
\global\long
\global\long\def\b#1{\boldsymbol{#1}}
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long\def\smldblangle#1{\ensuremath{\savebox{\@brx}{\(\m@th{\langle}\)}
\mathopen{\copy\@brx\kern-0.5\wd\@brx\usebox{\@brx}}#1\savebox{\@brx}{\(\m@th{\rangle}\)}
\mathclose{\copy\@brx\kern-0.5\wd\@brx\usebox{\@brx}}}}
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long\def\set#1{\mathscr{#1}}
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\global\long
\title{Common Trends and Long-Run Identification \\ in Nonlinear Structural
VARs}
\author{\Shortunderstack{James A.\ Duffy\footnotemark[1]{}\\\small{University of Oxford}}\hspace{3cm}
\Shortunderstack{Sophocles Mavroeidis\footnotemark[2]{}\\\small{University
of Oxford}}}
\date{\vspace*{0.3cm}September 2024}
\maketitle
\footnotetext[1]{Department\ of Economics and Corpus Christi College;
\texttt{[email removed]}}
\footnotetext[2]{Department of Economics and University College;
\texttt{[email removed]}}
\setcounter{footnote}{0}
\begin{abstract}
\noindent While it is widely recognised that linear (structural) VARs
may fail to capture important aspects of economic time series, the
use of nonlinear SVARs has to date been almost entirely confined to
the modelling of stationary time series, because of a lack of understanding
as to how common stochastic trends may be accommodated within nonlinear
models. This has unfortunately circumscribed the range of series to
which such models can be applied -- and/or required that these series
be first transformed to stationarity, a potential source of misspecification
-- and prevented the use of long-run identifying restrictions in
these models. To address these problems, we develop a flexible class
of additively time-separable nonlinear SVARs, which subsume models
with threshold-type endogenous regime switching, both of the piecewise
linear and smooth transition varieties. We extend the Granger--Johansen
representation theorem to this class of models, obtaining conditions
that specialise exactly to the usual ones when the model is linear.
We further show that, as a corollary, these models are capable of
supporting the same kinds of long-run identifying restrictions as
are available in linearly cointegrated SVARs.
\end{abstract}
\vfill
\noindent First version: April 2024. We thank B.~Beare, J.~Dolado,
A.\ Escribano, S.~Johansen, Y.~Lu, B.\ Nielsen, B.~Rossi, and
seminar participants at UC3 Madrid, Oxford, Sydney, the 2024 BSE Summer
Forum and the 26th Dynamic Econometrics Conference, for their helpful
comments on earlier drafts of this paper.
\thispagestyle{plain}
\pagenumbering{roman}
\newpage{}
\thispagestyle{plain}
\setcounter{tocdepth}{2}
{\singlespacing
\tableofcontents{}
}
\newpage{}
\newpage{}
\pagenumbering{arabic}
\section{Introduction}
For more than four decades, the linear structural VAR (SVAR) model,
\begin{align}
\Phi_{0}z_{t} & =c+\sum_{i=1}^{k}\Phi_{i}z_{t-i}+u_{t} & u_{t} & =\Upsilon\varepsilon_{t}\label{eq:svar-intro}
\end{align}
has played a central role in empirical macroeconomics. Structural
VARs provide a satisfyingly coherent framework within which the problem
of identifying causal relations, in the presence of simultaneity,
may be approached by regarding the $p$ endogenous variables $z_{t}$
as generated by an underlying sequence of $p$ exogenous structural
shocks $\varepsilon_{t}$. Viewed through the lens of these models, the identification
problem reduces to one of identifying these structural shocks and
their associated impulse responses: equivalently, of identifying the
mapping $\Upsilon$ between $\varepsilon_{t}$ and $u_{t}$.
The early literature on structural VARs, following the seminal contribution
of \citet{Sims80}, developed contemporaneously with major advances
in the modelling of common (stochastic) trends in macroeconomic time
series. This latter work culminated in the development of the theory
of the cointegrated VAR, and in particular the Granger--Johansen
representation theorem (GJRT), which showed how a VAR could be configured
so as to accommodate the presence of common stochastic trends, and
the cointegrating relations that are dual to them (\citealp{EG87Ecta};
\citealp{Joh91Ecta}). This importantly demonstrated that the SVAR
model could be reconciled with some of the evident properties of the
(levels of) nonstationary macroeconomic time series, obviating the
need to difference these series to stationarity prior to analysis
(indeed, an implication of this work was that such pre-filtering was
not only unnecessary, but could induce model misspecification). An
important corollary to the GJRT was that by uncovering the mapping
between the common trends and the endogenous variables, it made available
a class of long-run identifying restrictions, relating to whether
certain structural shocks were regarded as having permanent, or merely
transient, effects (\citealp{BlanchardQuah89}; \citealp{KPSW91AER}).
Subsequent work has sought to extend the (S)VAR model so as to allow
for various forms of nonlinearity: typically as modelled via some
sort of regime switching, possibly with smooth transitions between
these regimes (see e.g.\ \citealp{Chan09}; \citealp{TTG10}; \citealp{HT13}).
More recently, interest in structural macroeconomic models that incorporate
occasionally binding constraints, of which the zero lower bound on
interest rates is a leading instance, has had its counterpart in the
development of new class of regime-switching nonlinear VARs. These
newer models are notably distinguished from the earlier nonlinear
VAR literature by being able to accommodate \emph{endogenous} regime
switching, i.e.\ of allowing the current regime to depend on the
current values of the endogenous variables, rather than being pre-determined,
or being dependent on some exogenous switching process (\citealp{SM21};
\citealp{AMSV21}).
However, there remains a significant disconnect between the literature
on nonlinear (S)VARs, and that on cointegration and common trends.
In this respect, the situation is reminiscent of that of the literature
on linear VARs in the early 1980s, with there being little theoretical
understanding of how common trends may be accommodated within nonlinear
VARs, such that the prior removal of common trends by pre-filtering
remains a necessity. One area where significant efforts have been
made to remedy this pertain to a certain class of `nonlinear VECM'
models. These are VAR models that are specified directly in VECM form,
with a linear cointegrating space, but in which the stationary components
of the model (the equilibrium errors and the lagged differences) are
permitted to enter nonlinearly (see e.g.\ \citealp{EM02JTSA}; \citealp{BR04EctJ};
\citealp{Saik05JoE,Saik08ET}; \citealp{KR10JoE,KR13ET}). This class
of models has proved sufficiently tractable to facilitate a nonlinear
extension of the GJRT, but the assumptions made in this literature
are such as to leave a substantial portion of the space of nonlinear
VAR models unexplored.
In this paper, we generalise \ref{eq:svar-intro} in the direction
of the additively time-separable SVAR
\begin{align}
f_{0}(z_{t}) & =c+\sum_{i=1}^{k}f_{i}(z_{t-i})+u_{t} & u_{t} & =\Upsilon\varepsilon_{t}\label{eq:SVAR-intro}
\end{align}
where $f_{0}:\mathbb{R}^{p}\ensuremath{\rightarrow}\mathbb{R}^{p}$ is continuous and has
a continuous inverse (i.e.\ it is a homeomorphism), and each $f_{i}:\mathbb{R}^{p}\ensuremath{\rightarrow}\mathbb{R}^{p}$
is continuous.\footnote{For a detailed explanation of why the structural shocks enter \ref{eq:SVAR-intro}
in an apparently similar manner to \ref{eq:svar-intro}, see \ref{subsec:identification}
below.} We aim not only to extend the GJRT, but also to preserve exactly
the kind of long-run identifying restrictions that are yielded by
the GJRT in a linear setting, such that it remains possible to fruitfully
discriminate between structural shocks on the basis of the permanence
(or transience) of their effects. Via an associated VECM representation,
this leads us to consider SVARs in which the \emph{image} of the map
\[
\pi(z)\coloneqq\sum_{i=1}^{k}f_{i}(z)-f_{0}(z)
\]
is an $r$-dimensional subspace of $\mathbb{R}^{p}$, a requirement that
we term the \emph{common row space condition} (CRSC); effectively,
this entails a decomposition of the form $\pi(z)=\alpha\theta(z)$. This
reverses a key assumption of the nonlinear VECM literature, where
$\theta(z)=\beta^{\mathsf{T}}z$ is a fixed linear function, but $\alpha$
is allowed to vary (albeit only as a function of $\beta^{\mathsf{T}}z_{t-1}$
and $\Delta z_{t-1}$, or their lags). As we discuss, the principal
motivation for the CRSC is that it ensures that only $q=p-r$ structural
shocks may have permanent effects -- whereas if the CRSC is violated,
all $p$ structural shocks may have permanent effects, even though
$z_{t}$ has only $q<p$ common stochastic trends.
Our conditions on the nonlinear SVAR \ref{eq:SVAR-intro} are otherwise
rather general, and unlike the preceding literature allow for common
stochastic trends to enter $z_{t}$ \emph{nonlinearly}, thereby giving
rise to nonlinear cointegrating relations between elements of $z_{t}$.
(A rigorous discussion of `cointegration' in this nonlinear setting
is deferred to a companion paper, in which the limiting distribution
of the standardised process $n^{-1/2}z_{\smlfloor{n\lambda}}$ is
derived; for a discussion of this in the context of the censored and
kinked SVAR, see \citealp{DMW22}.) These conditions are presented
in \ref{sec:model}, followed by a discussion of the structural shock
identification problem that arises in this model. We further show
that our conditions reduce to exactly those of the linear cointegrated
VAR, when each $\{f_{i}\}_{i=0}^{k}$ is linear: and thus they do
not appear to be particularly restrictive.
The principal contribution of this paper is to provide an extension
of the GJRT to the setting of a nonlinear VAR (\ref{sec:common-trends}),
demonstrating how the model \ref{eq:SVAR-intro} may be configured
so as to generate common stochastic trends. This in turn permits an
analysis of the stability of the steady state equilibria of the model,
and a characterisation of the directions in which the shocks $u_{t}$
have only transient effects (\ref{subsec:attractors}), and thereby
yields long-run identifying restrictions on $\Upsilon$ (\ref{subsec:long-run-id}).
The GJRT also facilitates the calculation of the implied long-run
multipliers associated with the shocks (\ref{sec:lr-multipliers}).
We subsequently return, in \ref{sec:without-the-crsc}, to a discussion
of the role played by the CRSC in ensuring the availability of long-run
identifying restrictions. Finally, \ref{subsec:piecewise-affine} elaborates
on a class of models in which each $f_{i}$ is piecewise affine
(or a smooth, convex combination of piecewise affine functions) showing
how the abstract conditions introduced in \ref{sec:model} may be specialised
to this class of models. \ref{sec:conclusion} concludes. Proofs of
all results appear in the appendices.
\begin{notation*}
$e_{m,i}$ denotes the $i$th column of an $m\times m$ identity matrix;
when $m$ is clear from the context, we write this simply as $e_{i}$.
$\smlnorm{\cdot}$ denotes the Euclidean norm on $\mathbb{R}^{m}$. Matrix
norms are always those induced by the corresponding vector norm. For
$X$ a random vector and $p\geq1$, $\smlnorm X_{p}\coloneqq(\ensuremath{\mathbb{E}}\smlnorm X^{p})^{1/p}$.
For a matrix $\alpha\in\mathbb{R}^{p\times r}$ with $\operatorname{rk}\alpha=r$,
$\alpha_{\perp}$ denotes some $p\times(p-r)$ matrix, with $\operatorname{rk}\alpha_{\perp}=p-r$,
such that $\alpha_{\perp}^{\mathsf{T}}\alpha=0$. For $\{A_{i}\}_{i=1}^{k}$
a collection of $m\times n$ matrices, $[A_{i}]_{i=1}^{k}$ denotes
the $mk\times n$ matrix formed as $(A_{1}^{\mathsf{T}},\ldots,A_{m}^{\mathsf{T}})^{\mathsf{T}}$.
The \emph{convex hull} of $\{A_{i}\}_{i=1}^{k}$ is denoted $\operatorname{co}\{A_{i}\}_{i=1}^{k}\coloneqq\{\sum_{i=1}^{k}\lambda_{i}A_{i}\mid\lambda_{i}\geq0,\ \sum_{i=1}^{k}\lambda_{i}=1\}$.
For a continuously differentiable function $f:\mathbb{R}^{\ell}\ensuremath{\rightarrow}\mathbb{R}^{m}$,
$\partial_{x}f(x_{0})$ denotes the Jacobian of $f$ at $x=x_{0}$.
\end{notation*}
\section{Model}
\label{sec:model}
\subsection{Framework: a nonlinear VAR($k$)}
\label{subsec:framework}
We consider an additively time-separable nonlinear VAR($k$) model
for a $p$-dimensional time series $\{z_{t}\}_{t\in\mathbb{Z}}$, of
the form
\begin{equation}
f_{0}(z_{t})=c+\sum_{i=1}^{k}f_{i}(z_{t-i})+u_{t}\label{eq:nlVAR}
\end{equation}
where $f_{i}:\mathbb{R}^{p}\ensuremath{\rightarrow}\mathbb{R}^{p}$ for $i\in\{0,1,\ldots,k\}$,
and $\{u_{t}\}_{t\in\mathbb{Z}}$ is a sequence of `shocks' taking
values in $\mathbb{R}^{p}$. (As a convenient location normalisation,
we shall suppose throughout that $f_{i}(0)=0$ for all $i\in\{0,\ldots,k\}$.)
By defining
\begin{equation}
g_{j}(z)\coloneqq-\sum_{i=j+1}^{k}f_{i}(z)\label{eq:ge}
\end{equation}
for $j\in\{0,\ldots,k-1\}$, we can write the model in nonlinear
VECM form as
\begin{equation}
\Deltaf_{0}(z_{t})=c+\pi(z_{t-1})+\sum_{i=1}^{k-1}\Deltag_{i}(z_{t-i})+u_{t}\label{eq:VECM-first}
\end{equation}
where
\begin{equation}
\pi(z)\coloneqq-[g_{0}(z)+f_{0}(z)]=-f_{0}(z)+\sum_{i=1}^{k}f_{i}(z)\label{eq:pi-def}
\end{equation}
(see also \ref{lem:vecm} below).
In the linear VAR, where $\pi(z)=\Pi z$ is linear, the key assumption
required for the model to generate common stochastic trends (i.e.\ cointegrated
$I(1)$ processes) is that $\operatorname{rk}\Pi=r=p-q$, where $q$ is the number
of unit roots in the autoregressive polynomial. Equivalently, this
condition may be expressed in terms of the image of the map $z\ensuremath{\mapsto}\Pi z$,
\[
\operatorname{im}\Pi=\{\Pi z\mid z\in\mathbb{R}^{p}\}=\operatorname{sp}\Pi,
\]
being an $r$-dimensional (linear) subspace of $\mathbb{R}^{p}$. In extending
the GJRT to the setting of the nonlinear VAR \ref{eq:nlVAR}, the role
of $\Pi$ (or more precisely, of $z\ensuremath{\mapsto}\Pi z$) will now be played
by $\pi(z)$, and our high-level assumptions below will ensure that
\[
\operatorname{im}\pi=\{\pi(z)\mid z\in\mathbb{R}^{p}\}
\]
remains an $r$-dimensional (linear) subspace of $\mathbb{R}^{p}$, exactly
as in the linear case. We term this the \emph{common row space} \emph{condition}
(CRSC). (Note that there is no corresponding restriction on $\ker\pi$;
in particular, this will \emph{not} be required to be a $q$-dimensional
subspace of $\mathbb{R}^{p}$.) For a further discussion of this condition
-- in particular, of the role that it plays in ensuring that the
long-run identifying restrictions familiar from a linear cointegrated
VAR are also available in the present setting -- see \ref{sec:without-the-crsc}
below.
\subsection{High-level assumptions}
\label{subsec:high-level}
To state our main assumptions on the model \ref{eq:nlVAR}, we first
provide a more compact representation for \ref{eq:VECM-first}; its
proof appears in \ref{app:auxproof}.
\begin{lem}
\label{lem:vecm}Suppose $\{z_{t}\}$ follows \ref{eq:nlVAR}. Then
\ref{eq:VECM-first} holds. Defining $\b{\zeta}_{t}\coloneqq[\zeta_{j,t}]_{j=1}^{k-1}$,
where
\[
\zeta_{j,t}\coloneqq\sum_{i=j}^{k-1}g_{i}(z_{t-i+j}),
\]
and letting $\b z_{t}\coloneqq(z_{t}^{\mathsf{T}},\b{\zeta}_{t-1}^{\mathsf{T}})^{\mathsf{T}}$,
we also have
\begin{align}
\begin{bmatrix}\Deltaf_{0}(z_{t})\\
\Delta\b{\zeta}_{t-1}
\end{bmatrix} & =\begin{bmatrix}c\\
0_{p(k-1)}
\end{bmatrix}+\begin{bmatrix}\pi(z_{t-1})+g_{1}(z_{t-1})\\
\b{g}(z_{t-1})
\end{bmatrix}+\begin{bmatrix}0 & D_{1}\\
0 & D
\end{bmatrix}\begin{bmatrix}z_{t-1}\\
\b{\zeta}_{t-2}
\end{bmatrix}+\begin{bmatrix}u_{t}\\
0_{p(k-1)}
\end{bmatrix}\nonumber \\
& \eqqcolon\b c+\b{\pi}(z_{t-1})+\b D\b z_{t-1}+\b u_{t}.\label{eq:VECM-system}
\end{align}
for $\b{g}(z)\coloneqq[g_{j}(z)]_{j=1}^{k-1}$,
\[
D\coloneqq\begin{bmatrix}-I_{p} & I_{p}\\
& \ddots & \ddots\\
& & -I_{p} & I_{p}\\
& & & -I_{p}
\end{bmatrix}\in\mathbb{R}^{p(k-1)\times p(k-1)}
\]
and $D_{1}$ the first $p$ rows of $D$.
\end{lem}
\begin{rem}
\label{rem:vecm-representation}
\refstepcounter{subremark}
(\roman{subremark}){}.
{}\label{subrem:higher-order}
The preceding holds exactly as stated when $k\geq3$; the model with
$k\in\{1,2\}$ may be handled as a special case of $k=3$ with at
least one of $f_{2}$ and $f_{3}$ being identically zero. Alternatively,
we may specialise the above to $k\in\{1,2\}$ by taking $D$ to be
an `empty' matrix when $k=1$, and $D=-I_{p}$ when $k=2$. (In
the proofs and statements of all results, to avoid these kinds of
complications, we shall implicitly maintain throughout, without loss
of generality, that $k\geq3$.)
\refstepcounter{subremark}
(\roman{subremark}){}.
{} \ref{eq:VECM-system} differs notably from the companion
form representation that is typically employed in the analysis of
linear VECMs. In particular, the state vector is not
\[
\vec{z}_{t}\coloneqq(z_{t}^{\mathsf{T}},\ldots,z_{t-k+1}^{\mathsf{T}})^{\mathsf{T}},
\]
but rather
\begin{equation}
\b z_{t}=\begin{bmatrix}z_{t}\\
\b{\zeta}_{t-1}
\end{bmatrix}=\begin{bmatrix}z_{t}\\
\zeta_{1,t-1}\\
\vdots\\
\zeta_{k-1,t-1}
\end{bmatrix}=\begin{bmatrix}z_{t}\\
\sum_{i=1}^{k-1}g_{i}(z_{t-i})\\
\vdots\\
g_{k-1}(z_{t-1})
\end{bmatrix}=\begin{bmatrix}z_{t}\\
\sum_{i=1}^{k-1}\Gamma_{i}z_{t-i}\\
\vdots\\
\Gamma_{k-1}z_{t-1}
\end{bmatrix},\label{eq:bz-in-linear}
\end{equation}
where the final equality holds only in a \emph{linear} VAR. By comparison
with $\vec{z}_{t}$, we have defined $\b z_{t}$ so that the r.h.s.\ of
\ref{eq:VECM-system} involves a nonlinear transformation of only $z_{t-1}$,
and is otherwise linear in $\b{\zeta}_{t-2}$. This permits a major
simplification of the analysis of the model when $k\geq2$; for this
the assumption of additive separability is crucially important.
\end{rem}
We next define a class ${\cal M}_{r}$ of nonlinear VAR models, that
will be shown to be consistent with the presence of $q=p-r$ common
stochastic trends in $\{z_{t}\}$, and $r$ `cointegrating relations'
between those series. For some $\theta:\mathbb{R}^{p}\ensuremath{\rightarrow}\mathbb{R}^{r}$,
define
\begin{align}
\b{\theta}(z) & \coloneqq\begin{bmatrix}\theta(z)\\
\b{g}(z)
\end{bmatrix} & \b D_{0} & \coloneqq\begin{bmatrix}0_{r\times p} & 0\\
0 & D
\end{bmatrix} & E & \coloneqq\begin{bmatrix}I_{p}\\
0_{p(k-2)\times p}
\end{bmatrix},\label{eq:bold-theta}
\end{align}
and for $\alpha\in\mathbb{R}^{p\times r}$ with $\operatorname{rk}\alpha=r$,
\begin{align}
\b{\alpha} & \coloneqq\begin{bmatrix}\alpha & E^{\mathsf{T}}\\
0 & I_{p(k-1)}
\end{bmatrix} & \b{\alpha}_{\perp} & \coloneqq\begin{bmatrix}I_{p}\\
-E
\end{bmatrix}\alpha_{\perp},\label{eq:bold-alpha}
\end{align}
Recall that a function $f:{\cal X\ensuremath{\rightarrow}{\cal Y}}$ is a \emph{homeomorphism}
if it is continuous, bijective, and has a continuous inverse. $f$
is \emph{bi-Lipschitz} if there exists a $C_{0}<\infty$ such that
$C_{0}^{-1}\smlnorm{x-x^{\prime}}\leq\smlnorm{f(x)-f(x^{\prime})}\leq C_{0}\smlnorm{x-x^{\prime}}$
for all $x,x^{\prime}\in{\cal X}$; we term any such $C_{0}$ a \emph{bi-Lipschitz
constant} for $f$. For $\mathcal{A}\subset\mathbb{R}^{m\times m}$ a
bounded collection of matrices, let $\rho_{{\scriptstyle \mathrm{JSR}}}(\mathcal{A})\coloneqq\limsup_{t\ensuremath{\rightarrow}\infty}\sup_{B\in\mathcal{A}^{t}}\rho(B)^{1/t}$
denote its joint spectral radius (JSR; e.g.\ \citealp{Jungers09},
Defn.\ 1.1), where $\mathcal{A}^{t}\coloneqq\{\prod_{s=1}^{t}M_{s}\mid M_{s}\in\mathcal{A}\}$
is the set of $t$-fold products of matrices in ${\cal A}$, and $\rho(B)$
the spectral radius of $B$.
\needspace{3cm}
\begin{defn}
\label{def:cvar}Let $r\in\{0,\ldots,p\}$, $\alpha\in\mathbb{R}^{p\times r}$
have $\operatorname{rk}\alpha=r$, $\abv b\in\mathbb{R}$, and $\abv{\rho}\in[0,1)$.
We say that the VAR($k$) model \ref{eq:nlVAR}, parametrised by $c\in\mathbb{R}^{p}$
and $\{f_{i}\}_{i=0}^{k}$, belongs to class ${\cal M}_{r}(\alpha,\abv b,\abv{\rho})$
if:
\begin{enumerate}[label=(${\cal M}$.\roman*), leftmargin=1.75cm]
\item \label{enu:cvar:homeo}$f_{0}:\mathbb{R}^{p}\ensuremath{\rightarrow}\mathbb{R}^{p}$ is
a homeomorphism;
\item \label{enu:cvar:rank} there exists $\mu\in\mathbb{R}^{r}$ and $\theta:\mathbb{R}^{p}\ensuremath{\rightarrow}\mathbb{R}^{r}$
such that
\begin{align*}
c & =\alpha\mu & \pi(z) & =\alpha\theta(z)
\end{align*}
for all $z\in\mathbb{R}^{p}$;
\item \label{enu:cvar:jsr}for every $\b z=(z^{\mathsf{T}},\b{\zeta}^{\mathsf{T}})^{\mathsf{T}}$
and $\b z^{\prime}=(z^{\prime\mathsf{T}},\b{\zeta}^{\prime\mathsf{T}})^{\mathsf{T}}$
in $\mathbb{R}^{p}\times\mathbb{R}^{p(k-1)}$, there exists a $\b{\beta}\in{\cal B}$
(possibly depending on $\b z$ and $\b z^{\prime}$) such that
\begin{equation}
[(\b{\theta}\ensuremath{\circ}f_{0}^{-1})(z)+\b D_{0}\b z]-[(\b{\theta}\ensuremath{\circ}f_{0}^{-1})(z^{\prime})+\b D_{0}\b z^{\prime}]=\b{\beta}^{\mathsf{T}}(\b z-\b z^{\prime})\label{eq:gradient}
\end{equation}
where ${\cal B}\subset\mathbb{R}^{p\times[p(k-1)+r]}$ is closed with
$\max_{\b{\beta}\in{\cal B}}\smlnorm{\b{\beta}}\leq\abv b$ and
\begin{equation}
\rho_{{\scriptstyle \mathrm{JSR}}}(\{I_{p(k-1)+r}+\b{\beta}^{\mathsf{T}}\b{\alpha}\mid\b{\beta}\in{\cal B}\})\leq\abv{\rho}.\label{eq:jsr}
\end{equation}
\end{enumerate}
Let ${\cal M}_{r}^{\ast}(\alpha,\abv b,\abv{\rho})$ denote those models
in ${\cal M}_{r}(\alpha,\abv b,\abv{\rho})$ for which $f_{0}=I_{p}$,
the identity on $\mathbb{R}^{p}$. We denote by ${\cal M}_{r}$ (resp.\ ${\cal M}_{r}^{\ast}$)
the union of all classes ${\cal M}_{r}(\alpha,\abv b,\abv{\rho})$ (resp.\ ${\cal M}_{r}^{\ast}(\alpha,\abv b,\abv{\rho})$)
with $\alpha\in\mathbb{R}^{p\times r}$ having $\operatorname{rk}\alpha=r$, $\abv b<\infty$
and $\abv{\rho}<1$.
\end{defn}
\begin{rem}
\label{rem:cvar}
\refstepcounter{subremark}
(\roman{subremark}){}.
{} We allow $f_{0}$ to be non-trivial
to permit (linear and nonlinear) \emph{structural} VARs to be accommodated
within the present setting: see \ref{subsec:identification} for a
further discussion. In the linear case, $f_{0}^{-1}(z)=\Phi_{0}^{-1}z$
is trivially continuous; requiring $f_{0}$ to be a homeomorphism
extends this to a nonlinear setting. When deriving the properties
of the series generated by the models in class ${\cal M}_{r}$, we shall
first rewrite the system in terms of $z_{t}^{\ast}\coloneqqf_{0}(z_{t})$,
which is itself generated by a model in class ${\cal M}_{r}^{\ast}$
(see \ref{lem:unstruct} in \ref{app:auxiliary}).
\refstepcounter{subremark}
(\roman{subremark}){}.
{} It follows from the preceding conditions, and principally
from \ref{enu:cvar:jsr}, that $f_{i}$ is continuous for each $i\in\{1,\ldots,k\}$;
indeed, these functions are Lipschitz when $f_{0}=I_{p}$ (see \ref{lem:co-cts}
in \ref{app:auxiliary}).
\refstepcounter{subremark}
(\roman{subremark}){}.
{} A fundamental role will be played throughout the following
by the map $\chi:\mathbb{R}^{p}\ensuremath{\rightarrow}\mathbb{R}^{p}$ defined by
\begin{equation}
\chi(z)\coloneqq\begin{bmatrix}\psi(z)\\
\theta(z)
\end{bmatrix},\label{eq:co-map}
\end{equation}
where
\begin{equation}
\psi(z)\coloneqq\alpha_{\perp}^{\mathsf{T}}\left[f_{0}(z)-\sum_{i=1}^{k-1}g_{i}(z)\right]\eqqcolon\alpha_{\perp}^{\mathsf{T}}h(z).\label{eq:ct-def}
\end{equation}
As developed in \ref{subsec:gjrt} below, if $\{z_{t}\}$ is generated
by a model from class ${\cal M}_{r}$, then $\psi(z_{t})$ and $\theta(z_{t})$
respectively provide a representation for the `common trends' and
`equilibrium errors' in the system. Since $\chi$ is guaranteed
to be a homeomorphism under our assumptions (see \ref{lem:co-cts}
in \ref{app:auxiliary}), these maps provide an exhaustive description
of $z_{t}$. As a further consequence of the invertibility of $\chi$,
we have $\operatorname{im}\theta=\mathbb{R}^{r}$ and hence $\operatorname{im}\pi=\operatorname{sp}\alpha$, an $r$-dimensional
subspace of $\mathbb{R}^{p}$; models in class ${\cal M}_{r}$ thus satisfy
the common row space condition introduced in \ref{subsec:framework}
above.
\refstepcounter{subremark}
(\roman{subremark}){}.
{} The case $r=0$ corresponds to $c=0$ and $\pi(z)=0$
for all $z\in\mathbb{R}^{p}$. The results in this paper will be seen
to cover this case, if the appropriate adjustments are made by deleting
$\theta$ and $\mu$ from those objects in which they appear, so that
e.g.\ now $\b{\theta}=\b{g}$ and $\chi=\psi$. (To keep this paper
to a manageable length, we do not treat this case explicitly in our
proofs.) At the other extreme, the case $r=p$ corresponds to a stationary
system with no common trends. For this reason, we generally restrict
the statements of our results to the case where $r\in\{0,\ldots,p-1\}$.
\refstepcounter{subremark}
(\roman{subremark}){}.
{}\label{subrem:cvar:shortmem} \ref{enu:cvar:jsr} is
our principal assumption on the stability of the weakly dependent
components of the system, which comprise the `equilibrium errors'
$\xi_{t}\coloneqq\mu+\theta(z_{t})$, and the `lagged differences' $\Delta\zeta_{j,t}=\sum_{i=j}^{k-1}\Deltag_{i}(z_{t-i+j})$
for $j\in\{1,\ldots,k-1\}$. In a linear cointegrated VAR, these reduce
to the familiar $\xi_{t}=\mu+\beta^{\mathsf{T}}z_{t}$ and $\Delta\zeta_{j,t}=\sum_{i=j}^{k-1}\Gamma_{j}\Delta z_{t-i+j}$,
which are stationary (up to initialisation) by the GJRT. Though in
the present setting, $\xi_{t}$ and $\{\Delta\zeta_{j,t}\}_{j=1}^{k-1}$
will generally \emph{not} be stationary, \ref{eq:jsr} nonetheless
ensures that these are of strictly smaller order than $z_{t}$ itself,
just as in a linearly cointegrated system. As noted in \ref{subsec:linear:model}
below, in a linear VAR \ref{enu:cvar:jsr} reduces to precisely the
familiar conditions on the roots of the associated autoregressive
polynomial; whereas the form in which it is stated here is applicable
to models which depart so far from linearity that no useful notion
of `autoregressive roots' is available.
\refstepcounter{subremark}
(\roman{subremark}){}.
{}\label{subrem:cvar:lipschitz} The requirement that \ref{eq:gradient}
hold with $\smlnorm{\b{\beta}}\leq\abv b$ implies $\b{\theta}\ensuremath{\circ}f_{0}^{-1}$
is Lipschitz, with a Lipschitz constant that is bounded by $\abv b$;
the converse is also true (see \ref{lem:lipcondition} in \ref{app:auxiliary}).
Excepting cases where $f_{0}$ and $\b{\theta}$ interact in a very
particular way, this condition signals that our concern here is principally
with models in which the equilibrium errors are defined by a function
$\b{\theta}$ for which $\b{\theta}(z)=O(\smlnorm z)$ as $\smlnorm z\ensuremath{\rightarrow}\infty$,
i.e.\ which exhibits at most linear asymptotic growth. For example,
if $f_{0}=I_{p}$ and $\theta(z)=y-m(x)$, for $z=(y^{\mathsf{T}},x^{\mathsf{T}})^{\mathsf{T}}$,
then \ref{enu:cvar:jsr} requires $m$ to be Lipschitz, and thus that
$m(x)=O(\smlnorm x)$ as $\smlnorm x\ensuremath{\rightarrow}\infty$, for $r\in\{0,\ldots,p-1\}$.
\end{rem}
Our assumptions on the data generating process may now be stated.
\renewcommand{{\upshape{{\scalefont{0.76}#1}}}}{{\upshape{{\scalefont{0.76}CVAR}}}}
\begin{assumption}
\label{ass:cvar} Given $\vec{z}_{0}=(z_{0}^{\mathsf{T}},\ldots,z_{-k+1}^{\mathsf{T}})^{\mathsf{T}}$,
$\{z_{t}\}_{t\in\mathbb{Z}}$ follows \ref{eq:nlVAR} for $t\geq1$,
where $(c,\{f_{i}\}_{i=0}^{k})$ belong to class ${\cal M}_{r}$.
\end{assumption}
For the purposes of much of this paper, the sequence of innovations
$\{u_{t}\}$ may be treated as a given non-random sequence in $\mathbb{R}^{p}$,
since our results are either non-asymptotic, or take limits under
the assumption that the shocks cease at some finite horizon $\tau$.
However, it will occasionally be useful to indicate the implications
of our results in cases where the following holds.
\renewcommand{{\upshape{{\scalefont{0.76}#1}}}}{{\upshape{{\scalefont{0.76}ERR}}}}
\begin{assumption}
\label{ass:coerr}$\{u_{t}\}_{t\in\mathbb{Z}}$ is a random sequence
in $\mathbb{R}^{p}$, such that $\sup_{t\in\mathbb{Z}}\smlnorm{u_{t}}_{2+\delta_{u}}<\infty$
for some $\delta_{u}>0$, and
\begin{equation}
U_{n}(\lambda)\coloneqq n^{-1/2}\sum_{t=1}^{\smlfloor{n\lambda}}u_{t}\ensuremath{\rightsquigarrow} U(\lambda)\label{eq:Unwkc}
\end{equation}
on $D[0,1]$, where $U$ is a $p$-dimensional Brownian motion with
positive-definite variance $\Sigma$.
\end{assumption}
\needspace{3cm}
\subsection{Relationship to the linear VAR}
\label{subsec:linear:model}
To help cast further light on the defining conditions of class ${\cal M}_{r}$,
we now show that when \ref{eq:nlVAR} is specialised to a linear VAR,
these conditions reduce to exactly the usual conditions for a linear
VAR to generate common $I(1)$ components, as per the GJRT (\citealp{Joh95},
Ch.\ 4). This provides a clear indication that our conditions on
the general, nonlinear VAR are not overly restrictive. By a \emph{linear
VAR}, we mean one in which $f_{i}(z)=\Phi_{i}z$ for some $\Phi_{i}\in\mathbb{R}^{p\times p}$,
for $i\in\{0,\ldots,k\}$, so that \ref{eq:nlVAR} becomes
\begin{equation}
\Phi_{0}z_{t}=c+\sum_{i=1}^{k}\Phi_{i}z_{t-i}+u_{t}.\label{eq:linearVAR}
\end{equation}
Let $\Phi(\lambda)\coloneqq\Phi_{0}-\sum_{i=1}^{k}\Phi_{i}\lambda^{i}$
denote the associated autoregressive polynomial.
\begin{prop}
\label{prop:lin-rep-var}Suppose $(c,\{f_{i}\}_{i=0}^{k})$ is a
linear VAR, where for some $r\in\{0,\ldots,p\}$,
\begin{enumerate}[label={\upshape{{\scalefont{0.76}LIN.\arabic*}}}, leftmargin=1.75cm]
\item \label{enu:lin:homeo}$\Phi_{0}$ is invertible;
\item \label{enu:lin:rank}$c\in\operatorname{sp}\Phi(1)$ and $\operatorname{rk}\Phi(1)=r=p-q$;
and
\item \label{enu:lin:roots}$\Phi(\lambda)$ has $q$ roots at unity, and
all others outside the unit circle.
\end{enumerate}
Then $(c,\{f_{i}\}_{i=1}^{k})$ belongs to class ${\cal M}_{r}$, with:
$g_{j}(z)=\Gamma_{j}z$, where $\Gamma_{j}\coloneqq-\sum_{i=j+1}^{k}\Phi_{i}$
for $j\in\{1,\ldots,k-1\}$; and
\[
\chi(z)=\begin{bmatrix}\psi(z)\\
\theta(z)
\end{bmatrix}=\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}(\Phi_{0}-\sum_{i=1}^{k-1}\Gamma_{i})\\
\beta^{\mathsf{T}}
\end{bmatrix}z\eqqcolon\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}\mathrm{H}\\
\beta^{\mathsf{T}}
\end{bmatrix}z\eqqcolon\mathrm{X} z
\]
for $\alpha,\beta\in\mathbb{R}^{p\times r}$ such that $\alpha\beta^{\mathsf{T}}=-\Phi(1)$.
\end{prop}
The proof of \ref{prop:lin-rep-var} demonstrates that the factorisation
required of $\pi(z)$ in \ref{enu:cvar:rank} has its direct counterpart
in condition \ref{enu:lin:rank} on the rank of $\Phi(1)$, while
the high-level condition \ref{enu:cvar:jsr}, which here involves
only the matrices
\begin{align}
\b{\beta}^{\mathsf{T}} & \coloneqq\begin{bmatrix}\beta^{\mathsf{T}}\Phi_{0}^{-1} & 0\\
\b{\Gamma}\Phi_{0}^{-1} & D
\end{bmatrix}, & \b{\alpha} & \coloneqq\begin{bmatrix}\alpha & E^{\mathsf{T}}\\
0 & I_{p(k-1)}
\end{bmatrix}, & \b{\Gamma} & \coloneqq[\Gamma_{i}]_{i=1}^{k-1}\coloneqq\left[\begin{smallmatrix}\Gamma_{1}\\
\vdots\\
\Gamma_{k-1}
\end{smallmatrix}\right],\label{eq:lin-boldbeta}
\end{align}
is equivalent to the familiar condition \ref{enu:lin:roots} on the
autoregressive roots.
\subsection{The structural nonlinear VAR($k$)}
\label{subsec:identification}
In macroeconomics, a conventional starting point for empirical work
is the \emph{structural} \emph{(linear) VAR} model, in which the observed
series $\{z_{t}\}$ are regarded as being generated by an underlying
$p$-dimensional i.i.d.\ sequence of \emph{structural shocks} $\{\varepsilon_{t}\}$,
whose elements are mutually orthogonal, and each of which have an
economic interpretation (as e.g.\ an aggregate supply shock, a monetary
policy shock, etc.).\footnote{In some treatments (e.g.\ \citealp{LMS17JoE}; \citealp{GMR20REStud})
the components of $\varepsilon_{t}$ are taken to be not merely uncorrelated,
but mutually independent. Under non-Gaussianity, this is known to
yield certain additional identifying restrictions. We emphasise that
in the present work, only orthogonality between the components of
$\varepsilon_{t}$ is maintained.} This perspective may also be adopted within the framework of \ref{eq:nlVAR},
reproduced here as
\begin{equation}
f_{0}(z_{t})=c+\sum_{i=1}^{k}f_{i}(z_{t-i})+u_{t},\tag{\ref{eq:nlVAR}}
\end{equation}
by supposing that there is a parametrisation $(c^{\varepsilon},\{f_{i}^{\varepsilon}\}_{i=0}^{k})$
of the model such that $\{z_{t}\}$ satisfies
\begin{equation}
f_{0}^{\varepsilon}(z_{t})=c^{\varepsilon}+\sum_{i=1}^{k}f_{i}^{\varepsilon}(z_{t-i})+\varepsilon_{t}.\label{eq:struct-form}
\end{equation}
We may suppose, as a convenient location and scale normalisation,
that each of the elements of $\varepsilon_{t}$ have mean zero and unit variance,
so that $\{\varepsilon_{t}\}$ is i.i.d.\ with $\ensuremath{\mathbb{E}}\varepsilon_{t}=0$ and
$\ensuremath{\mathbb{E}}\varepsilon_{t}\varepsilon_{t}^{\mathsf{T}}=I_{p}$. For the purposes of this
section, we shall also require $\varepsilon_{t}$ to be continuously distributed,
with density $\varrho^{\varepsilon}$. (For simplicity of presentation, we also
treat the order $k$ of the VAR as known, though only a known finite
upper bound on the order of the VAR is required for the discussion
that follows.)
The question then arises as to whether, or to what extent, the structural
shocks \{$\varepsilon_{t}\}$ are identified by observation of $\{z_{t}\}$.
To phrase this question more precisely, we first clarify how the model
\ref{eq:nlVAR} is parametrised. For $i\in\{0,\ldots,k\}$, let $\mathscr{F}_{i}$
denote a collection of functions $\mathbb{R}^{p}\ensuremath{\rightarrow}\mathbb{R}^{p}$ to
which each $f_{i}$ respectively belongs. We also specialise \ref{ass:coerr}
to the case where $\{u_{t}\}$ is i.i.d., with a density $\varrho$ on
$\mathbb{R}^{p}$ that belongs to some set $\mathscr{R}$, normalised to have
$\ensuremath{\mathbb{E}} u_{t}=0$ and $\ensuremath{\mathbb{E}} u_{t}u_{t}^{\mathsf{T}}=I_{p}$. The parameter
space for \ref{eq:nlVAR} is then
\[
\set P\coloneqq\{(\tilde{c},\{\tilde{f}_{i}\}_{i=0}^{k},\tilde{\varrho})\mid\tilde{c}\in\mathbb{R},\ \tilde{f}_{i}\in\mathscr{F}_{i}\text{ for }i\in\{0,\ldots,k\},\ \tilde{\varrho}\in\mathscr{R}\}.
\]
Under the regularity conditions given in \ref{app:identification},
each $(\tilde{c},\{\tilde{f}_{i}\}_{i=0}^{k},\tilde{\varrho})\in\set P$
completely specifies the density of $z_{t}$ conditional on $\vec{z}_{t-1}=(z_{t-1}^{\mathsf{T}},\ldots,z_{t-k}^{\mathsf{T}})^{\mathsf{T}}$.
Following \citet[pp.~949f.]{Matz09Ecta}, we say that two points in
$\set P$ are \emph{observationally equivalent} if they imply exactly
the same conditional density function, and thus the same likelihood
for $\{z_{t}\}_{t=1}^{n}$, conditional on the initial values $\vec{z}_{0}$
(which as per \ref{ass:cvar} we take to be arbitrary, and thus uninformative
about the model parameters).
Let $\Upsilon\in\mathbb{R}^{p\times p}$ be an orthogonal matrix, and
suppose that $\mathscr{R}$ is such that the density $\tilde{\varrho}$ of
$\tilde{u_{t}}\coloneqq\Upsilon\varepsilon_{t}$ also lies in $\mathscr{R}$. Then
it is evident from pre-multiplying \ref{eq:struct-form} through by
$\Upsilon$, to obtain
\[
\tilde{f}_{0}(z_{t})\coloneqq\Upsilonf_{0}^{\varepsilon}(z_{t})=\Upsilon c^{\varepsilon}+\sum_{i=1}^{k}\Upsilonf_{i}^{\varepsilon}(z_{t-i})+\Upsilon\varepsilon_{t}\eqqcolon\tilde{c}+\sum_{i=1}^{k}\tilde{f}_{i}(z_{t-i})+\tilde{u}_{t},
\]
that $(\tilde{c},\{\tilde{f}_{i}\}_{i=0}^{k},\tilde{\varrho})$ must
be observationally equivalent to $(c^{\varepsilon},\{f_{i}^{\varepsilon}\}_{i=0}^{k},\varrho^{\varepsilon})$.
Thus, as is familiar from the \emph{linear} structural VAR, the model
parameters and the structural shocks are at best identified up to
an unknown orthogonal transformation.
Remarkably, despite the much broader class of nonlinear transformations
permitted by \ref{eq:nlVAR}, such orthogonal transformations delineate
the \emph{only} loci of non-identification in the nonlinear VAR. By
\ref{thm:shockid} in \ref{app:identification}, under certain conditions
on the sets $\{\mathscr{F}_{i}\}_{i=0}^{k}$ and $\mathscr{R}$, and the parameters
generating $\{z_{t}\}$, there exists a $\varrho\in\mathscr{R}$ such that
the parameters $(c,\{f_{i}\}_{i=0}^{k},\varrho)$ of \ref{eq:nlVAR}
are observationally equivalent to those $(c^{\varepsilon},\{f_{i}^{\varepsilon}\}_{i=0}^{k},\varrho^{\varepsilon})$
of \ref{eq:struct-form}, if \emph{and only if} there is an orthogonal
matrix $\Upsilon\in\mathbb{R}^{p\times p}$ such that
\begin{equation}
c=\Upsilon c^{\varepsilon}\text{ and }f_{i}(z)=\Upsilonf_{i}^{\varepsilon}(z),\ \forall z\in\mathbb{R}^{p},\ i\in\{0,\ldots,k\},\label{eq:identified-set}
\end{equation}
and thus the shocks $\{u_{t}\}$ in \ref{eq:nlVAR} satisfy
\begin{equation}
u_{t}=f_{0}(z_{t})-c-\sum_{i=1}^{k}f_{i}(z_{t-i})=\Upsilon\left[f_{0}^{\varepsilon}(z_{t})-c^{\varepsilon}-\sum_{i=1}^{k}f_{i}^{\varepsilon}(z_{t-i})\right]=\Upsilon\varepsilon_{t}.\label{eq:u-to-epsilon}
\end{equation}
(For a discussion of the conditions under which the preceding holds,
see \ref{app:identification}: these are essentially weak smoothness
conditions on the elements of $\{\mathscr{F}_{i}\}_{i=0}^{k}$ and $\mathscr{R}$,
together with the requirement that $\{f_{i}^{\varepsilon}\}_{i=1}^{k}$be
such as to ensure sufficient dependence of the r.h.s.\ of the model
on lags of $z_{t}$.) Conversely, should \ref{eq:identified-set} fail
to hold, then there will be at least some realisations of $\{z_{t}\}$
for which the likelihoods of $(c,\{f_{i}\}_{i=0}^{k},\varrho)$ and
$(c^{\varepsilon},\{f_{i}^{\varepsilon}\}_{i=0}^{k},\varrho^{\varepsilon})$ will be distinct,
and so the data is to this extent informative about these two alternative
parametrisations of the model.\footnote{We do not\emph{ }claim, on the basis of \ref{thm:shockid}, that the
parameters of the model are consistently estimable up to an orthogonal
transformation. While it seems reasonable to suppose that consistent
nonparametric estimation of the model would be possible (under regularity
conditions) if $\{z_{t}\}$ is stationary and ergodic, the usual connection
between identification and consistent estimation is attenuated when
$\{z_{t}\}$ possesses some stochastic (or indeed, deterministic)
trends, because of the non-recurrence of those trends in higher dimensions
(see \citealp[Sec.~6]{Bing01Hdbk}; \citealp[p.~62]{GP13JoE}). Consistent
estimation of the model parameters (up to $\Upsilon$) would in such
cases likely require further restrictions on the functional form of
$f_{i}$.}
Accordingly, when subsequently discussing the problem of structural
shock identification, we shall suppose that the shocks $u_{t}$ appearing
in \ref{eq:nlVAR} are related to $\varepsilon_{t}$ in precisely the manner
of \ref{eq:u-to-epsilon}, yielding the \emph{structural nonlinear
VAR model}
\begin{align}
f_{0}(z_{t}) & =c+\sum_{i=1}^{k}f_{i}(z_{t-i})+u_{t} & u_{t} & =\Upsilon\varepsilon_{t}\label{eq:nlSVAR}
\end{align}
where $\Upsilon\in\mathbb{R}^{p\times p}$ is an orthogonal matrix. In
other words, on the strength of \ref{thm:shockid}, we shall regard
observation of $\{z_{t}\}$ as being sufficient to identify the model
parameters up to (and only up to) $\Upsilon$, and thence the structural
shocks up to transformation by this same matrix. To identify the structural
shocks, we thus seek additional restrictions sufficient to pin down
(at least some of) the unknown elements of $\Upsilon$, such as will
be provided by the long-run identifying restrictions developed in
\ref{subsec:long-run-id} below.
\section{Granger--Johansen representation theory}
\label{sec:common-trends}
\label{subsec:gjrt}
The starting point for the analysis of the linear cointegrated VAR
is the Granger--Johansen representation theorem (GJRT), which decomposes
$z_{t}$ into the sum of: (i) an initial condition, (ii) a stochastic
trend, and (iii) a weakly dependent process (which is stationary if
$z_{t}$ is suitably initialised). This may be rendered in various
ways: to facilitate the comparison with \ref{thm:gjrt} below (from
which it follows in the linear case), we shall here write this for
a linear VAR satisfying \ref{enu:lin:homeo}{\upshape{{\scalefont{0.76}--3}}}, as
\begin{equation}
z_{t}=\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}\mathrm{H}\\
\beta^{\mathsf{T}}
\end{bmatrix}^{-1}\left(\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}\abv{h}(\vec{z}_{0})+\alpha_{\perp}^{\mathsf{T}}\sum_{s=1}^{t}u_{s}\\
-\mu
\end{bmatrix}+S_{\chi}^{\mathsf{T}}\b{\xi}_{t}\right),\label{eq:GJ-type-rep}
\end{equation}
where: (i) $\abv{h}(\vec{z}_{0})$ captures the dependence on
the initial values $\vec{z}_{0}=(z_{0}^{\mathsf{T}},\ldots,z_{-k+1}^{\mathsf{T}})^{\mathsf{T}}$;
(ii) $\alpha_{\perp}^{\mathsf{T}}\sum_{s=1}^{t}u_{s}$ is a stochastic
trend of dimension $q=\operatorname{rk}\alpha_{\perp}=p-r$; and (iii)
\begin{equation}
\begin{bmatrix}\mu+\beta^{\mathsf{T}}z_{t}\\
\Delta\b{\zeta}_{t}
\end{bmatrix}=\b{\xi}_{t}=(I_{p(k-1)+r}+\b{\beta}^{\mathsf{T}}\b{\alpha})\b{\xi}_{t-1}+\b{\beta}^{\mathsf{T}}\b u_{t}\label{eq:eqerr-linear}
\end{equation}
follows a VAR, where $\b{\alpha}$ and $\b{\beta}$ are as in \ref{eq:lin-boldbeta}
above. The stability of this VAR follows from \ref{prop:lin-rep-var},
because membership of ${\cal M}_{r}$ implies, via \ref{enu:cvar:jsr},
that all the eigenvalues of $I_{p(k-1)+r}+\b{\beta}^{\mathsf{T}}\b{\alpha}$
must lie inside the unit circle. (It may also be noted from \ref{eq:bz-in-linear}
above that, in this case, $\Delta\b{\zeta}_{t}$ is itself a linear
function of $\{\Delta z_{t-i}\}_{i=0}^{k-2}$.) $\b{\xi}_{t}$ may
thus be rendered stationary through an appropriate choice of the initial
conditions ($\vec{z}_{0}$ or, equivalently, $\b z_{0}$).
Before providing our extension of the GJRT to the general setting
of \ref{eq:nlVAR}, we first note that the nonlinear counterpart of
$\b{\xi}_{t}$ will continue to follow an autoregression of the form
\ref{eq:eqerr-linear}, but now with time-varying coefficients, in
particular with $\b{\beta}=\b{\beta}_{t}$ now depending on the level
of $z_{t}$. While such processes cannot be stationary, we still have
a well-defined notion of \emph{stability} for such processes, which
is sufficient to ensure that $\b{\xi}_{t}$ remains of strictly smaller
stochastic order than the common trends.
\begin{defn}
\label{def:expstable}Suppose that $\{w_{t}\}_{t\in\ensuremath{\mathbb{N}}}$ follows
the time-varying VAR,
\begin{equation}
w_{t}=c_{t}+A_{t}w_{t-1}+B_{t}v_{t},\label{eq:wt-time-varying}
\end{equation}
where $A_{t}\in\set A\subset\mathbb{R}^{m\times m}$, $B_{t}\in\set B\subset\mathbb{R}^{m\times\ell}$,
and $c_{t}\in\set C\subset\mathbb{R}^{m}$ for all $t\in\ensuremath{\mathbb{N}}$, from
some given $w_{0}$. We say that $\{w_{t}\}$ is \emph{exponentially
stable} if $\set B$ and $\set C$ are bounded, and there exists a
$C<\infty$ and a $\rho\in[0,1)$ such that
\[
\norm{\prod_{s=1}^{t}A_{s}}\leq C\rho^{t},\ \forall t\in\ensuremath{\mathbb{N}}.
\]
\end{defn}
A sufficient condition for exponential stability is that $\rho_{{\scriptstyle \mathrm{JSR}}}(\set A)<1$
(e.g.\ \citealp{Jungers09}, Cor.~1.1), which motivates the restriction
on the JSR that appears in \ref{enu:cvar:jsr}. Notable consequences
are that if $v_{t}=0$ for all $t\geq\tau$, then $w_{t}\ensuremath{\rightarrow}0$,
while if $\{v_{t}\}$ and $w_{0}$ have uniformly bounded $2+\delta$
moments, then $\max_{1\leq t\leq n}\smlnorm{w_{t}}=o_{p}(n^{1/2})$.
(For this last, see Lemma~A.1 in \citealp{DMW22}.)
We now state our counterpart of the Granger--Johansen representation
theorem for the model \ref{eq:nlVAR}, which constitutes the main result
of this paper.
\begin{thm}
\label{thm:gjrt}Suppose \ref{ass:cvar} holds. Then $\chi:\mathbb{R}^{p}\ensuremath{\rightarrow}\mathbb{R}^{p}$
in \ref{eq:co-map} is a homeomorphism, and
\begin{equation}
z_{t}=\chi^{-1}\left(\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}\abv{h}(\vec{z}_{0})+\alpha_{\perp}^{\mathsf{T}}\sum_{s=1}^{t}u_{s}\\
-\mu
\end{bmatrix}+S_{\chi}^{\mathsf{T}}\b{\xi}_{t}\right),\label{eq:gjrt-z}
\end{equation}
where
\[
\abv{h}(\vec{z}_{s})\coloneqqf_{0}(z_{s})-\sum_{i=1}^{k-1}g_{i}(z_{s-i}),
\]
$S_{\chi}^{\mathsf{T}}$ is the $p\times[p(k-1)+r]$ matrix given by
\[
S_{\chi}^{\mathsf{T}}\coloneqq\begin{bmatrix}-\alpha_{\perp}^{\mathsf{T}} & 0\\
0 & I_{r}
\end{bmatrix}\begin{bmatrix}0_{p\times r} & I_{p} & \cdots & I_{p}\\
I_{r} & 0_{r\times p} & \cdots & 0_{r\times p}
\end{bmatrix},
\]
and
\[
\b{\xi}_{t}\coloneqq\b{\mu}+\b{\theta}(z_{t})+\b D_{0}\b z_{t}=\begin{bmatrix}\mu+\theta(z_{t})\\
\Delta\b{\zeta}_{t}
\end{bmatrix}\eqqcolon\begin{bmatrix}\xi_{t}\\
\Delta\b{\zeta}_{t}
\end{bmatrix}
\]
follows the exponentially stable VAR
\begin{equation}
\b{\xi}_{t}=(I_{p(k-1)+r}+\b{\beta}_{t}^{\mathsf{T}}\b{\alpha})\b{\xi}_{t-1}+\b{\beta}_{t}^{\mathsf{T}}\b u_{t}\label{eq:gjrt-xi}
\end{equation}
with $\b{\beta}_{t}\in{\cal B}$ for all $t\in\ensuremath{\mathbb{N}}$.
\end{thm}
\begin{rem}
\label{rem:-gjrt}
\refstepcounter{subremark}
(\roman{subremark}){}.
{}\label{subrem:gjrt-lin} That the preceding
specialises to \ref{eq:GJ-type-rep} in the linear case follows from
the fact that $\psi(z)=\alpha_{\perp}^{\mathsf{T}}\mathrm{H} z$ and $\theta(z)=\beta^{\mathsf{T}}z$
by \ref{prop:lin-rep-var}, and that ${\cal B}=\{\b{\beta}\}$ for
$\b{\beta}$ as in \ref{eq:lin-boldbeta}. To illustrate more clearly
the connection between \ref{eq:GJ-type-rep} and conventional statements
of the GJRT, we note that the part of the r.h.s.\ of \ref{eq:GJ-type-rep}
that depends on the common stochastic trends may be written as
\begin{equation}
\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}\mathrm{H}\\
\beta^{\mathsf{T}}
\end{bmatrix}^{-1}\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}\sum_{s=1}^{t}u_{s}\\
0_{r\times1}
\end{bmatrix}=\beta_{\perp}(\alpha_{\perp}^{\mathsf{T}}\mathrm{H}\beta_{\perp})^{-1}\alpha_{\perp}^{\mathsf{T}}\sum_{s=1}^{t}u_{s}\eqqcolon P_{\beta_{\perp}}\sum_{s=1}^{t}u_{s}\label{eq:gjrt-linear}
\end{equation}
by \ref{lem:lin-co-map} in \ref{app:examples}, which agrees precisely
with \citet[Ch.~4]{Joh95}.
\refstepcounter{subremark}
(\roman{subremark}){}.
{} Suppose that $\{u_{t}\}$ satisfies \ref{ass:coerr}.
Then it follows by Lemma~A.1 in \citet{DMW22} that $\max_{1\leq t\leq n}\smlnorm{\b{\xi}_{t}}=o_{p}(n^{1/2})$,
and so is strictly of smaller order than $\sum_{s=1}^{\smlfloor{n\lambda}}u_{s}$.
This is crucial for obtaining the limiting distribution of the standardised
process $n^{-1/2}z_{\smlfloor{n\lambda}}$: though here further assumptions
are required, because of the presence of the nonlinear map $\chi$
in \ref{eq:gjrt-z}. Results of this kind, which were obtained in the
setting of the CKSVAR by \citet{DMW22}, are the subject of a companion
paper to the present work.
\refstepcounter{subremark}
(\roman{subremark}){}.
{} Applying $\chi$ to both sides of \ref{eq:gjrt-z}, we
obtain
\[
\chi(z_{t})=\begin{bmatrix}\psi(z_{t})\\
\theta(z_{t})
\end{bmatrix}=\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}\abv{h}(\vec{z}_{0})+\alpha_{\perp}^{\mathsf{T}}\sum_{s=1}^{t}u_{s}-\alpha_{\perp}^{\mathsf{T}}\sum_{i=1}^{k-1}\Delta\zeta_{it}\\
\xi_{t}-\mu
\end{bmatrix}.
\]
Asymptotically, under \ref{ass:coerr}, the dominant component of $\psi(z_{t})$
would be the $q$ common trends $\alpha_{\perp}^{\mathsf{T}}\sum_{s=1}^{t}u_{s}$,
which would in turn dominate $\theta(z_{t})$, since the latter equals
the first $r$ components of the stable autoregressive process $\b{\xi}_{t}$.
In this sense, the action of $\chi$ separates $z_{t}$ into its `common
trend' and `equilibrium error' components. Since $\theta(z_{t})$
is purged of these common trends -- while being restricted by the
requirement that $\chi$ be a homeomorphism -- it may be said to provide
a representation of the `nonlinear cointegrating relations' that
exist between the elements of $z_{t}$.
\end{rem}
\section{Consequences of the representation theorem}
\label{sec:consequences}
\subsection{Attractor spaces}
\label{subsec:attractors}
A first application of \ref{thm:gjrt} is to verify the stability of
the (non-stochastic) steady state solutions to \ref{eq:nlVAR}. By
considering \ref{eq:VECM-first} when $z_{t-1}=\cdots=z_{t-k}=z$ for
some $z\in\mathbb{R}^{p}$ and $u_{t}=0$, we see that for $z$ to be
a steady state equilibrium, it must satisfy
\[
0=c+\pi(z)=\alpha[\mu+\theta(z)]
\]
or equivalently $\theta(z)=-\mu$, since $\operatorname{rk}\alpha=r$ and $\theta:\mathbb{R}^{p}\ensuremath{\rightarrow}\mathbb{R}^{r}$.
Thus the set of steady state equilibria is given by the $q$-dimensional
manifold
\begin{equation}
\set M_{\mu}\coloneqq\{z\mid\pi(z)=-c\}=\{z\in\mathbb{R}^{p}\mid\theta(z)=-\mu\}=\chi^{-1}(\mathbb{R}^{q}\times\{-\mu\})\label{eq:ctspc-mu}
\end{equation}
where the fact that $\mathscr{M}_{\mu}$ is indeed a $q$-dimensional manifold
follows from the final equality, since $\chi:\mathbb{R}^{p}\ensuremath{\rightarrow}\mathbb{R}^{p}$
is a homeomorphism.
For the purposes of analysing the stability of $\mathscr{M}_{\mu}$, we
shall suppose that the shocks $\{u_{t}\}$ are set to zero after some
$\tau\in\ensuremath{\mathbb{N}}$: that is, $u_{s}=0$ for all $s\geq\tau+1$, and
then consider how $z_{t}$ evolves from time $\tau$ forward. In other
words, we shall fix the state of the model at time $\tau-1$ at
\begin{equation}
\vec{z}_{\tau-1}=(z_{\tau-1}^{\mathsf{T}},z_{\tau-2}^{\mathsf{T}},\ldots,z_{\tau-k}^{\mathsf{T}})^{\mathsf{T}}=(\mathfrak{z}_{(1)}^{\mathsf{T}},\mathfrak{z}_{(2)}^{\mathsf{T}},\ldots,\mathfrak{z}_{(k)}^{\mathsf{T}})^{\mathsf{T}}\eqqcolon\mathfrak{z};\label{eq:tau-general-init}
\end{equation}
allow one final shock $u_{\tau}=u\in\mathbb{R}^{p}$ to occur at time
$\tau$; and then evaluate the (non-stochastic) limit of $z_{t}=z_{t}(u;\mathfrak{z})$
as $t\ensuremath{\rightarrow}\infty$. We aim to show that $\mathscr{M}_{\mu}$ is strictly
stable, in the sense of the following.
\begin{defn}
\label{def:stability}${\cal S}\subset\mathbb{R}^{p}$ is \emph{stable}
if for every $(u,\mathfrak{z})\in\mathbb{R}^{p}\times\mathbb{R}^{kp}$, there exists
a $z_{\infty}\in{\cal S}$ such that $z_{t}(u,\mathfrak{z})\ensuremath{\rightarrow} z_{\infty}\in{\cal S}$.
$z_{\infty}\in\mathbb{R}^{p}$ is a \emph{non-trivial attractor} if for
every $\mathfrak{z}\in\mathbb{R}^{kp}$, there exists a non-empty $\set U(z_{\infty};\mathfrak{z})\subset\mathbb{R}^{p}$
such that $z_{t}(u,\mathfrak{z})\ensuremath{\rightarrow} z_{\infty}$ for all $u\in\set U(z_{\infty};\mathfrak{z})$;
we term $\set U(z_{\infty};\mathfrak{z})$ the \emph{domain of attraction}
for $z_{\infty}$, given $\mathfrak{z}$. ${\cal S}$ is \emph{strictly
stable} if it is stable and contains only non-trivial attractors.
\end{defn}
In other words, if $\mathscr{M}_{\mu}$ is strictly stable, then whatever
the current state $\vec{z}_{\tau-1}=\mathfrak{z}$ of the model and the
given limiting $z_{\infty}\in\mathscr{M}_{\mu}$, there is always a value
for the $\tau$-dated shock $u_{\tau}$ that would lead $z_{t}$ to
converge to $z_{\infty}$; every element in $\mathscr{M}_{\mu}$ may thus
ultimately be `reached' from $\vec{z}_{\tau-1}$. (Note that this
is not so trivial a matter as choosing $u_{\tau}=u$ such that $z_{\tau}=z_{\infty}$
immediately, since what is required is that $z_{t}$ converge to a
\emph{steady state equilibrium} at $z_{\infty}$.) Since $\{\b{\xi}_{t}\}$
is exponentially stable by \ref{thm:gjrt}, it follows that $\b{\xi}_{t}\ensuremath{\rightarrow}0$
as $t\ensuremath{\rightarrow}\infty$ (see the discussion following \ref{def:expstable}
above). Hence, noting that $\sum_{s=\tau}^{t}u_{s}=u_{\tau}=u$, we
have by that result that
\[
z_{t}(u;\mathfrak{z})=\chi^{-1}\left(\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}\abv{h}(\mathfrak{z})+\alpha_{\perp}^{\mathsf{T}}u\\
-\mu
\end{bmatrix}+S_{\chi}^{\mathsf{T}}\b{\xi}_{t}\right)\ensuremath{\rightarrow}\chi^{-1}\left(\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}[\abv{h}(\mathfrak{z})+u]\\
-\mu
\end{bmatrix}\right)\eqqcolon z_{\infty}(u;\mathfrak{z})\in\mathscr{M}_{\mu}.
\]
Because the r.h.s.\ depends on $\mathfrak{z}$ only through $\abv{h}(\mathfrak{z})$,
and as $u\in\mathbb{R}^{p}$ varies $\alpha_{\perp}^{\mathsf{T}}[\abv{h}(\mathfrak{z})+u]$
ranges freely over $\mathbb{R}^{q}$, we can induce $z_{\infty}(u;\mathfrak{z})$
to take any desired value in $\mathscr{M}_{\mu}$. We thus obtain the following.
\begin{thm}
\label{thm:stab}Suppose \ref{ass:cvar} holds. Then
\begin{enumerate}
\item \label{enu:stab:cvg}for every $(u,\mathfrak{z})\in\mathbb{R}^{p}\times\mathbb{R}^{kp}$,
as $t\ensuremath{\rightarrow}\infty$
\begin{equation}
z_{t}(u;\mathfrak{z})\ensuremath{\rightarrow} z_{\infty}(u;\mathfrak{z})=\chi^{-1}\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}[\abv{h}(\mathfrak{z})+u]\\
-\mu
\end{bmatrix}\in\mathscr{M}_{\mu};\label{eq:zt-det-lim}
\end{equation}
\end{enumerate}
\begin{enumerate}[resume]
\item \label{enu:stab:dom}for any $(z,\mathfrak{z})\in\mathscr{M}_{\mu}\times\mathbb{R}^{kp}$
the $r$-dimensional affine subspace
\begin{equation}
\set U(z;\mathfrak{z})\coloneqq(\operatorname{sp}\alpha)+[h(z)-\abv{h}(\mathfrak{z})]\label{eq:domain-of-attraction}
\end{equation}
is the domain of attraction for $z$, given $\mathfrak{z}$;
\end{enumerate}
and hence $\mathscr{M}_{\mu}$ is strictly stable.
\end{thm}
\begin{rem}
\refstepcounter{subremark}
(\roman{subremark}){}.
{} In view of the preceding, we are justified in referring
to $\mathscr{M}_{\mu}$ as the \emph{attractor space} for $\{z_{t}\}$.
\refstepcounter{subremark}
(\roman{subremark}){}.
{} If the model is in a steady state equilibrium at some
$z\in\mathscr{M}_{\mu}$, immediately prior to the incidence of $u_{\tau}$,
so that $\mathfrak{z}_{(i)}=z\in\mathscr{M}_{\mu}$ for all $i\in\{1,\ldots,k-1\}$,
then
\[
\abv{h}(\mathfrak{z})=f_{0}(z)-\sum_{i=1}^{k-1}g_{i}(z)=h(z)
\]
and in this case the domain of attraction \ref{eq:domain-of-attraction}
reduces to $\operatorname{sp}\alpha$.
\refstepcounter{subremark}
(\roman{subremark}){}.
{} In the linear VAR, $\theta(z)=\beta^{\mathsf{T}}z$, and thus
the attractor space simplifies to the $q$-dimensional affine subspace
\[
\mathscr{M}_{\mu}=\{z\in\mathbb{R}^{p}\mid\beta^{\mathsf{T}}z=-\mu\},
\]
while the domain of attraction for $z_{\infty}\in\mathscr{M}_{\mu}$ is
the $r$-dimensional affine subspace
\[
\set U(z_{\infty};\mathfrak{z})=(\operatorname{sp}\alpha)+[\mathrm{H} z_{\infty}-\abv{h}(\mathfrak{z})].
\]
\end{rem}
\subsection{Long-run identifying restrictions}
\label{subsec:long-run-id}
As a direct consequence of the preceding, we obtain the following
characterisation of the linear combinations of the shocks $u_{t}$
in \ref{eq:nlVAR} that do \emph{not} have permanent effects.
\begin{cor}
\label{cor:subspace-noeffect}Suppose \ref{ass:cvar} holds. Then for
every $\mathfrak{z}\in\mathbb{R}^{p}$,
\[
\{u\in\mathbb{R}^{p}\mid z_{\infty}(u;\mathfrak{z})=z_{\infty}(0;\mathfrak{z})\}=\operatorname{sp}\alpha
\]
i.e.~a shock has no permanent effect on $z_{t}$, if and only if
it lies in $\operatorname{sp}\alpha$.
\end{cor}
The significance of this result may be explained as follows. Suppose
that, as in \ref{eq:nlSVAR} above, we have the nonlinear structural
VAR
\begin{align*}
f_{0}(z_{t}) & =c+\sum_{i=1}^{k}f_{i}(z_{t-i})+u_{t} & u_{t} & =\Upsilon\varepsilon_{t}\tag{\ref{eq:nlSVAR}}
\end{align*}
belonging to class ${\cal M}_{r}$. As noted in \ref{subsec:identification},
the parameters of this model are identified up to the unknown orthogonal
matrix $\Upsilon\in\mathbb{R}^{p\times p}$, about which the data is entirely
uninformative, and thus we seek additional restrictions that would
pin down (at least some of) the elements of $\Upsilon$. To that end,
suppose we partition the structural shocks as
\[
\varepsilon_{t}=(\varepsilon_{(1),t}^{\mathsf{T}},\varepsilon_{(2),t}^{\mathsf{T}})^{\mathsf{T}}
\]
where $\varepsilon_{(1),t}$ takes values in $\mathbb{R}^{m}$, and collects
the $m\in\{1,\ldots,r\}$ structural shocks that are regarded as having
\emph{no} permanent effect on $z_{t}$. Then \ref{cor:subspace-noeffect}
implies that
\begin{equation}
\Upsilon\begin{bmatrix}I_{m}\\
0_{(p-m)\times m}
\end{bmatrix}\subset\operatorname{sp}\alpha\implies\alpha_{\perp}^{\mathsf{T}}\Upsilon\begin{bmatrix}I_{m}\\
0_{(p-m)\times m}
\end{bmatrix}=0,\label{eq:lr-ident}
\end{equation}
which provides $qm$ `long-run identifying restrictions' on $\Upsilon$.\footnote{Note that \ref{cor:subspace-noeffect}, or equivalently \ref{eq:lr-ident},
implies that at most $r$ elements of $\varepsilon_{t}$ may be such that
the corresponding column of $\Upsilon$ lies in $\operatorname{sp}\alpha$, i.e.\ \emph{up
to} $r$ structural shocks may have purely transitory effects. As
is familiar from the linear structural VAR, there is nothing here
to preclude e.g.\ \emph{all} shocks from having permanent effects;
the number (and identities) of the $m\in\{0,\ldots,r\}$ structural
shocks having purely transitory effects will thus depend on the identifying
conditions that define those shocks.}
As discussed further in \ref{sec:without-the-crsc} below, the availability
of these long-run identifying restrictions is largely a consequence
of the common row space condition, and provides one of the principal
motivations for maintaining that condition in our model. Remarkably,
despite the nonlinearity permitted by \ref{eq:nlSVAR}, the identifying
restrictions \ref{eq:lr-ident} take exactly the same form as in a
linear SVAR (see \citealp{KL17book}, Sec.~10.2).
\subsection{Long-run multipliers}
\label{sec:lr-multipliers}
Having obtained a characterisation of the attractor space $\mathscr{M}_{\mu}$
for $\{z_{t}\}$, we may also say something more about the permanent
effect of a shock at time $t=\tau$. The \emph{limiting impulse responses}
or \emph{long-run multipliers} can be computed by differentiating
$z_{\infty}(u;\mathfrak{z})$ with respect to $u$, i.e.~by computing the
Jacobian
\begin{equation}
\partial_{u}\left.z_{\infty}(u;\mathfrak{z})\right|_{u=0}=\partial_{v}\left.\chi^{-1}\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}\abv{h}(\mathfrak{z})+v\\
-\mu
\end{bmatrix}\right|_{v=0}\alpha_{\perp}^{\mathsf{T}}\eqqcolon\nabla(\mathfrak{z})\alpha_{\perp}^{\mathsf{T}}.\label{eq:multipliers}
\end{equation}
In view of \ref{eq:nlSVAR}, postmultiplying this by $\Upsilon$ yields
the long-run multipliers with respect to the structural shocks $\varepsilon_{t}$,
at time $t=\tau$.
One concern we might have here, in the general nonlinear case, is
the apparent dependence of these long-run multipliers on $\vec{z}_{t-1}=\mathfrak{z}\in\mathbb{R}^{kp}$,
which potentially ranges over a very `large' ($kp$-dimensional)
space. However, it will be noted that $z_{\infty}(u;\mathfrak{z})$ only
depends on $\mathfrak{z}$ through $\alpha_{\perp}^{\mathsf{T}}\abv{h}(\mathfrak{z})$:
and as shown in the proof of the next result, it is always possible
to find a $z_{\mathfrak{z}}\in\mathscr{M}_{\mu}$ such that $\psi(z_{\mathfrak{z}})=\alpha_{\perp}^{\mathsf{T}}\abv{h}(\mathfrak{z})$.
Therefore, for the purposes of computing the full set of possible
long-run effects of shocks in the model, it suffices to consider cases
in which the model is in a steady state equilibrium (i.e.\ where
$z_{\tau-1}=\dots=z_{\tau-k}=z$ for some $z\in\mathscr{M}_{\mu}$) immediately
prior to the incidence of the shock.
Because $\chi^{-1}$ is not necessarily differentiable everywhere,
the long-run multipliers as defined in \ref{eq:multipliers} may not
exist for every $z\in\mathscr{M}_{\mu}$. However, in most cases of interest,
the subset $N\subset\mathscr{M}_{\mu}$ at which this occurs will be exceptionally
small, corresponding e.g.\ in the case of piecewise affine models
(see \ref{subsec:piecewise-affine} below) to points exactly on the
boundary between two regimes. It is also possible that the Jacobian
of $\chi^{-1}$ may fail to be invertible at exceptional points: though
it must be invertible almost everywhere, since $\chi$ is invertible.
\begin{thm}
\label{thm:multipliers}Suppose \ref{ass:cvar} holds. Let $N\subset\mathscr{M}_{\mu}$
denote the set of all $z\in\mathscr{M}_{\mu}$ at which
\[
v\ensuremath{\mapsto}\chi^{-1}\begin{bmatrix}\psi(z)+v\\
-\mu
\end{bmatrix}
\]
is \emph{not} differentiable with respect to $v$, when $v=0$, and
$N_{0}$ denote the union of $N$ with the set of $z\in\mathscr{M}_{\mu}\backslash N$
at which the Jacobian of the preceding is non-invertible. Then
\begin{enumerate}
\item \label{enu:mult:set}the full set of long-run multiplier matrices
is given by
\begin{align}
\Theta_{\infty}(z)\coloneqq\partial_{u}\left.\chi^{-1}\begin{bmatrix}\psi(z)+\alpha_{\perp}^{\mathsf{T}}u\\
-\mu
\end{bmatrix}\right|_{u=0} & =\partial_{v}\left.\chi^{-1}\begin{bmatrix}\psi(z)+v\\
-\mu
\end{bmatrix}\right|_{v=0}\alpha_{\perp}^{\mathsf{T}}\label{eq:lrmult}
\end{align}
for $z\in\mathscr{M}_{\mu}\backslash N$; and
\item \label{enu:mult:rank}$\operatorname{rk}\Theta_{\infty}(z)=q$ for every $z\in\mathscr{M}_{\mu}\backslash N_{0}$.
\end{enumerate}
\end{thm}
\begin{rem}
\refstepcounter{subremark}
(\roman{subremark}){}.
{} In the linear VAR, it follows by \ref{prop:lin-rep-var}
and \ref{lem:lin-co-map} that
\[
\partial_{u}\left.z_{\infty}(u;\mathfrak{z})\right|_{u=0}=\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}\mathrm{H}\\
\beta^{\mathsf{T}}
\end{bmatrix}^{-1}\partial_{u}\left.\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}[\abv{h}(\mathfrak{z})+u]\\
-\mu
\end{bmatrix}\right|_{u=0}=\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}\mathrm{H}\\
\beta^{\mathsf{T}}
\end{bmatrix}^{-1}\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}\\
0
\end{bmatrix}=\beta_{\perp}(\alpha_{\perp}^{\mathsf{T}}\mathrm{H}\beta_{\perp})^{-1}\alpha_{\perp}^{\mathsf{T}}
\]
consistent with \ref{eq:gjrt-linear} above. In this case, the limiting
IRF is invariant to the state of the process prior to the incidence
of the final shock\@.
\refstepcounter{subremark}
(\roman{subremark}){}.
{} In nonlinear VARs such as \ref{eq:nlVAR}, impulse responses
with respect to a shock incident at time $\tau$ are well known to
be dependent on both the current state $\vec{z}_{\tau-1}$ of the
model, and on the shocks that occur subsequent to time $\tau$. In
computing the long-run multipliers above, we have allowed for dependence
on the current state, but deliberately forced $u_{t}=0$ for $t\geq\tau+1$.
This not only simplifies the analysis but, we would argue, provides
a reasonable basis on which to determine whether the structural VAR
permits certain (small) shocks to have permanent effects on the model
variables, particularly in a nonstationary model of this kind, where
subsequent shocks may push $z_{t}$ in any region of the state space
(though it will, in some sense, still remain `attracted' to $\mathscr{M}_{\mu}$).
Nonetheless, let us suppose that, following \citet{KPP96JoE}, one
is also interested in `generalised' impulse responses that are computed
conditional on $\vec{z}_{\tau-1}=\mathfrak{z}$, but which average over
all (potential) histories of the shocks subsequent to $t=\tau$.
Then while we could not hope to obtain as clean a characterisation
of the limiting IRF as is given by the preceding theorem, the fact
that the major implication of \ref{eq:multipliers}, that
\[
\ker\partial_{u}\left.z_{\infty}(u;\mathfrak{z})\right|_{u=0}=\alpha
\]
is invariant to the state of the process suggests that this property
would continue to hold for the generalised impulse responses. However,
because of the possible long-range dependence of $\b{\xi}_{t}$ on
past shocks, via $\b{\beta}_{t}$ (and hence $z_{t}$), these calculations
are far from straightforward, and are therefore deferred to future
work.
\end{rem}
\section{The common row space condition}
\label{sec:without-the-crsc}
As introduced in \ref{subsec:framework} above, a key characteristic
of models in class ${\cal M}_{r}$ is the common row space condition
(CRSC), that $\operatorname{im}\pi=\{\pi(z)\mid z\in\mathbb{R}^{p}\}$ is an $r$-dimensional
linear subspace of $\mathbb{R}^{p}$; we noted there that no such restriction
is imposed on $\ker\pi$. By contrast, the existing literature on
nonlinear VECM models effectively reverses this condition by requiring
that $\ker\pi$ be a $q$-dimensional subspace of $\mathbb{R}^{p}$, without
necessarily restricting $\operatorname{im}\pi$ (see in particular \citealp{KR10JoE}).
Loosely speaking, if we suppose that $\pi$ may be decomposed, for
each $z\in\mathbb{R}^{p}$, as
\begin{equation}
\pi(z)=\alpha(z)\beta(z)^{\mathsf{T}}\label{eq:factor}
\end{equation}
where $\alpha,\beta:\mathbb{R}^{p}\ensuremath{\rightarrow}\mathbb{R}^{r}$, then the CRSC permits
taking $\alpha(z)=\alpha$, if $\beta(z)$ is appropriately normalised,
but leaves $\beta(z)$ otherwise unrestricted;\footnote{If $\beta(z)$ is instead normalised in some other way, such as e.g.\ $\beta(z)^{\mathsf{T}}=[I_{r},-A(z)]$,
then $\alpha(z)$ may indeed vary with $z$. In this sense, it is
not quite correct to regard the CRSC as requiring the loadings on
the equilibrium errors to be \emph{constant}, but rather that those
loadings should lie in a fixed $r$-dimensional linear subspace; there
may be nonlinear adjustment towards equilibrium, provided that it
occurs along this subspace.} whereas in the nonlinear VECM literature, $\beta(z)=\beta$ is constant,
with the result that there is a globally linear cointegrating space.
The representation theory developed in the nonlinear VECM literature
is thus strictly complementary to the present work. (Note that even
the case where $\beta(z)=\beta$ in our framework is \emph{not} encompassed
by that literature, because we allow the dynamics of the implied VECM
\ref{eq:VECM-first} to depend on the \emph{level} of $z_{t}$, whereas
that literature requires that these be governed entirely by the stationary
equilibrium errors $\beta^{\mathsf{T}}z_{t}$ and differences $\Delta z_{t}$.)
Recall that in the linear structural VAR model, common stochastic
trends arise because some subset of the structural shocks (say, $q$
in number) have permanent effects on $z_{t}$, while the remaining
$r=p-q$ shocks have only transitory effects. This provides not only
an underpinning economic explanation for the patterns of long-run
co-movement present in the data, but also a fruitful source of long-run
identifying restrictions (following \citealp{BlanchardQuah89}, and
\citealp{KPSW91AER}). The results developed in Sections~\ref{sec:common-trends}
and \ref{sec:consequences} above show that these properties are preserved
when we generalise the cointegrated linear SVAR to a nonlinear SVAR
\ref{eq:nlSVAR} belonging to class ${\cal M}_{r}$. However, as the example
developed in this section illustrates, these properties are highly
sensitive to departures from the CRSC, even when the assumption of
a globally linear cointegrating space is maintained (i.e.\ when $\beta(z)=\beta$
for all $z\in\mathbb{R}^{p}$ in \ref{eq:factor} above). In models where
the CRSC fails, it will generally be the case that the \emph{direction}
in which an impulse $u_{t}=\delta$ has no permanent effect on $z_{t}$
will vary with the \emph{magnitude} $\delta$. Thus, if we want to
work with nonlinear SVAR models in which structural shocks may be
distinguished on the basis of their long-run effects -- and in which,
as a corollary, common stochastic trends arise because only a subset
of these shocks have permanent effects -- then it would appear that
the CRSC cannot be easily dispensed with.
To illustrate the consequences of a failure of the CRSC, we consider
a nonlinear VAR(1), specified in VECM form as
\[
\Delta z_{t}=a(\beta^{\mathsf{T}}z_{t-1})\beta^{\mathsf{T}}z_{t-1}+u_{t}
\]
where for $\Lambda:\mathbb{R}^{r}\ensuremath{\rightarrow}[0,1]$ a smooth function satisfying
$\Lambda(0)=0$ and $\lim_{\smlnorm{\xi}\ensuremath{\rightarrow}\infty}\Lambda(\xi)=1$,
and $\alpha,\tilde{\alpha}\in\mathbb{R}^{p\times r}$, each having rank
$r$,
\[
a(\xi)\coloneqq\Lambda(\xi)\tilde{\alpha}+[1-\Lambda(\xi)]\alpha,
\]
so that the model is one in which the kernel of $\pi(z)=a(\beta^{\mathsf{T}}z)\beta^{\mathsf{T}}$
is a fixed $q$-dimensional subspace (given by $\operatorname{sp}\beta$), but
the image of $\pi$ is not a linear subspace. (If we further suppose
that the eigenvalues of $I_{r}+\beta^{\mathsf{T}}\alpha$ lie strictly
inside the unit circle, then the model satisfies conditions (A.2)
and (A.3) of \citet{KR10JoE}, who provide a GJRT for these models.)
This model may be regarded as smoothly combining two regimes: an `inner'
or `near equilibrium' regime in which
\begin{align*}
\Delta z_{t} & =\tilde{\alpha}\beta^{\mathsf{T}}z_{t-1}+u_{t} & u_{t} & =\Upsilon\varepsilon_{t}
\end{align*}
and an `outer' or `far from equilibrium' regime where $\tilde{\alpha}$
is replaced by $\alpha$. If the eigenvalues of $I_{r}+\beta^{\mathsf{T}}\tilde{\alpha}$
associated to the inner regime are also strictly inside the unit circle,
then $\mathscr{M}_{0}=\beta_{\perp}$ is a strictly stable attractor in
the sense of \ref{def:stability} above.
Now suppose we were to repeat the analysis of \ref{subsec:attractors}
for this model: i.e.\ fixing the state $z_{\tau-1}=\mathfrak{z}\in\mathbb{R}^{p}$
of the model in time $\tau$, having the model impacted by a final
shock $u_{\tau}=u$, and then computing $\lim_{t\ensuremath{\rightarrow}\infty}z_{t}(u;\mathfrak{z})=z_{\infty}(u;\mathfrak{z})$
in the absence of any further shocks (so $u_{t}=0$ for all $t\geq\tau+1$).
We may then ask: what values of $u$ would leave $z_{\infty}(u;\mathfrak{z})$
unchanged? That is, we would like to determine the set
\[
\{u\in\mathbb{R}^{p}\mid z_{\infty}(u;\mathfrak{z})=z_{\infty}(0;\mathfrak{z})\}.
\]
Let us further suppose that $\beta^{\mathsf{T}}z_{\tau-1}=\beta^{\mathsf{T}}\mathfrak{z}=0$,
so that $z_{\tau-1}\in\mathscr{M}_{0}$, and the model is in equilibrium
prior to the incidence of $u_{\tau}$. Locally to $\mathscr{M}_{0}$, we
would expect shocks in the direction of $\operatorname{sp}\tilde{\alpha}$ to have
no permanent effects; whereas further from $\mathscr{M}_{0}$, shocks in
directions lying progressively closer to $\operatorname{sp}\alpha$ should have
no permanent effects.
\begin{figure}
\begin{centering}
\includegraphics[viewport=150bp 180bp 692bp 415bp,clip,scale=0.8]{figures/nl-vecm}
\par\end{centering}
\caption{Directions of shocks having only transitory effects}
\label{fig:transitory}
\end{figure}
\ref{fig:transitory} illustrates this is indeed the case for a bivariate
($p=2$) model with
\begin{align*}
\tilde{\alpha} & =-\begin{bmatrix}1\\
0.5
\end{bmatrix} & \alpha & =-\begin{bmatrix}1\\
0.25
\end{bmatrix} & \beta & =\begin{bmatrix}1\\
-1
\end{bmatrix},
\end{align*}
and $\Lambda(\xi)=2\smlabs{\Phi(\xi)-0.5}$, where $\Phi$ denotes
the Gaussian cdf. (Without loss of generality, we take $\mathfrak{z}=0$.)
For values of $\smlnorm u$ between $0$ and $10$, the figure reports
the (unique) direction $\delta_{\smlnorm u}$ such that $u_{\tau}=\smlnorm u\delta_{\smlnorm u}$
has no permanent effect on $z_{t}$; for the sake of comparability
with $\tilde{\alpha}$ and $\alpha$, both of which have unit first
element, this is reported in terms of the ratio $\delta_{\smlnorm u,2}/\delta_{\smlnorm u,1}$.
We see here that $\delta_{\smlnorm u,2}/\delta_{\smlnorm u,1}$ takes
the value $0.5=\tilde{\alpha}_{2}$ when $\smlnorm u=0$, and tends
towards $0.25=\alpha_{1}$ as $\smlnorm u$ grows (equalling $0.26$
when $\smlnorm u=20$). As a consequence
\[
\ensuremath{\bigcap}_{u\in\mathbb{R}^{p}}\operatorname{sp}\delta_{\smlnorm u}=\emptyset,
\]
and as such there is \emph{no} direction along which the shocks $u_{t}$
will only have transient effects. There is thus no possibility of
discriminating between the underlying structural shocks $\varepsilon_{t}$
according to whether not they have permanent effects: each must have
some such effect, to an extent that varies with the magnitude of the
shock.
\section{Application to piecewise affine VARs}
\label{subsec:piecewise-affine}
Finally, we introduce a class of regime-switching models in which
the conditions required for our results may be verified relatively
straightforwardly. Suppose that for each $i\in\{0,\ldots,k\}$,
\begin{equation}
f_{i}(z)=\sum_{\ell=1}^{L}\ensuremath{\mathbf{1}}\{z\in\set Z^{(\ell)}\}(\bar{\phi}_{i}^{(\ell)}+\Phi_{i}^{(\ell)}z),\label{eq:pwa}
\end{equation}
where $\{\set Z^{(\ell)}\}_{\ell=1}^{L}$ is a collection of convex
sets that partition $\mathbb{R}^{p}$, $\{\bar{\phi}_{i}^{(\ell)}\}_{\ell=1}^{L}\subset\mathbb{R}^{p}$
and $\{\Phi_{i}^{(\ell)}\}_{\ell=1}^{L}\subset\mathbb{R}^{p\times p}$.\footnote{If instead $f_{i}(z)=\sum_{\ell=1}^{L}\ensuremath{\mathbf{1}}\{z\in\set Z_{i}^{(\ell)}\}(\bar{\phi}_{i}^{(\ell)}+\Phi_{i}^{(\ell)}z)$,
with a partition $\{\set Z_{i}^{(\ell)}\}_{\ell=1}^{L_{i}}$ that
depends on $i$, the model can nonetheless be written in terms of
$\{f_{i}\}_{i=0}^{k}$ of the form \ref{eq:pwa}, i.e.\ with a lag-independent
partition $\{\set Z^{(\ell)}\}_{\ell=1}^{L}$, by forming each $\set Z^{(\ell)}$
from a refinement of the sets in $\{\set Z_{i}^{(\ell)}\}_{\ell=1}^{L_{i}}$,
as $i$ ranges over $\{0,\ldots,k\}$.} When these parameters are such that $f_{i}$ is continuous for
each $i\in\{0,\ldots,k\}$, we shall say that each $f_{i}$ is a
\emph{piecewise affine function}, and refer to the model as a \emph{piecewise
affine VAR}. (We do \emph{not} consider cases in which $f_{i}$
may be discontinuous; so continuity is always implied when a function
or VAR is described as being piecewise affine.)
Piecewise affine VARs, of which the CKSVAR of \citet{SM21} is a recent
instance, provide a flexible but tractable means of introducing nonlinearity
into a vector autoregressive model. Defining
\[
\ensuremath{\mathbf{1}}^{(\ell)}(z)\coloneqq\ensuremath{\mathbf{1}}\{z\in\set Z^{(\ell)}\},
\]
we see that
\[
f_{0}(z_{t})-\sum_{i=1}^{k}f_{i}(z_{t-i})=\sum_{\ell=1}^{L}\ensuremath{\mathbf{1}}^{(\ell)}(z_{t})(\bar{\phi}_{i}^{(\ell)}+\Phi_{i}^{(\ell)}z_{t})-\sum_{i=1}^{k}\sum_{\ell=1}^{L}\ensuremath{\mathbf{1}}^{(\ell)}(z_{t-i})(\bar{\phi}_{i}^{(\ell)}+\Phi_{i}^{(\ell)}z_{t-i}).
\]
This is a kind of endogenous regime-switching VAR, in which the sets
$\{\set Z^{(\ell)}\}_{\ell=1}^{L}$ demarcate $L$ distinct `regimes'.
However, unlike in typical models of this kind, there is no requirement
that the same `regime' apply e.g.\ to $z_{t}$ and $z_{t-1}$ --
each may lie in a different member of $\{\set Z^{(\ell)}\}_{\ell=1}^{L}$
-- and so it might be more correct to say that there are a total
of $L^{k}$ autoregressive regimes, once all the possible patterns
of membership of $\{z_{t-i}\}_{i=0}^{k}$ in the sets $\{\set Z^{(\ell)}\}_{\ell=1}^{L}$
are allowed for. Nonetheless, it turns out that a fruitful approach
to analysing the long-run behaviour of these systems focuses attention
on those $L$ `linear submodels' that arise when each of $\{z_{t-i}\}_{i=0}^{k}$
lie in the \emph{same} $\set Z^{(\ell)}$, to each $\ell\in\{1,\ldots,L\}$
of which we may associate the autoregressive polynomial
\begin{equation}
\Phi^{(\ell)}(\lambda)\coloneqq\Phi_{0}^{(\ell)}-\sum_{i=1}^{k}\Phi_{i}^{(\ell)}\lambda^{i}.\label{eq:Phi-ell}
\end{equation}
\subsection{Simplification of the JSR condition}
The conditions for membership of ${\cal M}_{r}$, for piecewise affine
VARs, will parallel \ref{enu:lin:homeo}{\upshape{{\scalefont{0.76}--3}}}. Before stating
these, we first give a result that reduces \ref{enu:cvar:jsr} to
a condition on the JSR of a collection of only $L$ matrices (rather
than a collection of uncountably many matrices). To state it, observe
that
\begin{align}
\pi(z) & =-\sum_{\ell=1}^{L}\ensuremath{\mathbf{1}}^{(\ell)}(z)\left[\left(\bar{\phi}_{0}^{(\ell)}-\sum_{i=1}^{k}\bar{\phi}_{i}^{(\ell)}\right)+\left(\Phi_{0}^{(\ell)}-\sum_{i=1}^{k}\Phi_{i}^{(\ell)}\right)z\right]\nonumber \\
& \eqqcolon\sum_{\ell=1}^{L}\ensuremath{\mathbf{1}}^{(\ell)}(z)(\bar{\pi}^{(\ell)}+\Pi^{(\ell)}z).\label{eq:pi-piecewise}
\end{align}
In this context, we note that \ref{enu:cvar:rank} effectively requires
that
\begin{equation}
\Pi^{(\ell)}=-\Phi^{(\ell)}(1)=\alpha\beta^{(\ell)\mathsf{T}}\label{eq:Pi-ell}
\end{equation}
where $\alpha,\beta^{(\ell)}\in\mathbb{R}^{p\times r}$ with $\operatorname{rk}\alpha=\operatorname{rk}\beta^{(\ell)}=r$.
Further, let $\Gamma_{j}^{(\ell)}\coloneqq-\sum_{j=i+1}^{k}\Phi_{j}^{(\ell)}$,
$\b{\Gamma}^{(\ell)}\coloneqq[\Gamma_{i}^{(\ell)}]_{i=1}^{k-1}$, and
\begin{align}
\b{\beta}^{(\ell)\mathsf{T}} & \coloneqq\begin{bmatrix}\beta^{(\ell)\mathsf{T}}(\Phi_{0}^{(\ell)})^{-1} & 0\\
\b{\Gamma}^{(\ell)}(\Phi_{0}^{(\ell)})^{-1} & D
\end{bmatrix} & \b{\alpha} & \coloneqq\begin{bmatrix}\alpha & E^{\mathsf{T}}\\
0 & I_{p(k-1)}
\end{bmatrix}.\label{eq:beta_l}
\end{align}
\begin{lem}
\label{lem:affine-conditions}Suppose that $(c,\{f_{i}\}_{i=0}^{k})$
is a piecewise affine model, for which there exists an $\alpha\in\mathbb{R}^{p\times r}$
with $\operatorname{rk}\alpha=r$ such that
\begin{enumerate}
\item $c=\alpha\mu$, $\bar{\pi}^{(\ell)}=\alpha\bar{\mu}^{(\ell)}$ and
$\Pi^{(\ell)}=\alpha\beta^{(\ell)\mathsf{T}}$, for $\mu,\bar{\mu}^{(\ell)},\beta^{(\ell)}\in\mathbb{R}^{p\times r}$,
for all $\ell\in\{1,\ldots,L\}$; and
\item $f_{0}:\mathbb{R}^{p}\ensuremath{\rightarrow}\mathbb{R}^{p}$ is a homeomorphism.
\end{enumerate}
Then $(c,\{f_{i}\}_{i=0}^{k})$ satisfies \ref{eq:gradient} with ${\cal B}$
equal to $\operatorname{co}\{\b{\beta}^{(\ell)}\}_{\ell=1}^{L}$, and so
\[
\rho_{{\scriptstyle \mathrm{JSR}}}(\{I_{p(k-1)+r}+\b{\beta}^{\mathsf{T}}\b{\alpha}\mid\b{\beta}\in{\cal B}\})=\rho_{{\scriptstyle \mathrm{JSR}}}(\{I_{p(k-1)+r}+\b{\beta}^{(\ell)\mathsf{T}}\b{\alpha}\}_{\ell=1}^{L}).
\]
\end{lem}
Since the $p(k-1)+r$ eigenvalues of $I_{p(k-1)+r}+\b{\beta}^{(\ell)\mathsf{T}}\b{\alpha}$
correspond to the inverses of the non-unit roots of $\Phi^{(\ell)}(\lambda)$,
a necessary, though not sufficient condition for
\[
\rho_{{\scriptstyle \mathrm{JSR}}}(\{I_{p(k-1)+r}+\b{\beta}^{(\ell)\mathsf{T}}\b{\alpha}\}_{\ell=1}^{L})<1
\]
is that the eigenvalues of $I_{p(k-1)+r}+\b{\beta}^{(\ell)\mathsf{T}}\b{\alpha}$
should lie strictly inside the unit circle, for all $\ell\in\{1,\ldots,L\}$.
Thus, if $\operatorname{rk}\Phi^{(\ell)}(1)=q=p-r$, as per \ref{eq:Pi-ell}, then
$\Phi^{(\ell)}(\lambda)$ will have $q$ roots at unity, and all others
strictly outside the unit circle (this follows e.g.\ from arguments
given in the proof of \ref{prop:lin-rep-var}). In other words, for
a piecewise affine model to belong to class ${\cal M}_{r}(\alpha,\abv b,\abv{\rho})$
(for some $\abv{\rho}<1$), it is essentially necessary that each
of its $L$ linear submodels satisfy the usual conditions for a linear
VAR to give rise to $q$ common trends and $r$ cointegrating relations
(i.e.\ conditions \ref{enu:lin:rank} and \ref{enu:lin:roots} above).
\subsection{Membership of $\protect{\cal M}_{r}$}
It remains to consider \ref{enu:cvar:homeo}, i.e.\ the requirement
that $f_{0}$ be a homeomorphism. For the purposes of verifying
this condition, two important special cases of the piecewise affine
VAR are:
\begin{itemize}
\item the \emph{piecewise linear VAR} (PLVAR), in which there exists a basis
$\{a_{i}\}_{i=1}^{p}$ for $\mathbb{R}^{p}$ such that each $\set Z^{(\ell)}$
can be written as a union of cones of the form
\[
\set C_{{\cal I}}\coloneqq\{z\in\mathbb{R}^{p}\mid a_{i}^{\mathsf{T}}z\geq0,\ \forall i\in{\cal I}\text{ and }a_{i}^{\mathsf{T}}z<0,\ \forall i\notin{\cal I}\}
\]
where ${\cal I}$ ranges over the subsets of $\{1,\ldots,p\}$, and
$\bar{\phi}_{i}^{(\ell)}=0$ for all $i$ and $\ell$; and
\item the \emph{threshold affine VAR} (TAVAR), in which there exists an
$a\in\mathbb{R}^{p}\backslash\{0\}$ and thresholds $\{\tau_{\ell}\}_{\ell=0}^{L}$
with $\tau_{\ell}<\tau_{\ell+1}$, $\tau_{0}=-\infty$ and $\tau_{L}=+\infty$,
such that
\[
\set Z^{(\ell)}=\{z\in\mathbb{R}^{p}\mid a^{\mathsf{T}}z\in(\tau_{\ell-1},\tau_{\ell}]\},
\]
i.e.\ the sets $\{\set Z^{(\ell)}\}$ take the forms of `bands'
in $\mathbb{R}^{p}$. (In typical examples, $a=e_{p,i}$, i.e.\ it picks
out one `threshold variable' from the elements of $z_{t}$.)
\end{itemize}
In these models, the results of \citet{GLM80Ecta} provide necessary
and sufficient conditions for $f_{0}$ to be invertible, which can
be expressed in terms of the determinants of $\{\Phi_{0}^{(\ell)}\}_{\ell=1}^{L}$.
We thus have the following
\begin{prop}
\label{prop:affine}Suppose is $(c,\{f_{i}\}_{i=0}^{k})$ is either
a piecewise linear or threshold affine VAR, such that:
\begin{enumerate}[label={\upshape{{\scalefont{0.76}PWA.\arabic*}}}, leftmargin=1.75cm]
\item \label{enu:pwa:homeo}$\operatorname{sgn}\det\Phi_{0}^{(\ell)}=\operatorname{sgn}\det\Phi_{0}^{(1)}\neq0$
for all $\ell\in\{1,\ldots,L\}$;
\item \label{enu:pwa:rank}$c=\alpha\mu$, $\bar{\pi}^{(\ell)}=\alpha\bar{\mu}^{(\ell)}$
, and $\Pi^{(\ell)}=\alpha\beta^{(\ell)\mathsf{T}}$, where $\alpha,\beta^{(\ell)}\in\mathbb{R}^{p\times r}$
have rank $r$, for all $\ell\in\{1,\ldots,L\}$; and
\item \label{enu:pwa:jsr}$\rho_{{\scriptstyle \mathrm{JSR}}}(\{I_{p(k-1)+r}+\b{\beta}^{(\ell)\mathsf{T}}\b{\alpha}\}_{\ell=1}^{L})\leq\abv{\rho}$.
\end{enumerate}
Then $(c,\{f_{i}\}_{i=0}^{k})$ belongs to class ${\cal M}_{r}(\alpha,\abv b,\abv{\rho})$,
for $\abv b\geq\max_{1\leq\ell\leq L}\smlnorm{\b{\beta}^{(\ell)}}$,
with
\begin{equation}
\chi(z)=\begin{bmatrix}\psi(z)\\
\theta(z)
\end{bmatrix}=\sum_{\ell=1}^{L}\ensuremath{\mathbf{1}}^{(\ell)}(z)\left\{ \begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}\bar{\psi}^{(\ell)}\\
\bar{\mu}^{(\ell)}
\end{bmatrix}+\begin{bmatrix}\alpha_{\perp}^{\mathsf{T}}\mathrm{H}^{(\ell)}\\
\beta^{(\ell)\mathsf{T}}
\end{bmatrix}z\right\} \eqqcolon\sum_{\ell=1}^{L}\ensuremath{\mathbf{1}}^{(\ell)}(z)\{\bar{\chi}^{(\ell)}+\mathrm{X}^{(\ell)}z\}\label{eq:co-pwa}
\end{equation}
where
\begin{align*}
\bar{\psi}^{(\ell)} & \coloneqq\bar{\phi}_{0}^{(\ell)}-\sum_{i=1}^{k-1}\bar{\gamma}_{i}^{(\ell)} & \mathrm{H}^{(\ell)} & \coloneqq\Phi_{0}^{(\ell)}-\sum_{i=1}^{k-1}\Gamma_{i}^{(\ell)}
\end{align*}
for $\bar{\gamma}_{i}^{(\ell)}\coloneqq-\sum_{j=i+1}^{k}\bar{\phi}_{j}^{(\ell)}$.
Moreover, the same conclusion holds if $(c,\{f_{i}\}_{i=0}^{k})$
is a general piecewise affine model satisfying \ref{enu:pwa:rank}
and \ref{enu:pwa:jsr} above, if $f_{0}$ is a homeomorphism.
\end{prop}
Since it may be verified that a linear VAR satisfying \ref{enu:lin:homeo}{\upshape{{\scalefont{0.76}--3}}}
is itself a piecewise linear model satisfying \ref{enu:pwa:homeo}{\upshape{{\scalefont{0.76}--3}}}
above, \ref{prop:lin-rep-var} may be construed as a special case of
the preceding (though for expository reasons, a separate proof of
that result is given in \ref{app:examples}).
\subsection{Smooth transitions}
The model \ref{eq:pwa} may also be extended to allow for smooth transitions
between the $L$ regimes. In the literature on smooth transition (vector)
autoregressive models, the conventional approach (e.g.\ \citealp[Sec.~3.3]{HT13})
is to replace the indicator functions $\ensuremath{\mathbf{1}}\{z\in\set Z^{(\ell)}\}$
by smooth maps $\Lambda^{(\ell)}(z)$, so that now
\[
f_{i}^{\mathrm{ST}}(z)=\sum_{\ell=1}^{L}\Lambda^{(\ell)}(z)(\bar{\phi}_{i}^{(\ell)}+\Phi_{i}^{(\ell)}z),
\]
where $\Lambda^{(\ell)}(z)\in[0,1]$ and $\sum_{\ell=1}^{L}\Lambda^{(\ell)}(z)=1$
for all $z\in\mathbb{R}^{p}$, so that $f_{i}^{\mathrm{ST}}(z)$ is
a always a smooth, convex combination of the affine functions $\{z\ensuremath{\mapsto}\bar{\phi}_{i}^{(\ell)}+\Phi_{i}^{(\ell)}z\}_{\ell=1}^{L}$.
While models of this kind may be accommodated within our framework
(under certain regularity conditions), the fact that the \emph{gradient}
of $f_{i}^{\mathrm{ST}}$ is \emph{not} a convex combination of
those underlying affine regimes makes it difficult to reduce \ref{enu:cvar:jsr}
to a bound on the JSR of a finite collection of matrices, in the manner
of \ref{prop:affine}. As an alternative specification that allows
for smooth transitions between regimes, while also keeping \ref{enu:cvar:jsr}
tractable, consider
\begin{equation}
f_{i,K}(z)\coloneqq\int_{\mathbb{R}^{p}}f_{i}(z+u)K(u)\ensuremath{\,\ensuremath{\mathrm{d}}} u\label{eq:smoothed}
\end{equation}
where $f_{i}$ is a (continuous) piecewise affine function as in
\ref{eq:pwa} above, and $K$ is a smooth kernel with mean zero. Then
we have the following.
\begin{prop}
\label{prop:smoothed}Suppose that:
\begin{enumerate}
\item $(c,\{f_{i}\}_{i=0}^{k})$ is either a piecewise linear or threshold
affine VAR satisfying \ref{enu:pwa:homeo}{\upshape{{\scalefont{0.76}--3}}};
\item \label{enu:f0K}$\Phi_{0}^{(\ell)}=\Phi_{0}$ for all $\ell\in\{1,\ldots,L\}$,
for some nonsingular $\Phi_{0}$;
\item $K:\mathbb{R}\ensuremath{\rightarrow}\mathbb{R}$ is continuous and non-negative, with $\int_{\mathbb{R}^{p}}K(u)\ensuremath{\,\ensuremath{\mathrm{d}}} u=1$
and $\int_{\mathbb{R}^{p}}uK(u)\ensuremath{\,\ensuremath{\mathrm{d}}} u=0$;
\item $f_{i,K}$ is constructed by smoothing $f_{i}$ with $K$ as in
\ref{eq:smoothed}, for $i\in\{0,\ldots,k\}$.
\end{enumerate}
Then $(c,\{f_{i,K}\}_{i=0}^{k})$ belongs to class ${\cal M}_{r}(\alpha,\abv b,\abv{\rho})$,
for $\abv b\geq\max_{1\leq\ell\leq L}\smlnorm{\b{\beta}^{(\ell)}}$,
for $\b{\beta}^{(\ell)}$ as in \ref{eq:beta_l} above.
\end{prop}
The requirement in \ref{enu:f0K} that $f_{0}$ be linear is needed
principally to facilitate the verification of \ref{enu:cvar:jsr},
because of the role played by $f_{K,0}^{-1}$ there. We expect
that it should be possible to extend the preceding to allow $f_{0}$
to be a general piecewise affine function, albeit possibly at the
cost of additional regularity conditions. (Indeed, it may be shown
that the smooth counterpart $f_{0,K}$ of $f_{0}$ is a homeomorphism
if every $\Phi\in\operatorname{co}\{\Phi_{0}^{(\ell)}\}_{\ell=1}^{L}$ is invertible,
thereby satisfying \ref{enu:cvar:homeo}.)
\section{Conclusion}
\label{sec:conclusion}
This paper has extended the Granger--Johansen representation theorem
to a flexible class of additively time-separable, nonlinear SVAR models.
This shows that such models may be applied directly to time series
in which (common) stochastic trends are present, without the need
for pre-filtering, thus avoiding the potential for misspecification
that this entails. As an important corollary to our results, we show
that these models are capable of supporting the same kinds of long-run
identifying restrictions as are available in linear cointegrated SVARs.
A companion paper to the present work provides limit theory for $n^{-1/2}z_{\smlfloor{n\lambda}}$,
and a further discussion of nonlinear cointegration, under the assumptions
maintained here.
\bibliographystyle{ecta}
\bibliography{cksvar}