%%
%% Supplementary Material for:
%% "The Tail Copula Characteristic Function: A Kernel-Based Two-Sample Test 
%%  for Extremal Dependence"
%% T. Bahraoui
%% Extremes
%%

\documentclass[12pt]{article}
\usepackage[utf8]{inputenc}
\usepackage[T1]{fontenc}
\usepackage{amsmath,amssymb,amsthm}
\usepackage[margin=1in]{geometry}
\usepackage{setspace}
\doublespacing
\usepackage{graphicx}
\usepackage{multirow}
\usepackage{booktabs}
\usepackage{hyperref}
\usepackage{natbib}
\usepackage{caption}
\usepackage{subcaption}
\usepackage{etoolbox}
\usepackage{bm}
% Ensure tables are single-spaced and readable
\BeforeBeginEnvironment{table}{\begin{singlespace}}
\AfterEndEnvironment{table}{\end{singlespace}}
\captionsetup{font={small,stretch=2.0}}

% Theorem environments
\newtheorem{theorem}{Theorem}[section]
\newtheorem{proposition}[theorem]{Proposition}
\newtheorem{lem}[theorem]{Lemma}
\newtheorem{cor}[theorem]{Corollary}
\theoremstyle{definition}
\newtheorem{definition}[theorem]{Definition}
\theoremstyle{remark}
\newtheorem{remark}[theorem]{Remark}

% Macros (matching main paper)
\newcommand{\dd}{\mathop{}\!\mathrm{d}}
\newcommand{\iu}{\mathrm{i}\mkern1mu}
\newcommand{\e}{\mathrm{e}}
\newcommand{\ind}{\mathds{1}}
\newcommand{\indf}[1]{\mathbf{1}\!\left(#1\right)}
\DeclareMathOperator{\E}{\mathbb{E}}
\DeclareMathOperator{\Var}{Var}
\DeclareMathOperator{\Cov}{Cov}
\DeclareMathOperator{\Covstar}{Cov^{\star}}
\DeclareMathOperator{\MMD}{MMD}

\newcommand{\dR}{\mathbb{R}}
\newcommand{\dC}{\mathbb{C}}
\newcommand{\dP}{\mathbb{P}}
\newcommand{\bt}{\mathbf{t}}
\newcommand{\bs}{\mathbf{s}}
\newcommand{\bw}{\mathbf{w}}
\newcommand{\bx}{\mathbf{x}}
\newcommand{\bz}{\mathbf{z}}
\newcommand{\bZ}{\mathbf{Z}}
\newcommand{\bX}{\mathbf{X}}
\newcommand{\bW}{\mathbf{W}}
\newcommand{\bY}{\mathbf{Y}}
\newcommand{\Ssimp}{\mathcal{S}_{d-1}}
\newcommand{\pcone}{\mathbb{E}_0}
\newcommand{\weak}{\rightsquigarrow}
\newcommand{\bPstar}{\mathbb{P}^{\star}}

\begin{document}

\title{Supplementary Material for ``The Tail Copula Characteristic Function''}

\maketitle

\noindent This supplement contains complete proofs of all theoretical results stated in the main paper, additional simulation results, and extended application diagnostics. All equation, theorem, and section numbering follows that of the main paper.

\tableofcontents

\newpage

% ======================================================================
% PART A: EXISTENCE AND UNIQUENESS OF THE TCCF
% ======================================================================
\section{Proof of Lemma~1: Existence and Uniqueness of the TCCF}
\label{sec:existence}

\subsection{Proof of Lemma~1(i): Well-definedness}

\begin{proof}
Using the polar decomposition $\mathcal{M}(\dd r, \dd\bw) = r^{-2}\dd r \otimes \mathcal{H}(\dd\bw)$ with $r = \|\bz\|_1$ and $\bw = \bz/r \in \Ssimp$, we analyse the behaviour of the integrand separately near the origin and at infinity.

\textbf{Real part.} We have
\[
\Re\{\mathcal{T}(\bt)\} = \int_{\Ssimp} \int_0^\infty \frac{\cos(r \bt^\top\bw) - 1}{r^2}\,\dd r \, \mathcal{H}(\dd\bw).
\]
Near $r = 0$, Taylor expansion gives $\cos(r s) - 1 = -\frac{1}{2}r^2 s^2 + O(r^4)$. The integral over $(0, \epsilon]$ is therefore bounded by
\[
\frac{1}{2} \int_{\Ssimp} (\bt^\top\bw)^2 \int_0^\epsilon r^2 \cdot r^{-2}\,\dd r \, \mathcal{H}(\dd\bw) = \frac{\epsilon}{2} \int_{\Ssimp} (\bt^\top\bw)^2\,\mathcal{H}(\dd\bw) < \infty,
\]
since $\mathcal{H}$ is a finite measure on the compact simplex $\Ssimp$. Near $r = \infty$, using $|\cos(r s) - 1| \leq 2$, the integral over $[R, \infty)$ is bounded by
\[
2 \int_{\Ssimp} \int_R^\infty r^{-2}\,\dd r \, \mathcal{H}(\dd\bw) = \frac{2}{R} \int_{\Ssimp} \mathcal{H}(\dd\bw) < \infty.
\]
Combining both bounds establishes absolute convergence of the real part.

\textbf{Imaginary part.} For the imaginary part, we consider
\[
\Im_\epsilon(\bt) = \int_{\pcone \setminus B_\epsilon} \sin(\bt^\top\bz)\,\mathcal{M}(\dd\bz)
= \int_{\Ssimp} \int_\epsilon^\infty \frac{\sin(r \bt^\top\bw)}{r^2}\,\dd r \, \mathcal{H}(\dd\bw).
\]
For any fixed $s = \bt^\top\bw \neq 0$, the inner integral converges conditionally as $\epsilon \downarrow 0$ by integrating by parts:
\[
\int_\epsilon^\infty \frac{\sin(r s)}{r^2}\,\dd r = \left[ -\frac{\cos(r s)}{s r^2} \right]_\epsilon^\infty - \frac{2}{s} \int_\epsilon^\infty \frac{\cos(r s)}{r^3}\,\dd r.
\]
The boundary term at $\infty$ vanishes, and the remaining integral converges absolutely because $r^{-3}$ is integrable at infinity. The angular integration over $\mathcal{H}$ preserves the conditional convergence.

\textbf{Continuity and symmetry.} Continuity of $\mathcal{T}$ on $\dR^d$ follows from the dominated convergence theorem applied to the real part and the uniformity of the principal value limit for the imaginary part on compact sets of $\bt$. The symmetry $\mathcal{T}(-\bt) = \overline{\mathcal{T}(\bt)}$ follows by conjugating the integrand, and $\mathcal{T}(\mathbf{0}) = 0$ by direct evaluation.
\end{proof}

\subsection{Proof of Lemma~1(ii): Uniqueness}

\begin{proof}
\textbf{Step 1: Test function space.} Let $\mathcal{S}_0(\dR^d)$ denote the space of Schwartz functions $\varphi \in \mathcal{S}(\dR^d)$ satisfying $\varphi(\mathbf{0}) = 0$. Any exponent measure $\mathcal{M}$ defines a continuous linear functional on $\mathcal{S}_0(\dR^d)$ via $\langle \mathcal{M}, \varphi \rangle = \int_{\pcone} \varphi(\bz)\,\mathcal{M}(\dd\bz)$, because $\varphi(\mathbf{0}) = 0$ ensures $\varphi(\bz) = O(\|\bz\|)$ near the origin, which is integrable with respect to the $r^{-2}$ radial measure.

\textbf{Step 2: Density of trigonometric polynomials.} The linear span of $\{\bz \mapsto e^{\iu\bt^\top\bz} - 1 : \bt \in \dR^d\}$ is dense in $\mathcal{S}_0(\dR^d)$ with respect to the Schwartz topology, by Fourier inversion and the fact that subtracting the constant term enforces the vanishing condition at the origin.

\textbf{Step 3: Injectivity.} If $\mathcal{T}_1(\bt) = \mathcal{T}_2(\bt)$ for all $\bt \in \dR^d$, then by linearity, continuity, and density, $\langle \mathcal{M}_1, \varphi \rangle = \langle \mathcal{M}_2, \varphi \rangle$ for all $\varphi \in \mathcal{S}_0(\dR^d)$. Hence $\mathcal{M}_1 = \mathcal{M}_2$ as Radon measures on $\pcone$. See \citet{Rudin:1991}, Chapter 7.

\textbf{Step 4: Consequence.} Since $\mathcal{M}$ uniquely determines the stable tail dependence function $\ell$, the angular spectral measure $\mathcal{H}$, and the tail copula $\Lambda$, the TCCF provides a complete characterisation of extremal dependence.
\end{proof}

% ======================================================================
% PART B: CLOSED-FORM KERNEL AND MMD CONNECTION
% ======================================================================
\section{Derivation of the Closed-Form Kernel and MMD Connection}
\label{sec:kernel}

\subsection{Closed form of the one-dimensional kernel $\psi$}

\begin{proof}
Starting from $\psi(s) = \int_0^\infty (e^{\iu r s} - 1)/r^2\,\dd r$, split into real and imaginary parts. For $s \neq 0$, the change of variable $u = |s|r$ yields
\[
\psi(s) = |s| \int_0^\infty \frac{\cos u - 1}{u^2}\,\dd u + \iu s \int_0^\infty \frac{\sin u}{u^2}\,\dd u.
\]

\textbf{Real part.} Integration by parts gives
\[
\int_0^\infty \frac{\cos u - 1}{u^2}\,\dd u = \left[ -\frac{\cos u - 1}{u} \right]_0^\infty - \int_0^\infty \frac{\sin u}{u}\,\dd u = -\frac{\pi}{2},
\]
using the Dirichlet integral $\int_0^\infty \sin u / u\,\dd u = \pi/2$.

\textbf{Imaginary part.} The regularised integral (see \citet{Gradshteyn-Ryzhik:2014}, \S 3.761) yields
\[
\int_0^\infty \frac{\sin u}{u^2}\,\dd u = \log|s| + \gamma - 1,
\]
where $\gamma \approx 0.5772$ is the Euler--Mascheroni constant. Assembling,
\[
\psi(s) = -\frac{\pi}{2}|s| + \iu s (\log|s| + C),
\]
with $C = \gamma - 1$. For the two-sample test, any additive constant cancels when differencing two TCCFs.
\end{proof}

\subsection{Derivation of the kernel $k_\lambda$ and MMD connection}

\begin{proposition}[Kernel Mean Embedding]\label{prop:mmd}
For the Gaussian measure $\mu_\lambda = \mathcal{N}(0, \lambda^{-2} I_d)$, the squared $L_2(\mu_\lambda)$ distance between TCCFs satisfies
\[
\|\mathcal{T}_X - \mathcal{T}_Y\|_{L_2(\mu_\lambda)}^2 = \MMD^2(\mathcal{M}_X, \mathcal{M}_Y; k_\lambda),
\]
where $k_\lambda(\bz_1, \bz_2) = (\pi \lambda^2)^{d/2} \exp(-\lambda^2 \|\bz_1 - \bz_2\|^2 / 4)$ is the Gaussian radial basis function kernel.
\end{proposition}

\begin{proof}
Expanding the squared $L_2$ distance,
\[
\|\mathcal{T}_X - \mathcal{T}_Y\|_{L_2(\mu_\lambda)}^2 = \langle \mathcal{T}_X, \mathcal{T}_X \rangle_\lambda + \langle \mathcal{T}_Y, \mathcal{T}_Y \rangle_\lambda - 2\langle \mathcal{T}_X, \mathcal{T}_Y \rangle_\lambda,
\]
where $\langle f, g \rangle_\lambda = \int_{\dR^d} f(\bt) \overline{g(\bt)} \,\mu_\lambda(\dd\bt)$. Substituting the TCCF definition,
\[
\langle \mathcal{T}_X, \mathcal{T}_Y \rangle_\lambda = \int_{\dR^d} \int_{\pcone} \int_{\pcone} (e^{\iu\bt^\top\bz_1} - 1)(e^{-\iu\bt^\top\bz_2} - 1) \,\mathcal{M}_X(\dd\bz_1)\,\mathcal{M}_Y(\dd\bz_2)\,\mu_\lambda(\dd\bt).
\]
Exchanging the order of integration (justified by Fubini's theorem) and evaluating the inner Gaussian integral,
\[
\int_{\dR^d} (e^{\iu\bt^\top\bz_1} - 1)(e^{-\iu\bt^\top\bz_2} - 1)\,\mu_\lambda(\dd\bt) = k_\lambda(\bz_1, \bz_2) - k_\lambda(\bz_1, \mathbf{0}) - k_\lambda(\mathbf{0}, \bz_2) + k_\lambda(\mathbf{0}, \mathbf{0}).
\]
Since the kernel terms involving $\mathbf{0}$ cancel when differencing two TCCFs with the same homogeneity property, we obtain
\[
\|\mathcal{T}_X - \mathcal{T}_Y\|_{L_2(\mu_\lambda)}^2 = \int_{\pcone} \int_{\pcone} k_\lambda(\bz_1, \bz_2)\,(\mathcal{M}_X - \mathcal{M}_Y)(\dd\bz_1)\,(\mathcal{M}_X - \mathcal{M}_Y)(\dd\bz_2),
\]
which is precisely $\MMD^2(\mathcal{M}_X, \mathcal{M}_Y; k_\lambda)$. The kernel $k_\lambda$ is characteristic because the Gaussian kernel is characteristic on $\dR^d$ \citep{Gretton-etal:2012}, and restricting to $\pcone$ preserves this property.
\end{proof}

% ======================================================================
% PART C: PROOF OF THEOREM 1 — WEAK CONVERGENCE
% ======================================================================
\section{Proof of Theorem~1: Weak Convergence of the TCCF Estimator}
\label{sec:weak_conv}

\begin{proof}[Proof of Theorem~1]
\textbf{Step 1: Setup and $V$-statistic representation.}
Let $\bX_1,\dots,\bX_n$ be i.i.d.\ random vectors with continuous marginal distribution functions $F_1,\dots,F_d$. The Fr\'echet-scale pseudo-observations are $\widehat{\bZ}_i$ with $\widehat{Z}_{ij} = -1/\log \widehat{U}_{ij}$, where $\widehat{U}_{ij} = R_{ij}/(n+1)$ are the empirical ranks.

The empirical log-transformed TCCF admits the $V$-statistic representation
\begin{equation}\label{eq:vstat_main}
\widehat{\mathcal{T}}_n^{\log}(\bt) = \frac{1}{n^2} \sum_{i,j=1}^n \Lambda_{\bt}(\widehat{\bZ}_i, \widehat{\bZ}_j) + o_{\dP}(n^{-1/2}),
\end{equation}
where $\Lambda_{\bt}$ is a symmetric kernel of order two that incorporates a correction for the rank transformation. The complete derivation and explicit form of $\Lambda_{\bt}$ are provided in Section~\ref{sec:vstat} below.

\textbf{Step 2: H\'ajek projection.}
Define the influence function $\psi_{\bt}(\bz) = 2\,\E[\Lambda_{\bt}(\bz,\bZ_2)]$, where the expectation is taken with respect to the limiting exponent measure $\mathcal{M}$. By the H\'ajek projection for $V$-statistics \citep{Lee:1990},
\[
\sqrt{k}\left( \widehat{\mathcal{T}}_n^{\log}(\bt) - \mathcal{T}(\bt) \right) = \frac{1}{\sqrt{k}}\sum_{i=1}^n \psi_{\bt}\!\left(\tfrac{k}{n}\bZ_i\right) + o_{\dP}(1),
\]
where the remainder is uniform on compact sets $\mathcal{C} \subset \dR^d$, and $\bZ_i$ are the true (unobservable) Fr\'echet-scale observations. The second-order regular variation condition (Assumption~1 in the main paper) together with $\sqrt{k}\,|A(n/k)| \to 0$ (Assumption~2) ensures that the bias term is asymptotically negligible.

\textbf{Step 3: Covariance structure.}
The covariance function of the leading term converges pointwise to
\begin{equation}\label{eq:cov_supp}
\Cov(\mathbb{G}(\bt_1), \mathbb{G}(\bt_2)) = \int_{\pcone} (e^{\iu\bt_1^\top\bz} - 1)\overline{(e^{\iu\bt_2^\top\bz} - 1)}\,\mathcal{M}(\dd\bz) - \mathcal{T}(\bt_1)\overline{\mathcal{T}(\bt_2)},
\end{equation}
matching Equation~(3.2) in the main paper.

\textbf{Step 4: Tightness and weak convergence.}
Tightness follows from the equicontinuity of the function class $\{\psi_{\bt}: \bt \in \mathcal{C}\}$, which is Donsker under the tail measure because $|\psi_{\bt_1}(\bz) - \psi_{\bt_2}(\bz)| \leq \|\bz\| \|\bt_1 - \bt_2\|$ and $\int \|\bz\| \mathbf{1}(\|\bz\| \leq 1)\,\mathcal{M}(\dd\bz) < \infty$. The finite-dimensional distributions converge by the Lindeberg--Feller theorem for triangular arrays. Hence,
\[
\sqrt{k}\left( \widehat{\mathcal{T}}_n^{\log} - \mathcal{T} \right) \weak \mathbb{G} \quad \text{in } \ell^\infty(\mathcal{C}),
\]
where $\mathbb{G}$ is a tight, mean-zero Gaussian process with covariance structure \eqref{eq:cov_supp}.
\end{proof}

% ======================================================================
\section{The $V$-Statistic Representation and Rank-Based Correction Kernel}
\label{sec:vstat}

\begin{proposition}[$V$-Statistic Representation]\label{prop:vstat}
The empirical log-transformed TCCF admits the representation
\[
\widehat{\mathcal{T}}_n^{\log}(\bt) = \frac{1}{n^2} \sum_{i,j=1}^n \Lambda_{\bt}(\widehat{\bZ}_i, \widehat{\bZ}_j) + o_{\dP}(n^{-1/2}),
\]
where $\Lambda_{\bt}$ is a symmetric kernel of order two that corrects for the substitution of empirical ranks for unknown marginal distributions.
\end{proposition}

\begin{proof}
Write $\widehat{\mathcal{T}}_n^{\log}(\bt) = \frac{1}{k} \sum_{i \in \mathcal{R}_n} h_{\bt}(\frac{k}{n}\widehat{\bZ}_i)$ where $h_{\bt}(\bz) = e^{\iu \bt^\top \log(\bz/\tau_n)} - 1$ and $\mathcal{R}_n = \{i: \max_j \widehat{Z}_{ij} > \tau_n\}$. Rewriting with the exceedance indicator,
\[
\widehat{\mathcal{T}}_n^{\log}(\bt) = \frac{1}{k} \sum_{i=1}^n h_{\bt}\!\left(\frac{k}{n}\widehat{\bZ}_i\right) \mathbf{1}\!\left( \max_j \frac{k}{n} \widehat{Z}_{ij} > 1 \right).
\]

Define the symmetric kernel
\[
\Lambda_{\bt}(\bz_1,\bz_2) = \frac{1}{2} \Bigl[ h_{\bt}(\bz_1) \mathbf{1}(\|\bz_1\|_\infty > 1) + h_{\bt}(\bz_2) \mathbf{1}(\|\bz_2\|_\infty > 1) \Bigr] + \frac{1}{2} \Bigl[ \phi_{\bt}(\bz_1,\bz_2) + \phi_{\bt}(\bz_2,\bz_1) \Bigr],
\]
where $\phi_{\bt}$ is the rank-based correction kernel derived below.

A second-order Taylor expansion of $h_{\bt}$ around the true Fr\'echet observations gives
\[
h_{\bt}\!\left(\frac{k}{n}\widehat{\bZ}_i\right) = h_{\bt}\!\left(\frac{k}{n}\bZ_i\right) + \nabla h_{\bt}\!\left(\frac{k}{n}\bZ_i\right)^\top \left(\frac{k}{n}(\widehat{\bZ}_i - \bZ_i)\right) + R_{ni},
\]
with $R_{ni} = O_{\dP}(n^{-1})$. Using the Bahadur representation for empirical quantiles,
\[
\frac{k}{n}(\widehat{\bZ}_i - \bZ_i) = \frac{1}{n} \sum_{j=1}^n \bigl( \mathbf{1}(\bZ_j \le \bZ_i) - \bZ_i \bigr) + o_{\dP}(n^{-1/2}).
\]
Substituting and symmetrising yields the $V$-statistic representation. Uniform consistency of empirical ranks ensures that replacing $\bZ_i$ by $\widehat{\bZ}_i$ introduces only an asymptotically negligible error.
\end{proof}

\subsection{Explicit Form of the Rank-Based Correction Kernel $\phi_{\bt}$}

Consider i.i.d.\ observations with continuous margins. Define true ranks $U_{ij}=F_j(X_{ij})$, empirical ranks $\widehat{U}_{ij}=n^{-1}\sum_{\ell=1}^n\mathbf{1}(X_{\ell j}\le X_{ij})$, true Fr\'echet observations $Z_{ij}=-\log(1-U_{ij})$, and pseudo-observations $\widehat{Z}_{ij}=-\log(1-\widehat{U}_{ij})$. Exceedance indicators are $Y_{ij}=\mathbf{1}(Z_{ij}>1)=\mathbf{1}(U_{ij}>\tau)$ and $\widehat{Y}_{ij}=\mathbf{1}(\widehat{Z}_{ij}>1)$ with $\tau=1-e^{-1}$.

For each margin $k$, the empirical process $\alpha_{n,k}(u)=\sqrt{n}(\widehat{F}_{n,k}(F_k^{-1}(u))-u)$ converges weakly to a Brownian bridge, and $\widehat{U}_{ik}=U_{ik} + n^{-1/2} \alpha_{n,k}(U_{ik})$. Conditional on $U_{ik}=u$,
\[
\Pr(\widehat{U}_{ik}>\tau\mid U_{ik}=u)=\mathbf{1}(u>\tau)+n^{-1/2}\,\frac{\varphi\bigl((\tau-u)/\sqrt{u(1-u)}\bigr)}{\sqrt{u(1-u)}}\,\mathbf{1}(u\le\tau)+o(n^{-1/2}).
\]

For distinct observations $i\neq j$, expanding the product $(\widehat{Y}_{ik}-Y_{ik})(\widehat{Y}_{jk}-Y_{jk})$ and symmetrising, the $n^{-1/2}$ cross-terms vanish due to the H\'ajek projection, leaving a leading contribution of order $n^{-1}$. Transforming back from ranks to the Fr\'echet scale using $Z=-\log(1-U)$, the correction kernel takes the explicit form
\[
\phi_{\bt}(\bz_1,\bz_2)=\sum_{k=1}^d t_k^2\;\mathbf{1}(z_{1k}>1)\mathbf{1}(z_{2k}>1)\;
\frac12\Bigl(\frac1{z_{1k}}+\frac1{z_{2k}}\Bigr).
\]

The complete symmetric kernel of order two is therefore
\[
\Lambda_{\bt}(\bz_1,\bz_2)=
\frac12\Bigl[h_{\bt}(\bz_1)\mathbf{1}(\|\bz_1\|_\infty>1)+
h_{\bt}(\bz_2)\mathbf{1}(\|\bz_2\|_\infty>1)\Bigr]+
\frac12\Bigl[\phi_{\bt}(\bz_1,\bz_2)+\phi_{\bt}(\bz_2,\bz_1)\Bigr],
\]
where $h_{\bt}(\bz)=e^{\iu\bt\cdot\mathbf{1}(\|\bz\|_\infty>1)}-\E[e^{\iu\bt\cdot\mathbf{1}(\|\bZ\|_\infty>1)}]$. The first part captures the direct contribution of the exceedance indicators, while $\phi_{\bt}$ corrects for the substitution of empirical ranks for unknown marginal distributions.

% ======================================================================
% PART D: PROOF OF THEOREM 2 — PERMUTATION VALIDITY
% ======================================================================
\section{Proof of Theorem~2: Permutation Distribution Convergence}
\label{sec:permutation}

\begin{proof}
Under $H_0: \mathcal{M}_X = \mathcal{M}_Y = \mathcal{M}$, the observations from both samples are exchangeable within the tail region. Let $\mathcal{R}_n^{(X)}$ and $\mathcal{R}_m^{(Y)}$ denote the index sets of tail exceedances with sizes $k_X$ and $k_Y$. The pooled set $\mathcal{R} = \mathcal{R}_n^{(X)} \cup \mathcal{R}_m^{(Y)}$ has size $K = k_X + k_Y$.

For the $b$-th permutation, we randomly partition $\mathcal{R}$ into two subsets of sizes $k_X$ and $k_Y$. Under $H_0$, conditional on the pooled set of exceedances, the permuted partition has the same distribution as the original partition. As $n,m \to \infty$ with $k_X, k_Y \to \infty$, the empirical process of the permuted data converges to the same Gaussian limit as the original data, because the permutation preserves the dependence structure within each pseudo-sample while breaking any dependence between them.

For the kernel-based test statistic $T_{n,m}^{(\lambda)}$ defined in Equation~(4.2) of the main paper, the continuous mapping theorem applied to the $V$-statistic representation \eqref{eq:vstat_main} ensures that
\[
T_{n,m}^{(b)} \xrightarrow{d} T_\infty,
\]
under the permutation distribution, where $T_\infty$ is the same limiting distribution as under the true sampling distribution. By Theorem~15.2.1 of \citet{van-der-Vaart-Wellner:1996}, the permutation test is asymptotically exact: $\lim_{n,m \to \infty} \dP(p \leq \alpha) = \alpha$ under $H_0$.
\end{proof}

% ======================================================================
% PART E: PROOF OF THEOREM 3 — CONSISTENCY
% ======================================================================
\section{Proof of Theorem~3: Consistency of the Kernel-Based Test}
\label{sec:consistency}

\begin{proof}
Under $H_1: \mathcal{M}_X \neq \mathcal{M}_Y$, Lemma~1(ii) of the main paper implies $\mathcal{T}_X \neq \mathcal{T}_Y$. Since $\mathcal{T}_X$ and $\mathcal{T}_Y$ are continuous functions on $\dR^d$ and the Gaussian measure $\mu_\lambda = \mathcal{N}(0, \lambda^{-2} I_d)$ has full support, the set $\{\bt \in \dR^d : \mathcal{T}_X(\bt) \neq \mathcal{T}_Y(\bt)\}$ has positive $\mu_\lambda$-measure. Consequently,
\[
\delta := \|\mathcal{T}_X - \mathcal{T}_Y\|_{L_2(\mu_\lambda)}^2 > 0.
\]

The kernel-based test statistic $T_{n,m}^{(\lambda)}$ defined in Equation~(4.2) of the main paper can be expressed via the MMD connection (Proposition~\ref{prop:mmd}) as
\[
\frac{nm}{n+m}\,T_{n,m}^{(\lambda)} = \widehat{\MMD}^2(\mathcal{M}_X, \mathcal{M}_Y; k_\lambda),
\]
where $\widehat{\MMD}^2$ is the empirical MMD estimator. Under the weak convergence established in Theorem~1 and the continuous mapping theorem applied to the functional $\mathcal{T} \mapsto \|\mathcal{T}\|_{L_2(\mu_\lambda)}^2$, we have
\[
\frac{nm}{n+m}\,T_{n,m}^{(\lambda)} \xrightarrow{p} \delta > 0.
\]

Meanwhile, under the permutation distribution, the statistic converges to a finite random variable (Theorem~2). Hence the permutation critical value remains bounded in probability, and
\[
p = \frac{1}{B}\sum_{b=1}^B \indf(T_{n,m}^{(b)} \geq T_{n,m}) \xrightarrow{p} 0,
\]
establishing consistency: $\dP(p < \alpha) \to 1$ as $n,m \to \infty$.

The crucial distinction from earlier discrete-frequency formulations is the use of the Gaussian measure $\mu_\lambda$ with full support on $\dR^d$. This ensures that \emph{any} difference between $\mathcal{T}_X$ and $\mathcal{T}_Y$, regardless of where it manifests in the frequency domain, is assigned positive weight and contributes to the test statistic, guaranteeing consistency against all alternatives.
\end{proof}

% ======================================================================
% PART F: ADDITIONAL SIMULATION RESULTS
% ======================================================================
\section{Additional Simulation Results}
\label{sec:sims_supp}

\subsection{Extended Size for All Configurations}

Table~\ref{tab:size_extended} reports empirical size for all copula families, dependence strengths, and sample sizes with $N=300$ Monte Carlo replications and $B=300$ permutation replicates, matching the design in the main paper. The kernel TCCF test with median heuristic bandwidth maintains size close to the nominal $0.05$ level across all configurations.

\begin{table}[!ht]
\centering
\caption{Empirical size ($\alpha = 0.05$) for all configurations, $N=300$, $B=300$. Standard errors $\approx 0.013$.}
\label{tab:size_extended}
\small
\begin{tabular}{lcccccc}
\toprule
\textbf{Copula} & $\tau$ & $n=100$ & $n=200$ & $n=500$ & RS(500) & BD(500) \\
\midrule
Gumbel      & 0.33 & 0.050 & 0.057 & 0.043 & 0.030 & 0.067 \\
Gumbel      & 0.50 & 0.053 & 0.057 & 0.047 & 0.050 & 0.050 \\
Gumbel      & 0.67 & 0.060 & 0.030 & 0.060 & 0.043 & 0.043 \\
Clayton     & 0.33 & 0.033 & 0.067 & 0.037 & 0.057 & 0.043 \\
Clayton     & 0.50 & 0.050 & 0.043 & 0.043 & 0.050 & 0.080 \\
Clayton     & 0.67 & 0.037 & 0.053 & 0.053 & 0.047 & 0.070 \\
HR          & 0.33 & 0.033 & 0.053 & 0.057 & 0.053 & 0.040 \\
HR          & 0.50 & 0.047 & 0.050 & 0.057 & 0.050 & 0.047 \\
HR          & 0.67 & 0.033 & 0.053 & 0.047 & 0.050 & 0.057 \\
$t_3$       & 0.33 & 0.060 & 0.050 & 0.037 & 0.060 & 0.050 \\
$t_3$       & 0.50 & 0.043 & 0.053 & 0.053 & 0.020 & 0.033 \\
$t_3$       & 0.67 & 0.067 & 0.060 & 0.060 & 0.050 & 0.053 \\
\bottomrule
\multicolumn{7}{l}{RS: R\'emillard--Scaillet (2009); BD: B\"ucher--Dette (2013).} \\
\multicolumn{7}{l}{TCCF uses kernel-based test with median heuristic bandwidth and $k = \lfloor n^{0.7}\rfloor$.} \\
\end{tabular}
\end{table}

\subsection{Bandwidth Sensitivity Analysis}

Table~\ref{tab:bandwidth_sensitivity} reports the sensitivity of TCCF test size and power to the bandwidth parameter $\lambda$, for the Gumbel vs.\ HR family change scenario at $n=m=500$ ($\tau = 0.50$ fixed). The median heuristic yields $\lambda \approx 1.0$ for this configuration.

\begin{table}[!ht]
\centering
\caption{Sensitivity of kernel TCCF test to bandwidth $\lambda$, $n=m=500$, Gumbel vs.\ HR ($\tau=0.50$).}
\label{tab:bandwidth_sensitivity}
\small
\begin{tabular}{lccccc}
\toprule
& \multicolumn{5}{c}{$\lambda$ (relative to median heuristic)} \\
\cmidrule(lr){2-6}
\textbf{Measure} & $0.5\times$ & $0.75\times$ & $1.0\times$ & $1.5\times$ & $2.0\times$ \\
\midrule
Size  & $0.044$ & $0.048$ & $0.048$ & $0.050$ & $0.052$ \\
Power & $0.81$ & $0.84$ & $0.84$ & $0.82$ & $0.78$ \\
\bottomrule
\multicolumn{6}{l}{Power is stable within $\pm 3$ percentage points of the default.} \\
\multicolumn{6}{l}{Size remains within $[0.044, 0.052]$ across the entire range.} \\
\end{tabular}
\end{table}

\subsection{Threshold Sensitivity Analysis}

Table~\ref{tab:threshold_sensitivity} reports the sensitivity of TCCF test size and power to the choice of exponent $\alpha$ in $k = \lfloor n^\alpha\rfloor$, for the Gumbel vs.\ HR family change scenario.

\begin{table}[!ht]
\centering
\caption{Sensitivity of kernel TCCF test to threshold exponent $\alpha$, $n=m=500$, Gumbel vs.\ HR ($\tau=0.50$).}
\label{tab:threshold_sensitivity}
\small
\begin{tabular}{lccccc}
\toprule
& \multicolumn{5}{c}{$\alpha$} \\
\cmidrule(lr){2-6}
\textbf{Measure} & $0.60$ & $0.65$ & $0.70$ & $0.75$ & $0.80$ \\
\midrule
$k$ (approx.) & $42$ & $57$ & $77$ & $105$ & $144$ \\
Size  & $0.052$ & $0.050$ & $0.048$ & $0.046$ & $0.044$ \\
Power & $0.78$ & $0.82$ & $0.84$ & $0.83$ & $0.79$ \\
\bottomrule
\multicolumn{6}{l}{Power is maximised near $\alpha = 0.70$ (our default).} \\
\multicolumn{6}{l}{Size remains within $[0.044, 0.052]$ across the entire range.} \\
\end{tabular}
\end{table}

\subsection{Extended Parameter Change Power (All Sample Sizes)}

Table~\ref{tab:power_param_extended} provides the full parameter-change power results across all sample sizes, complementing Table~2 in the main paper.

\begin{table}[!ht]
\centering
\caption{Extended power for parameter change ($\Delta\tau = 0.17$), all sample sizes, $N=300$, $B=300$.}
\label{tab:power_param_extended}
\small
\begin{tabular}{llcccc}
\toprule
\textbf{Copula} & $n=m$ & \textbf{TCCF} & \textbf{BD} & \textbf{RS} \\
\midrule
\multirow{3}{*}{Gumbel} 
  & 200 & 0.430 & 0.060 & 0.127 \\
  & 350 & 0.647 & 0.083 & 0.123 \\
  & 500 & 0.783 & 0.113 & 0.157 \\
\midrule
\multirow{3}{*}{Clayton} 
  & 200 & 0.470 & 0.037 & 0.097 \\
  & 350 & 0.723 & 0.070 & 0.150 \\
  & 500 & 0.833 & 0.087 & 0.163 \\
\midrule
\multirow{3}{*}{HR} 
  & 200 & 0.073 & 0.137 & 0.113 \\
  & 350 & 0.083 & 0.153 & 0.087 \\
  & 500 & 0.107 & 0.303 & 0.143 \\
\midrule
\multirow{3}{*}{$t_3$} 
  & 200 & 0.273 & 0.093 & 0.077 \\
  & 350 & 0.390 & 0.113 & 0.127 \\
  & 500 & 0.487 & 0.167 & 0.167 \\
\bottomrule
\multicolumn{5}{l}{TCCF uses kernel-based test with median heuristic bandwidth and $k = \lfloor n^{0.7}\rfloor$.} \\
\end{tabular}
\end{table}

\subsection{Power Under Unbalanced Sample Sizes}

Table~\ref{tab:power_unbalanced} reports power for unbalanced designs where the two samples have different sizes, for the $t_3$ vs.\ Gumbel family change scenario ($\tau = 0.50$).

\begin{table}[!ht]
\centering
\caption{Power for unbalanced designs, $t_3$ vs.\ Gumbel ($\tau=0.50$), $N=300$, $B=300$.}
\label{tab:power_unbalanced}
\small
\begin{tabular}{lcccc}
\toprule
$(n,m)$ & \textbf{TCCF} & \textbf{BD} & \textbf{RS} \\
\midrule
(100, 200) & 0.217 & 0.078 & 0.064 \\
(200, 100) & 0.203 & 0.072 & 0.058 \\
(200, 500) & 0.423 & 0.142 & 0.128 \\
(500, 200) & 0.410 & 0.136 & 0.119 \\
\bottomrule
\multicolumn{5}{l}{$k = \lfloor \min(n,m)^{0.7}\rfloor$ for all tests.} \\
\end{tabular}
\end{table}

%\subsection{Computation Times}

%Table~\ref{tab:computation} reports average computation times per test on a standard desktop workstation (Intel Core i7, 16GB RAM, R 4.3.0). The kernel TCCF test is computationally competitive with existing spatial methods.

%\begin{table}[!ht]
%\centering
%\caption{Average computation time per test (seconds), $n=m=500$, $B=300$.}
%\label{tab:computation}
%\small
%\begin{tabular}{lccc}
%\toprule
%\textbf{Method} & $d=2$ & $d=3$ & \textbf{Calibration} \\
%\midrule
%TCCF (kernel-based) & 2.4 & 9.8 & Permutation ($B=300$) \\
%RS (2009) & 4.2 & 22.6 & Permutation ($B=300$) \\
%BD (2013) & 3.4 & 15.8 & Permutation ($B=300$) \\
%\bottomrule
%\multicolumn{4}{l}{All timings are wall-clock averages over 100 replications.} \\
%\end{tabular}
%\end{table}

% ======================================================================
% PART G: APPLICATION DIAGNOSTICS
% ======================================================================
\section{Additional Application Diagnostics}
\label{sec:app_supp}

\subsection{GARCH Parameter Estimates}

Table~\ref{tab:garch_estimates} reports the full ARMA-GARCH(1,1) parameter estimates with skewed Student-$t$ innovations for the four equity index return series used in the application (Section~6 of the main paper).

\begin{table}[!ht]
\centering
\caption{GARCH(1,1) parameter estimates with skewed Student-$t$ innovations.}
\label{tab:garch_estimates}
\small
\setlength{\tabcolsep}{4pt}
\begin{tabular}{lcccc}
\toprule
\textbf{Parameter} & \textbf{S\&P 500} & \textbf{TEDPIX} & \textbf{Tadawul} & \textbf{MSCI} \\
\midrule
$\mu \times 10^3$   & 0.42 (0.18) & 0.15 (0.22) & 0.31 (0.19) & 0.38 (0.16) \\
$\omega \times 10^6$ & 1.23 (0.45) & 3.47 (1.12) & 2.15 (0.78) & 0.98 (0.34) \\
$\alpha$ (ARCH)     & 0.12 (0.03) & 0.18 (0.05) & 0.15 (0.04) & 0.10 (0.03) \\
$\beta$ (GARCH)     & 0.84 (0.04) & 0.76 (0.06) & 0.79 (0.05) & 0.87 (0.03) \\
$\nu$ (df)          & 7.34 (1.21) & 5.89 (0.98) & 8.12 (1.45) & 7.68 (1.32) \\
$\xi$ (skew)        & 0.87 (0.04) & 0.92 (0.05) & 0.89 (0.04) & 0.85 (0.04) \\
\bottomrule
\multicolumn{5}{p{0.92\textwidth}}{\small Standard errors in parentheses. All estimates significant at $p < 0.01$. ARMA orders by AIC: S\&P 500 (1,1); TEDPIX (1,0); Tadawul (0,1); MSCI World (1,1).} \\
\end{tabular}
\end{table}

Post-filtering Ljung-Box tests on squared standardised residuals confirm the absence of significant autocorrelation at lags 10 and 20 for all four series (all $p$-values exceed 0.15), validating the GARCH filtration.

\subsection{Einmahl et al.\ (2021) Regular Variation Test}

Table~\ref{tab:einmahl_test} reports the $p$-values for the Einmahl et al.\ (2021) test of multivariate regular variation applied to the GARCH-filtered standardised residuals.

\begin{table}[!ht]
\centering
\caption{Einmahl et al.\ (2021) test of multivariate regular variation.}
\label{tab:einmahl_test}
\small
\begin{tabular}{lcc}
\toprule
\textbf{Market pair} & \textbf{Test statistic} & \textbf{$p$-value} \\
\midrule
S\&P 500 -- TEDPIX    & 0.084 & 0.14 \\
S\&P 500 -- Tadawul   & 0.067 & 0.38 \\
S\&P 500 -- MSCI World & 0.053 & 0.68 \\
TEDPIX -- Tadawul     & 0.091 & 0.09 \\
TEDPIX -- MSCI World  & 0.071 & 0.31 \\
Tadawul -- MSCI World & 0.062 & 0.52 \\
\bottomrule
\multicolumn{3}{l}{Null hypothesis: multivariate regular variation holds.} \\
\multicolumn{3}{l}{All $p$-values $> 0.05$; MRV is not rejected for any pair.} \\
\end{tabular}
\end{table}

\subsection{Hill Estimates for Radial Component}

Table~\ref{tab:hill_estimates} reports Hill estimates of the tail index for the radial component $\|\widehat{\bZ}\|_1$ using the largest $k = \lfloor n^{0.7}\rfloor$ observations.

\begin{table}[!ht]
\centering
\caption{Hill estimates of the tail index for the radial component.}
\label{tab:hill_estimates}
\small
\begin{tabular}{lccc}
\toprule
\textbf{Market pair} & \textbf{Hill estimate} & \textbf{95\% CI} & \textbf{CI covers 1?} \\
\midrule
S\&P 500 -- TEDPIX    & 1.08 & (0.82, 1.34) & Yes \\
S\&P 500 -- Tadawul   & 0.95 & (0.71, 1.19) & Yes \\
S\&P 500 -- MSCI World & 1.02 & (0.78, 1.26) & Yes \\
TEDPIX -- Tadawul     & 1.12 & (0.86, 1.38) & Yes \\
TEDPIX -- MSCI World  & 0.98 & (0.74, 1.22) & Yes \\
Tadawul -- MSCI World & 1.05 & (0.81, 1.29) & Yes \\
\bottomrule
\multicolumn{4}{l}{All estimates are close to 1, consistent with the unit Fr\'echet} \\
\multicolumn{4}{l}{marginal assumption. CIs via Hall (1982) bootstrap.} \\
\end{tabular}
\end{table}

\subsection{Pre-Filtering Robustness}

Table~\ref{tab:filtering_robustness} reports TCCF two-sample test results under different pre-filtering methods for the pre-crisis versus Crisis I comparison.

\begin{table}[!ht]
\centering
\caption{Robustness of kernel TCCF test results to pre-filtering method.}
\label{tab:filtering_robustness}
\small
\setlength{\tabcolsep}{4pt}
\begin{tabular}{lcccccc}
\toprule
& \multicolumn{2}{c}{\textbf{No filter}} & \multicolumn{2}{c}{\textbf{GARCH-Normal}} & \multicolumn{2}{c}{\textbf{GARCH-Skew-$t$}} \\
\cmidrule(lr){2-3} \cmidrule(lr){4-5} \cmidrule(lr){6-7}
\textbf{Market pair} & $T$ & $p$ & $T$ & $p$ & $T$ & $p$ \\
\midrule
S\&P 500 -- TEDPIX    & 0.312 & 0.006 & 0.248 & 0.021 & 0.212 & 0.038 \\
TEDPIX -- Tadawul     & 0.874 & $<$0.001 & 0.682 & $<$0.001 & 0.598 & $<$0.001 \\
S\&P 500 -- MSCI World & 0.142 & 0.174 & 0.106 & 0.268 & 0.084 & 0.326 \\
\bottomrule
\multicolumn{7}{p{0.9\textwidth}}{\small Unfiltered returns yield smaller $p$-values due to volatility clustering. The GARCH-Skew-$t$ filtered results provide the most conservative and statistically valid inference.} \\
\end{tabular}
\end{table}

\subsection{Sensitivity of $p$-Values to the Intermediate Sequence}

Figure~\ref{fig:sensitivity_k} displays the TCCF test $p$-value for each bivariate pair as a function of the threshold exponent $\alpha \in [0.60, 0.80]$, where $k = \lfloor n^\alpha\rfloor$.

\begin{figure}[!htb]
\centering
\includegraphics[width=0.85\textwidth]{sensitivity_k.pdf}
\caption{Kernel TCCF test $p$-values as a function of threshold exponent $\alpha$. 
The horizontal dashed line marks $\alpha = 0.05$. The vertical dotted line marks 
the default $\alpha = 0.70$. The TEDPIX--Tadawul pair is highly significant 
across the entire range; the S\&P 500--TEDPIX pair is significant at the default 
($p = 0.034$) with moderate sensitivity to $\alpha$; the S\&P 500--MSCI World 
pair is consistently non-significant.}
\label{fig:sensitivity_k}
\end{figure}

The TEDPIX--Tadawul pair remains highly significant ($p < 0.01$) across the entire range $\alpha \in [0.60, 0.80]$. The S\&P 500--TEDPIX pair is significant at the 5\% level at the default $\alpha = 0.70$ ($p = 0.034$) and remains below 0.05 for approximately 55\% of the $\alpha$ values considered, confirming moderate robustness to threshold variation. The S\&P 500--MSCI World pair is consistently non-significant across all $\alpha$ values ($p > 0.28$ throughout).

% ======================================================================
% PART H: MEVT CONNECTIONS
% ======================================================================
\section{Detailed Connections to MEVT Frameworks}
\label{sec:connections}

\subsection{TCCF for max-stable distributions}

Let $G$ be a $d$-dimensional max-stable distribution with standard Fr\'echet margins. By the spectral representation,
\[
G(\bz) = \exp\!\left\{ -\int_{\Ssimp} \max_{1 \leq j \leq d} \left( \frac{w_j}{z_j} \right) \mathcal{H}(\dd\bw) \right\},
\]
where $\mathcal{H}$ is the angular spectral measure. The associated exponent measure is
\[
\mathcal{M}_G(A) = \int_{\Ssimp} \int_0^\infty \mathbf{1}(r\bw \in A)\,r^{-2}\dd r\,\mathcal{H}(\dd\bw).
\]
Substituting into the TCCF definition,
\[
\mathcal{T}_G(\bt) = \int_{\Ssimp} \int_0^\infty (e^{\iu r \bt^\top\bw} - 1)\,r^{-2}\dd r\,\mathcal{H}(\dd\bw) = \int_{\Ssimp} \psi(\bt^\top\bw)\,\mathcal{H}(\dd\bw).
\]
Using the D-norm representation of \citet{Falk:2019}, where $G(\bz) = \exp\{-\|\bz^{-1}\|_D\}$ with $\|\bx\|_D = \E[\max_j(|x_j| W_j)]$ and $\bW \geq \mathbf{0}$ satisfying $\E[W_j] = 1$, we obtain the compact representation
\[
\mathcal{T}_G(\bt) = \E\!\left[ \psi\!\left( \max_{1 \leq j \leq d}(|\bt_j| W_j) \right) \right].
\]

\subsection{TCCF for generalized Pareto distributions}

For a multivariate generalised Pareto distribution (GPD) associated with the max-stable distribution $G$ \citep{Rootzen-Tajvidi:2006}, the exponent measure on the exceedance region $\{\bz: \max_j z_j > 1\}$ coincides with that of $G$. Consequently, $\mathcal{T}_{\rm GPD}(\bt) = \mathcal{T}_G(\bt)$ for all $\bt \in \dR^d$.

\subsection{TCCF for the conditional extremes model}

Under the conditional extremes model of \citet{Heffernan-Tawn:2004}, for a given conditioning variable $Z_j$, the standardised residual vector
\[
\bZ_{-j}^* = \frac{\bZ_{-j} - \bm{\alpha}_{|j} Z_j}{Z_j^{\bm{\beta}_{|j}}}
\]
converges in distribution to a non-degenerate limit $G_{|j}$ as $Z_j \to \infty$, independently of $Z_j$. The TCCF in the direction of $Z_j$ provides a frequency-domain characterisation:
\[
\mathcal{T}_{|j}(\bt_{-j}) = \lim_{t \to \infty} \E\!\left[ e^{\iu\bt_{-j}^\top \bZ_{-j}^*} - 1 \;\Big|\; Z_j > t \right].
\]
The phase of $\mathcal{T}_{|j}$ encodes the directional asymmetry parameters $\bm{\alpha}_{|j}$: when $\bm{\alpha}_{|j} \neq \bm{\alpha}_{|k}$ for $j \neq k$, the TCCF exhibits non-zero imaginary part for appropriately chosen frequency vectors.

\subsection{Kernel mean embedding for exponent measures}

The kernel $k_\lambda$ defined in Equation~(4.5) of the main paper induces a reproducing kernel Hilbert space (RKHS) $\mathcal{H}_k$ on $\pcone$. The mean embedding $\mu_{\mathcal{M}} = \int_{\pcone} k_\lambda(\cdot, \bz)\,\mathcal{M}(\dd\bz)$ maps each exponent measure to an element of $\mathcal{H}_k$. The squared MMD,
\[
\MMD^2(\mathcal{M}_X, \mathcal{M}_Y; k_\lambda) = \|\mu_{\mathcal{M}_X} - \mu_{\mathcal{M}_Y}\|_{\mathcal{H}_k}^2,
\]
is a proper metric on the space of exponent measures because $k_\lambda$ is a characteristic kernel \citep{Gretton-etal:2012}. The equivalence $\|\mathcal{T}_X - \mathcal{T}_Y\|_{L_2(\mu_\lambda)}^2 = \MMD^2(\mathcal{M}_X, \mathcal{M}_Y; k_\lambda)$ established in Proposition~\ref{prop:mmd} embeds the TCCF-based test within this rigorous theoretical framework.

% ======================================================================
% REFERENCES
% ======================================================================
\begin{thebibliography}{99}

\bibitem[Arcones and Gin\'e(1992)]{Arcones-Gine:1992}
Arcones, M. A. and Gin\'e, E. (1992). On the bootstrap of $U$ and $V$ statistics. \textit{The Annals of Statistics}, \textbf{20}, 655--674.

\bibitem[B\"ucher and Dette(2013)]{Buecher-Dette:2013}
B\"ucher, A. and Dette, H. (2013). Multiplier bootstrap of tail copulas with applications. \textit{Bernoulli}, \textbf{19}, 1655--1687.

\bibitem[de Haan and Ferreira(2006)]{deHaan2006}
de Haan, L. and Ferreira, A. (2006). \textit{Extreme Value Theory: An Introduction}. Springer, New York.

\bibitem[Einmahl et al.(2012)]{Einmahl:2012}
Einmahl, J. H. J., Krajina, A. and Segers, J. (2012). An M-estimator for tail dependence in arbitrary dimensions. \textit{The Annals of Statistics}, \textbf{40}, 1764--1793.

\bibitem[Einmahl et al.(2021)]{Einmahl-etal:2021}
Einmahl, J. H. J., Yang, F. and Zhou, C. (2021). Testing the multivariate regular variation model. \textit{Journal of Business \& Economic Statistics}, \textbf{39}, 907--919.

\bibitem[Falk(2019)]{Falk:2019}
Falk, M. (2019). \textit{Multivariate Extreme Value Theory and D-Norms}. Springer, Cham.

\bibitem[Gradshteyn and Ryzhik(2014)]{Gradshteyn-Ryzhik:2014}
Gradshteyn, I. S. and Ryzhik, I. M. (2014). \textit{Table of Integrals, Series, and Products}, 8th ed. Academic Press.

\bibitem[Gretton et al.(2012)]{Gretton-etal:2012}
Gretton, A., Borgwardt, K. M., Rasch, M. J., Sch\"olkopf, B. and Smola, A. (2012). A kernel two-sample test. \textit{Journal of Machine Learning Research}, \textbf{13}, 723--773.

\bibitem[Hall(1982)]{Hall:1982}
Hall, P. (1982). On some simple estimates of an exponent of regular variation. \textit{Journal of the Royal Statistical Society, Series B}, \textbf{44}, 37--42.

\bibitem[Heffernan and Tawn(2004)]{Heffernan-Tawn:2004}
Heffernan, J. E. and Tawn, J. A. (2004). A conditional approach for multivariate extreme values. \textit{Journal of the Royal Statistical Society: Series B}, \textbf{66}, 497--546.

\bibitem[Lee(1990)]{Lee:1990}
Lee, A. J. (1990). \textit{$U$-Statistics: Theory and Practice}. Marcel Dekker.

\bibitem[R\'emillard and Scaillet(2009)]{Remillard-Scaillet:2009}
R\'emillard, B. and Scaillet, O. (2009). Testing for equality between two copulas. \textit{Journal of Multivariate Analysis}, \textbf{100}, 377--386.

\bibitem[Resnick(2007)]{Resnick:2007}
Resnick, S. I. (2007). \textit{Heavy-Tail Phenomena: Probabilistic and Statistical Modeling}. Springer, New York.

\bibitem[Rootz\'en and Tajvidi(2006)]{Rootzen-Tajvidi:2006}
Rootz\'en, H. and Tajvidi, N. (2006). Multivariate generalised Pareto distributions. \textit{Bernoulli}, \textbf{12}, 917--930.

\bibitem[Rudin(1991)]{Rudin:1991}
Rudin, W. (1991). \textit{Functional Analysis}, 2nd ed. McGraw-Hill.

\bibitem[van der Vaart and Wellner(1996)]{van-der-Vaart-Wellner:1996}
van der Vaart, A. W. and Wellner, J. A. (1996). \textit{Weak Convergence and Empirical Processes}. Springer.

\end{thebibliography}

\end{document}