%~Mouliné par MaN_auto v.0.40.4 (550756fa) 2026-07-21 14:36:40
\documentclass[CRMATH,Unicode,biblatex]{cedram}

\TopicFR{Statistiques}
\TopicEN{Statistics}

\addbibresource{CRMATH_ElBoukkouri_20250305.bib}

\usepackage{mathtools}

\newcommand{\void}{\,\cdot\,}
\newcommand{\dif}{\mathop{}\!{\operatorfont{d}}}

\newcommand{\R}{\mathbb{R}}
\newcommand{\N}{\mathbb{N}}
\newcommand{\HS}{\mathcal{H}}
\newcommand{\Cclass}{\mathcal{C}(\HS)}
\newcommand{\CCclass}{\mathcal{CC}(\HS)}
\newcommand{\X}{\mathbb{X}}

\DeclareMathOperator{\cov}{cov}

\newcommand{\finHS}{\HS \ni f}
\newcommand{\hinHS}{\HS \ni h}
\newcommand{\xyinRtwo}{\mathbb{R}^{2} \ni (x,y)}
\newcommand{\xinX}{\X \ni x}

%%%---------------------------------------------------------------------------

%%% ARROWS

%%% \to
\renewcommand*{\to}{\mathchoice{\longrightarrow}{\rightarrow}{\rightarrow}{\rightarrow}}

%%% \mapsto
\let\oldmapsto\mapsto
\renewcommand*{\mapsto}{\mathchoice{\longmapsto}{\oldmapsto}{\oldmapsto}{\oldmapsto}}

%%%---------------------------------------------------------------------------

%%% DELIMITERS

\DeclarePairedDelimiter{\parens}{\lparen}{\rparen}
\DeclarePairedDelimiter{\braces}{\{}{\}}
\DeclarePairedDelimiter{\abs}{\lvert}{\rvert}
\DeclarePairedDelimiter{\norm}{\lVert}{\rVert}
\DeclarePairedDelimiter{\bracks}{[}{]}
\DeclarePairedDelimiter{\angles}{\langle}{\rangle}

%%%---------------------------------------------------------------------------

\graphicspath{{./figures/}}

\newcommand*{\mk}{\mkern -1mu}
\newcommand*{\Mk}{\mkern -2mu}
\newcommand*{\mK}{\mkern 1mu}
\newcommand*{\MK}{\mkern 2mu}

\hypersetup{urlcolor=purple, linkcolor=blue, citecolor=red}

\newcommand*{\relabel}{\renewcommand{\labelenumi}{(\theenumi)}}
\newcommand*{\romanenumi}{\renewcommand*{\theenumi}{\roman{enumi}}\relabel}
\newcommand*{\Romanenumi}{\renewcommand*{\theenumi}{\Roman{enumi}}\relabel}
\newcommand*{\alphenumi}{\renewcommand*{\theenumi}{\alph{enumi}}\relabel}
\newcommand*{\Alphenumi}{\renewcommand*{\theenumi}{\Alph{enumi}}\relabel}
\let\oldtilde\tilde
\renewcommand*{\tilde}[1]{\mathchoice{\widetilde{#1}}{\widetilde{#1}}{\oldtilde{#1}}{\oldtilde{#1}}}
\let\oldexists\exists
\renewcommand*{\exists}{\mathrel{\oldexists}}
\let\oldforall\forall
\renewcommand*{\forall}{\mathrel{\oldforall}}

\title{General reproducing properties in RKHS with application to derivative and integral operators}
\alttitle{La propriété reproduisante dans les RKHS généralisée en application aux opérateurs dérivée et intégrale}

\author{\firstname{Fatima-Zahrae} \lastname{El-Boukkouri}\CDRorcid{0009-0004-2800-2718}\IsCorresp}
\address{Institut de Mathématiques de Toulouse, Universit\'e de Toulouse, INSA, 31077 Toulouse, France}
\email{el-boukkouri@insa-toulouse.fr}

\author{\firstname{Josselin} \lastname{Garnier}\CDRorcid{0000-0002-3518-4159}}
\address{CMAP, CNRS, \'Ecole polytechnique, Institut Polytechnique de Paris, 91120 Palaiseau, France}
\email{josselin.garnier@polytechnique.edu}

\author{\firstname{Olivier} \lastname{Roustant}\CDRorcid{0009-0004-4709-7177}}
\address[1]{Institut de Mathématiques de Toulouse, Universit\'e de Toulouse, INSA, 31077 Toulouse, France}
\email{roustant@insa-toulouse.fr}

\thanks{This research has been done in the frame of the Chair PILearnWater, part of the AI Cluster ANITI, funded by the France 2030 program under the Grant agreement ANR-23-IACL-0002}
\CDRGrant[ANR]{ANR-23-IACL-0002}

\subjclass{47B32}

\keywords{\kwd{RKHS} \kwd{reproducing property} \kwd{linear operator} \kwd{derivative} \kwd{mean embedding}}
\altkeywords{\kwd{RKHS} \kwd{propriété reproduisante} \kwd{opérateur linéaire} \kwd{dérivée} \kwd{mean embedding}}

\begin{abstract}
In this paper, we consider the reproducing property in Reproducing Kernel Hilbert Spaces (RKHS). We establish a reproducing property for the closure of the class of combinations of composition operators under minimal conditions. This allows to revisit the sufficient conditions for the reproducing property to hold for the derivative operator, as well as for the existence of the mean embedding function. These results provide a framework of application of the representer theorem for regularized learning algorithms that involve data for function values, gradients, or any other operator from the considered class.
\end{abstract}

\begin{altabstract}
Dans cet article, nous considérons la propriété reproduisante dans les espaces de Hilbert à noyaux reproduisants (RKHS). Nous établissons une propriété de reproduction pour l'adhérence de la classe des combinaisons d'opérateurs de composition sous des conditions minimales. Cela nous permet de revisiter les conditions suffisantes pour que la propriété de reproduction soit valable pour l'opérateur dérivé, ainsi que pour l'existence de la fonction \emph{mean embedding}. Ces résultats donnent un cadre d'application du théorème du représentant pour les algorithmes d'apprentissage régularisés qui impliquent des données sur les valeurs de fonctions, les gradients ou tout autre opérateur de la classe considérée.
\end{altabstract}

\COI{The authors do not work for, advise, own shares in, or receive funds from any organization that could benefit from this article, and have declared no affiliations other than their research organizations.}

\begin{document}
%\input{CR-pagedemetas}
%\end{document}
\maketitle

\section*{Introduction}

Machine learning algorithms often involve penalized regression problems of the form
\[
\min_{h \in \HS} \sum_{i=1}^n \parens[\big]{Lh (x_i) - y_i}^2 + \lambda \norm{h}^2.
\]
Here $(x_i, y_i)_{1 \leq i \leq n}$ is a dataset where $x_i$ belong to some set $\X$ and $y_i \in \R$, $\lambda \in \R_+$ is a penalty coefficient, $\HS$ is a Hilbert space of real-valued functions on $\X$, and $L$ is a linear operator from $\mathcal{H}$ into $\mathcal{F}(X,\mathbb{R})$, the space of real-valued functions on $\X$. When $\HS$ is a reproducing kernel Hilbert space (RKHS), it turns out that the solution lives in a finite-dimensional space, a famous result known as ``representer theorem'' \cite{wahba1990spline, Reproducing_kernel, Representer_Theorem}.

A key property to prove this result is the generalized ``reproducing property''
\begin{equation}\label{eq:LtildeIntro}
\forall x \in \X, \ \exists \widetilde{L}(x) \in \HS \ \text{s.t.\ $\forall h \in \HS$, $Lh(x) = \angles[\big]{h, \widetilde{L}(x)}$}.
\end{equation}
Note that, when $\widetilde{L}(x)$ exists, it is necessarily unique. Notice that in the particular case when $L$ has its values in $\HS$ and admits an adjoint $L^*$, we simply have $\widetilde{L}(x) = L^* \parens[\big]{K(x, \! \void)}$. When $L$ is the identity operator, we have $\widetilde{L}(x) = K(x, \! \void)$ where $K$ is the kernel associated to $\HS$,
\[
\forall x \in \X, \ \forall h \in \HS:
\quad h(x) = \angles[\big]{h, K(x, \! \void)},
\]
which is the original reproducing property of RKHS~\cite{Reproducing_kernel}. This immediately extends to finite linear combinations, i.e.\ when $L$ has the form
\[
Lf(x) = \sum_{i=1}^{q} \alpha_{i} f\parens[\big]{v_i(x)},
\]
where $\alpha_i \in \R$ and $v_i(x) \in \X$. In that case, $\widetilde{L}(x) = \sum_{i=1}^{q} \alpha_{i} K\parens[\big]{v_{i}(x), \! \void}$. However, the generalization to more complex operators such as derivative or integral operators is not straightforward as it involves a passage to the limit (see Section~\ref{sec:linearoperators}).

This question naturally arises in machine learning problems where the function to be learned is known to satisfy linear constraints. Typical examples include learning functions with prescribed mean or approximating solutions of linear partial differential equations from data~\cite{lindgren2022spde}. In such settings, observations involve linear operators acting on the unknown function rather than pointwise evaluations, making the generalized reproducing property~\eqref{eq:LtildeIntro} a key theoretical tool.

In this paper, we consider a broad class of operators corresponding to limits of linear combinations and we give a necessary and sufficient condition on $K$ for~\eqref{eq:LtildeIntro} to hold under the assumption that~\eqref{eq:LtildeIntro} remains true when passing to the limit. This allows us to revisit the reproducing properties in RKHS for derivative and integral operators. Firstly, we focus on the derivative operator, and show that the reproducing property holds if the cross derivative of the kernel exists and is continuous on the diagonal of $\X \times \X$. Thereby we retrieve the result presented in~\cite{christmann2008support,saitoh_sawano_book} by a different approach. We prove that this condition is less restrictive than the well-known $C^2$ class condition~\cite{Zhou, L2_kernel} and we exhibit a counterexample.

Secondly, we consider the mean embedding in RKHS within the frame of (improper) Riemann integrals. We prove that it is properly defined if the kernel is absolutely integrable. This is similar to the result of~\cite{carmeli2006vector, barp2024targeted, oates2022minimum} obtained for Lebesgue integrals. Then, we go a step further and show that, under the same condition, the reproducing property holds, which does not seem to be reported in the literature. We show that it is less restrictive than the condition that $x \mapsto \sqrt{K(x,x)}$ is integrable~\cite{Mean_embedding, muandet2017kernel} and we exhibit a counterexample.

\section{Reproducing property for linear operators}\label{sec:linearoperators}

A RKHS $\HS$ is a Hilbert space of real-valued functions defined on a set $\X$ such that for all $x \in \X$, the evaluation $\hinHS \mapsto h(x) $ is continuous. Giving a RKHS is equivalent to giving a positive semidefinite function, or kernel, $K$. The link between $\HS$ and $K$ is given by the reproducing property:
\[
\forall f \in \HS, \ \forall x \in \X:
\quad f(x) = \angles[\big]{f, K(x, \! \void)}_{\mathcal{H}}.
\]
This implies, by choosing $f=K(y, \! \void)$, that $K(x,y) = \angles[\big]{K(x, \! \void), K(y, \! \void)}_\HS$ for all $x, y \in \X$.

\subsection{Main result}

Before stating our main result in Theorem~\ref{THM Property}, we first recall the definition of so-called combination of composition operators, introduced in~\cite{composition_op}.

\begin{definition}
Let $v \colon \X \to \X$ be an arbitrary function. The composition operator $T_{v}$, with function~$v$, is defined by:
\[
T_{v} \colon \mathcal{H} \to \mathcal{F}(\X, \R),
\quad T_{v}(f) \coloneqq f \circ v.
\]
We call a combination of composition operators with functions $v_i$ and weights $\alpha_i \in \R$ for $ 1 \leq i \leq q $ the operator:
\[
L = \sum_{i=1}^q \alpha_i T_{v_i}.
\]
The class of combination of composition operators on functions on $\mathcal{H}$ is denoted $ \mathcal{CC(\HS)}$.
\end{definition}

Let $L = \sum_{i=1}^{q} \alpha_{i} T_{v_i} \in \CCclass$. Notice that for all $x \in \X$, $Lf(x) = \sum_{i=1}^{q} \alpha_{i} f\parens[\big]{v_i(x)}$. Thus, the linear form $\finHS \mapsto Lf(x)$ is continuous as a linear combination of evaluations of $\HS$. Following~\cite[Section~4.4]{berlinet2011reproducing}, we will denote by $\widetilde{L}(x) \in \HS$ its representer, i.e.\ the unique element of $\HS$ verifying
\begin{equation}\label{eq:Ltilde}
Lf(x) = \angles[\big]{f, \widetilde{L}(x)}.
\end{equation}
We have explicitly
\begin{equation}\label{eq:L_ntilde_expl}
\widetilde{L}(x) = \sum_{i=1}^{q} \alpha_{i} K\parens[\big]{v_{i}(x), \! \void}.
\end{equation}
Mind that $\widetilde{L}(x) \neq L\parens[\big]{K(x, \! \void)} = \sum_{i=1}^{q} \alpha_{i} K\parens[\big]{x, v_{i}(\void)}$.

The class $\CCclass$ is limited to finite linear combinations. We now define a natural extension of this class by considering its closure with respect to pointwise convergence.

\begin{definition}\label{def:Cclass}
We define the class of operators
\[
\Cclass= \braces[\Big]{L \colon \mathcal{H} \rightarrow \mathcal{F}(\X, \R): {\exists (L_n)_{n \in \mathbb{N}} \in \mathcal{CC(\HS)}^{\N}} \ \textnormal{s.t.\ $L(f) = \lim_{n \to +\infty} L_n(f) \ \forall f \in \mathcal{H}$}},
\]
where the limit is in the pointwise sense.
\end{definition}

If $L \in \Cclass$, writing $L_n = \sum_{i=1}^{q_n} \alpha_{i,n} T_{v_{i,n}}$ (where $q_n \in \N, \alpha_{i,n} \in \R, v_{i,n} \in \mathcal{F}({\X, \X})$), we have for all $f \in \HS $ and all $x \in \X$,
\[
L(f)(x) = \lim_{n \to +\infty} \sum_{i=1}^{q_n} \alpha_{i,n} f\parens[\big]{v_{i,n}(x)}.
\]
As an example, when $\X = \R$, the class $\Cclass$ includes derivative and integral operators:
\begin{itemize}
\item if the functions of $\HS$ are differentiable, the derivative operator $L \colon f \mapsto f'$ is written as $\lim_{n \to +\infty} L_n$ where $L_n = nT_{v_{2,n}}- n T_{v_{1, n}} $ with $v_{2,n} \colon x \mapsto x + \frac{1}{n}$ and $v_{1, n} \colon x \mapsto x $; here $\widetilde{L_n}(x) = n K\parens[\big]{v_{2,n}(x), \! \void} - n K\parens[\big]{v_{1,n}(x), \! \void}$;
\item if the functions of $\HS$ are continuous, the averaging operator $L \colon f \mapsto x^{-1} \int_0^x f(t) \dif t$ (with $L(f)(0)=f(0)$) is written as $\lim_{n \to +\infty} L_n$ where $L_n = \sum_{i=1}^n \frac{1}{n} T_{v_{i,n}}$ with $v_{i,n} \colon x \mapsto \frac{i}{n} x$; here $\widetilde{L_n}(x) = \sum_{i=1}^n \frac{1}{n} K\parens[\big]{v_{i,n}(x), \! \void}$.
\end{itemize}

We now recall the Loève criterion for convergence of sequences in Hilbert spaces.

\begin{proposition}[Loève criterion]\label{loève}
Let $\HS$ be a Hilbert space with inner product $\langle \void, \! \void \rangle$. Let $(h_n)_{n \in \N}$ be a sequence of $\mathcal{H}$. The following statements are equivalent:
\begin{enumerate}\romanenumi
\item \label{loève_i} $(h_n)_{n \in \N}$ converges in $\mathcal{H}$;
\item \label{loève_ii} the double sequence $\parens[\big]{\langle h_n, h_m \rangle}_{n, m \in \N}$ has a finite limit as $n, m$ tend to $+\infty$.
\end{enumerate}
\end{proposition}

In the whole paper the convergence of double sequences $(u_{n,m})_{n,m \in \N}$ ---~such as the one appearing in~\eqref{loève_ii}, is in the uniform sense (also known as Pringsheim sense~\cite{tripathy2005convergent}): if $c$ is its limit, $\forall \varepsilon > 0$, $\exists n_0 \in \N$, $\forall n,m \geq n_0$, $\abs{u_{n,m} - c} \leq \varepsilon$.

\begin{proof}
The proof can be found in~\cite[Section~3.5]{cramer} when $\HS$ is a space of square-integrable random variables. For completeness, we recall here the main arguments.

The direct sense \eqref{loève_i}~$\Rightarrow$~\eqref{loève_ii} is obvious by continuity of the scalar product.

Let us now prove \eqref{loève_ii}~$\Rightarrow$~\eqref{loève_i}. Denote by $c$ the limit of $\parens[\big]{\langle h_n, h_m \rangle}$ when $n, m$ tend to infinity. We have, for all $n, m \in \mathbb{N}$,
\begin{equation}\label{eq:cauchy}
\norm{h_n - h_m}^2 = \langle h_n, h_n\rangle + \langle h_m, h_m\rangle -2 \langle h_n, h_m\rangle.
\end{equation}
This implies that $\norm{h_n - h_m}^2 \to c + c - 2c = 0$ when $ n,m \to + \infty$. This proves that $(h_n)$ is a Cauchy sequence, and thus converges in $\HS$.
\end{proof}

For all operator $L$ and for all function $F \colon \X \times \X \to \R$ denote
\[
L^\ell F \colon \X \times \X \to \mathbb{R},
\quad (x, y) \mapsto L\parens[\big]{F(\void, y)}(x),
\]
and
\[
L^r F \colon \X \times \X \to \R,
\quad (x, y) \mapsto L\parens[\big]{F(x, \! \void)}(y).
\]
Notice that with these notations, for all $x, y \in \X$ we have $(L^\ell F)(\void, y) = L \parens[\big]{F(\void, y)}$ and $(L^r F)(x, \! \void) = L \parens[\big]{F(x, \! \void)}$. We will also simply write $L^\ell L^r$ instead of $L^\ell \circ L^r$.

\begin{theorem}\label{THM Property}
Let $\mathcal{H}$ be a RKHS with kernel $K$.
Let $(L_n)_{n \in \N}$ be a sequence of\/ $\CCclass$. We denote by $\widetilde{L_n}(x) \in \HS$ the representer of the linear form $\finHS \mapsto L_nf(x)$ (see~\eqref{eq:Ltilde}). The following conditions are equivalent:
\begin{enumerate}\romanenumi
\item \label{THM Property_i} $\parens[\big]{L_n^\ell L_m^r K(x,x)}_{n, m}$ converges when $n, m$ tend to $+\infty$ for all $x \in \X$;
\item \label{THM Property_ii} $\parens[\big]{\widetilde{L_n}(x)}_{n \in \N}$ converges in $\mathcal{H}$ for all $x \in \X $.
\end{enumerate}
In that case, denote $\widetilde{L}(x) \coloneqq \lim_{n \to +\infty} \widetilde{L_n}(x) \in \HS$. Then, for all $f \in \HS$, $L_n f$ converges pointwise. Define the operator $L \in \Cclass$ by $Lf(x) = \lim_{n \to +\infty} L_n f(x)$ (for all $f \in \HS$, $x \in \X$). Then, for all $x \in \X$ the mapping $\finHS \mapsto Lf(x)$ is continuous and the reproducing property holds:
\[
\forall x \in \X, \ \forall f \in \HS:
\quad Lf(x) = \angles[\big]{f, \widetilde{L}(x)}.
\]
Finally, $\norm[\big]{\widetilde{L}(x)}^2 = L^\ell L^r K(x,x)$.
\end{theorem}

\begin{proof}
Let $x \in \X$. We first prove that
\begin{equation}\label{eq:LLK}
\angles[\big]{\widetilde{L_n}(x), \widetilde{L_m}(x)} = L_n^\ell L_m^r K(x, x).
\end{equation}
Let $y \in \X$. Denote $F_m = L_m^r K$. Using the definition~\eqref{eq:Ltilde} of $\widetilde{L_m}(y)$ (with $f = K(x, \! \void)$), we have
\begin{equation}\label{eq:definitionF_m}
F_m (x, y) = \parens[\big]{L_m K(x, \! \void)}(y) = \angles[\big]{K(x, \! \void), \widetilde{L_m}(y)} = \widetilde{L_m}(y)(x).
\end{equation}
Thus, $F_m(\void, y) = \widetilde{L_m}(y)$ by uniqueness. Similarly, noting that $F_m(\void, y) \in \HS$ and using the definition~\eqref{eq:Ltilde} of $\widetilde{L_n}(x)$ (with $f = F_m(\void, y)$), we have
\begin{equation}\label{eq:LnFm}
L_n^\ell F_m (x, y) = \parens[\big]{L_n F_m(\void, y)}(x) = \angles[\big]{F_m(\void, y), \widetilde{L_n}(x)}.
\end{equation}
Finally, we get $ L_n^\ell L^r_m K (x, y) = \angles[\big]{\widetilde{L_m}(y), \widetilde{L_n}(x)}$, which gives~\eqref{eq:LLK} when $y=x$.

From Equation~\eqref{eq:LLK}, by the Loève criterion (Proposition~\ref{loève}), the sequence $\widetilde{L_n}(x)$ converges in the Hilbert space $\HS$ for all $x \in \X$ if and only if
\[
\lim_{n,m \to \infty} L_n^\ell L_m^r K (x, x)
\]
exists at each point $x \in \X$. This proves that \eqref{THM Property_i}~and~\eqref{THM Property_ii} are equivalent.

Suppose now that \eqref{THM Property_i}~or~\eqref{THM Property_ii} are verified and denote $\widetilde{L}(x) \coloneqq \lim_{n \to +\infty} \widetilde{L_n}(x) \in \HS$. Writing the equality~\eqref{eq:Ltilde} defining $\widetilde{L_n}(x)$, for a given $f \in \HS$ and $x \in \X$,
\[
L_n f(x) = \angles[\big]{f, \widetilde{L_n}(x)},
\]
and taking the limit when $n$ tends to infinity, we obtain that $L_n f(x)$ converges. Denoting by $Lf(x)$ its limit, we thus have:
\[
Lf(x) = \angles[\big]{f, \widetilde{L}(x)}.
\]
This ensures that the mapping $\finHS \mapsto Lf(x)$ is continuous.

Finally, by taking the limit in~\eqref{eq:LLK}, we get:
\[
\norm[\big]{\widetilde{L}(x)}^2 = L^\ell L^r K(x,x).
\]
Indeed, let $F = L^rK$ and $x,y \in \X$. By taking the limit in~\eqref{eq:definitionF_m}, we get $F(\void,y) = \widetilde{L}(y) \in \HS$. Then, taking the limit in~\eqref{eq:LnFm}, we obtain $L^\ell F (x, y) = \angles[\big]{F(\void, y), \widetilde{L}(x)}$, or equivalently,
\[
\angles[\big]{\widetilde{L}(x), \widetilde{L}(y)} = L^\ell L^r K(x,y). \qedhere
\]
\end{proof}

\section{Derivative reproducing property}\label{sec:derivative}

The reproducing property for the derivative operator has been extensively studied in the literature under various conditions. For example, in~\cite{Zhou}, this result is established for Mercer kernels of class $C^2$. A weaker condition is given in~\cite[Section~2]{saitoh_sawano_book} and~\cite[Corollary~4.36]{christmann2008support}, namely that the cross derivative $\frac{\partial^2 K}{\partial x_1 \partial x_2}$ exists and is continuous on $\X \times \X$. In what follows, we obtain a similar result, demanding the continuity at the neighborhood of the diagonal of $\X \times \X$, by following a different route with the class of compositions of operators. We argue that the continuity assumption, sometimes omitted~\cite{gaetan2010second, cramer}, is important when the RKHS is not supposed a priori to contain differentiable functions.

\begin{theorem}\label{thm:derivative}
Let $\HS$ be a RKHS of real-valued functions defined on a non-empty open set $\X \subset \R$, with reproducing kernel $K$. If $K$ is of class $C^1$ and $\frac{\partial^2 K}{\partial x_1 \partial x_2}$ exists and is continuous on a neighborhood of the diagonal, then:
\begin{enumerate}\alphenumi
\item \label{thm:derivative_a} for all $x \in \X$, $\frac{\partial K}{\partial x_1}(x, \! \void) $ is in $\HS$, and $\norm[\big]{\frac{\partial K}{\partial x_1}(x, \! \void)}^2 = \frac{\partial^2 K}{\partial x_1 \partial x_2}(x,x)$;
\item \label{thm:derivative_b} all functions of $\HS$ are differentiable, the mapping $\finHS \mapsto f'(x)$ is continuous, and the derivative reproducing property holds:
\[
\forall x \in \X, \ \forall f \in \HS:
\quad f'(x) = \angles[\Big]{f, \frac{\partial K}{\partial x_1}(x, \! \void)}.
\]
\end{enumerate}
\end{theorem}

We will need the well-known fact that convergence in $\HS$ implies pointwise convergence (see~\cite{berlinet2011reproducing}).

\begin{lemma}\label{lemma:RKHSandPointwiseConvergence}
Let $\HS$ a RKHS on $\X$ with kernel $K$, and let $(h_x)_{x \in \X}$ be a family of functions of $\HS$. If for some $x_0 \in \X$, $h_x$ converges in $\HS$ when $x \to x_0$, then $h_x$ converges pointwise to the same limit.
\end{lemma}

\begin{proof}
Let $h_\infty \in \HS$ be the limit of $h_x$ when $x \to x_0$. Let $y \in \X$. By the reproducing property and the Cauchy--Schwarz inequality,
\[
\abs[\big]{h_x(y) - h_\infty(y)}
= \abs[\big]{\angles[\big]{h_x - h_\infty, K(y, \! \void)}}
\leq \norm{h_x - h_\infty} \norm[\big]{K(y, \! \void)}.
\]
The result follows.
\end{proof}

\begin{proof}[Proof of Theorem~\ref{thm:derivative}]
Suppose that $K$ is of class $C^1$ and $\frac{\partial^2 K}{\partial x_1 \partial x_2}$ exists and is continuous on a neighborhood of the diagonal.

Let $(u_n)$ be a real-valued sequence converging to $0$ as $n$ tends to infinity with $u_n\neq~0$ $ \forall n \in \N$.

Let $n \in \N$ and $x \in \X$ and consider the operator $L_n \in \CCclass$ defined by
\[
L_n = \frac{T_{v_{2,n}} - T_{v_{1,n}}}{u_n},
\quad \text{where} \quad v_{2,n}(t) = t +u_n, \quad v_{1,n}(t) = t.
\]
We have
\[
L_n(f)(x) = \frac{f(x+u_n) - f(x)}{u_n} = \angles[\big]{f, \widetilde{L_n}(x)},
\]
with
\[
\widetilde{L_n}(x) = \frac{K(x + u_n, \! \void) -K(x, \! \void)}{u_n}.
\]
Notice that
\[
L_n^\ell L_m^r K(x,x) =\frac{K(x + u_n, x + u_m) - K(x+u_n, x) - K(x, x+u_m) + K(x,x)}{u_n u_m}.
\]
As $\frac{\partial^2 K}{\partial x_1 \partial x_2}$ exists and is continuous on a neighborhood of the diagonal, the function $(u', v') \mapsto \frac{\partial^2 K}{\partial x_1 \partial x_2}(x + u',x + v')$ is continuous on a neighborhood of $(0, 0)$. This justifies that there exists $n_0 \in \N$ such that for all $n,m \geq n_0$,
\[
L_n^\ell L_m^r K(x,x) = \frac{\int_0^{u_n} \int_0^{u_m}\frac{\partial^2 K}{\partial x_1 \partial x_2}(x + u',x + v') \dif u' \dif v'}{u_n u_m},
\]
which can be obtained by two successive integrations with respect to $u'$ and $v'$ respectively. As $ \frac{\partial^2 K}{\partial x_1 \partial x_2}$ is continuous at $(x,x)$, then for a given $\epsilon > 0$, there exists $ r > 0$ such that for all $u',v' \in [-r, r]$,
\[
\abs[\Bigg]{\frac{\partial^2 K}{\partial x_1 \partial x_2}(x + u', x+ v') - \frac{\partial^2 K}{\partial x_1 \partial x_2}(x,x)} < \epsilon.
\]
As the sequence $(u_n)$ converges to $0$, there exists $N_0 \in \N$, such that for all $n \geq N_0$, $u_n \in [-r, r] $. Then for all $n,m \geq \max(n_0, N_0)$,
\[
\abs[\Bigg]{L_n^\ell L_m^r K(x,x) - \frac{\partial^2 K}{\partial x_1 \partial x_2}(x,x)}
\leq \abs*{\int_0^{u_n} \int_0^{u_m} \frac{\abs*{\frac{\partial^2 K}{\partial x_1 \partial x_2}(x + u',x + v') - \frac{\partial^2 K}{\partial x_1 \partial x_2}(x,x)} \dif u' \dif v'}{u_n u_m}}
\leq \epsilon.
\]
This implies that $\parens[\big]{L_n^\ell L_m^r K(x,x)}_{n, m}$ converges when $n, m$ tend to $+\infty$ to $\frac{\partial^2 K}{\partial x_1 \partial x_2}(x,x)$.

Then using Theorem~\ref{THM Property}, we conclude that $\parens[\big]{\widetilde{L_n}(x)}_{n \in \N}$ converges in $\HS$. Denoting by $\widetilde{L}(x)$ its limit, we have by Lemma~\ref{lemma:RKHSandPointwiseConvergence},
\[
\frac{\partial K}{\partial x_1}(x, \! \void) = \widetilde{L}(x).
\]
As the limit of $\widetilde{L_n}(x)$ does not depend on the sequence $u_n$, we can conclude that $\frac{K(x + t, \! \void) - K(x, \! \void)}{t} $ converges in $\HS$ when $t$ tends to $0$ to $\frac{\partial K}{\partial x_1}(x, \! \void)$.

To conclude the proof of~\eqref{thm:derivative_a}, let us show that
\[
\norm[\Bigg]{\frac{\partial K}{\partial x_1}(x, \! \void)}^2 = \frac{\partial^2 K}{\partial x_1 \partial x_2}(x,x).
\]
From Theorem~\ref{THM Property}, for all $f \in \HS$, $L_n f$ converges pointwise, and with $Lf(x) = \lim_{n \to +\infty} L_n f(x)$, we have
\[
\norm[\big]{\widetilde{L}(x)}^2 = L^\ell L^r K(x,x).
\]
Recall that $\parens[\big]{L_n^\ell L_m^r K(x,x)}_{n, m}$ converges to $\frac{\partial^2 K}{\partial x_1 \partial x_2}(x,x)$ when $n, m$ tend to $+\infty$. In particular, $\parens[\big]{L_n^\ell L_n^r K(x,x)}$ converges to $\frac{\partial^2 K}{\partial x_1 \partial x_2}(x,x)$ when $n$ tends to $+\infty$. This implies that $ L^\ell L^r K(x,x) = \frac{\partial^2 K}{\partial x_1 \partial x_2}(x,x)$. And thus,
\[
\norm[\Bigg]{\frac{\partial K}{\partial x_1}(x, \! \void)}^2 = \frac{\partial^2 K}{\partial x_1 \partial x_2}(x,x).
\]

It remains to show~\eqref{thm:derivative_b}. Let $f \in \HS$ and $x \in \X$. We know that $\frac{\partial K}{\partial x_1}(x, \! \void)$ is in $\HS$. We have
\[
\frac{f(x + t) -f(x)}{t} = \angles[\Big]{f, \frac{K(x + t, \! \void) - K(x, \! \void)}{t}}.
\]
As $\frac{K(x + t, \! \void) - K(x, \! \void)}{t} $ converges to $\frac{\partial K}{\partial x_1}(x, \! \void)$ in $\HS$ when $t$ tends to $0$, we obtain that $f$ is differentiable at $x$ and
\[
f'(x) =
\angles[\Big]{f, \frac{\partial K}{\partial x_1}(x, \! \void)}.
\]
This implies that the mapping $\finHS \mapsto f'(x)$ is continuous.
\end{proof}

\begin{remark}
Theorem~\ref{thm:derivative} implies that if $Y$ is a centered second-order random process with covariance function $K$, then $Y$ is everywhere differentiable in quadratic mean (q.m.), the second-order cross derivative $\frac{\partial^2 K}{\partial x_1 \partial x_2}$ exists everywhere and $\cov\parens[\big]{Y'(s), Y'(t)} = \frac{\partial^2 K}{\partial x_1 \partial x_2}(s, t)$~\cite{gaetan2010second}.
\end{remark}

\begin{remark}
It may be not sufficient to assume only the existence of the cross derivative on the diagonal (without assuming its continuity on a neighborhood of the diagonal). Indeed, in the proof of Theorem~\ref{thm:derivative}, we can write
\[
L_n^\ell L_m^r K(x,x) = \frac{\frac{K(x + u_n, x + u_m) - K(x + u_n, x)}{u_m} - \frac{K(x, x + u_m) - K(x, x)}{u_m}}{u_n},
\]
which leads to the fact that $\lim_{n \to +\infty} \lim_{m \to +\infty} L_n^\ell L_m^r K(x,x) $ exists. Here, the limit is taken sequentially firstly in $m$ and secondly in $n$. This does not imply that the limit of this double sequence exists when $n, m$ tend to infinity, which is required by the Loève criterion (Proposition~\ref{loève}).

Finally, notice that the continuity of the cross derivative can be omitted if one assumes in addition from the outset that all functions in $\HS$ are differentiable, as proved in~\cite[Lemma~4]{barp2024targeted}.
\end{remark}

Theorem~\ref{thm:derivative} gives a sufficient condition on $K$ for a derivative reproducing property to hold, namely the existence of the cross-derivative and its continuity on a neighborhood of the diagonal. This is less restrictive than the $C^2$ condition on $K$ found in the literature. Indeed, the following example exhibits a kernel $K$ that is not of class $C^2$ but such that $\frac{\partial^2 K}{\partial x_1 \partial x_2}$ exists and is continuous.

\begin{example}[Non-$C^2$ kernel with finite cross derivative]
Consider the function:
\[
f \colon \R \to \R,
\quad f(x) =
\begin{cases}
	x^4 \sin\parens[\big]{\tfrac{1}{x}}
	& \text{if $x \neq 0$},
\\	0
	& \text{if $x = 0$}.
\end{cases}
\]
We notice that $f$ is twice differentiable on $(-\infty, 0)$ and $(0, \infty)$. For $x \in \R \backslash \{0\}$, the first and second derivatives are:
\[
f'(x) = 4x^3 \sin\parens[\Bigg]{\frac{1}{x}} - x^2 \cos\parens[\Bigg]{\frac{1}{x}}
\]
and
\[
f''(x) = 12x^2 \sin\parens[\Bigg]{\frac{1}{x}} - 6x \cos\parens[\Bigg]{\frac{1}{x}} - \sin\parens[\Bigg]{\frac{1}{x}}.
\]
Observe that
\[
\frac{f(x) - f(0)}{x} = \frac{x^4 \sin\parens[\big]{\tfrac{1}{x}}}{x} = x^3 \sin\parens[\Bigg]{\frac{1}{x}},
\]
which proves that $f$ is differentiable at $x = 0$ with $f'(0) = 0$. Similarly,
\[
\frac{f'(x) - f'(0)}{x} = \frac{f'(x)}{x} = \parens[\Bigg]{4x^2 \sin\parens[\Bigg]{\frac{1}{x}} - x \cos\parens[\Bigg]{\frac{1}{x}}},
\]
which shows that $f$ is twice differentiable at $x = 0$ with $f''(0) = 0$. We conclude that $f$ is twice differentiable on $\R$, with the second derivative given by:
\[
f''(x) =
\begin{cases}
	12x^2 \sin\parens[\big]{\tfrac{1}{x}} - 6x \cos\parens[\big]{\tfrac{1}{x}} - \sin\parens[\big]{\tfrac{1}{x}}
	& \text{if $x \neq 0$},
\\	0
	& \text{if $x = 0$}.
\end{cases}
\]
However, $f''$ has no limit at $0$, due to the term $\sin\parens[\big]{\tfrac{1}{x}}$, which means that $f$ is not $C^2$.

Now, consider the rank-one kernel $K$:
\[
K \colon \R\times\R \to \R, 
\quad K(x,y) = f(x)f(y).
\]
$K$ is not of class $C^2$ since for all $y \in \R$, the function $x \mapsto \frac{\partial^2 K}{\partial x_1^2}(x,y) = f''(x) f(y)$ is not continuous at $x=0$. Nevertheless, the cross derivative $ \frac{\partial^2 K}{\partial x_1 \partial x_2}(x, y) = f'(x) f'(y) $ exists for all $x, y \in \mathbb{R}$ and is continuous.
\end{example}

\section{Mean embedding in RKHS}\label{sec:meanembedding}

In this subsection, we consider a random variable $X$ defined on $\X \subset \R$ with probability density function $p$ and a RKHS $\mathcal{H}$ of real-valued functions defined on $\X$ with kernel $K$. We seek to establish the minimal condition under which there exists a function $\mu_p \in \HS$ satisfying:
\[
\mathbb{E}_{X \sim p} f(X) = \langle f, \mu_p \rangle,
\quad \forall f \in \HS.
\]
The function $\mu_p$ is called the mean embedding of $p$.

Fukumizu~et~al.~\cite{Mean_embedding} proved that $\mu_p$ exists if $\mathbb{E}_{X \sim p} \sqrt{K(X,X)} < \infty $ in the sense of the Lebesgue integral. However, the function $\mu_p$ exists under a less restrictive condition as we will demonstrate.

In what follows, our results will use the integral in the Riemann sense. To demonstrate it, we will need to use some notions related to the improper Riemann integral~\cite{Zorich}.

\begin{definition}[Admissible set]\label{def:admissible_set}
A set $E \subset \mathbb{R}^d$ is admissible if it is bounded and its boundary\/ $\partial E$ has Lebesgue measure zero. The integral of a function $f$ over $E$ is defined as:
\[
\int_E f(x) \dif x = \int_I f(x) \, \mathbb{1}_E(x) \dif x,
\]
where $I$ is some interval in $\mathbb{R}^d$ (i.e.\ a set of the form $\braces[\big]{x \in \R^d, \ a^i\leq x_i \leq b^i, \ i=1,\dots,d}$, for $a^i<b^i\in \mathbb{R}$) such that $E\subset I$.
\end{definition}

If the integral on the right-hand side of this equality exists, then we say that $f$ is Riemann integrable over $E$. Lebesgue’s criterion~\cite[Section~11.1.2, Theorem~1]{Zorich} for the existence of the Riemann integral over an interval implies that a function $f \colon E\to \R$ is integrable over an admissible set $E$ if and only if it is bounded and continuous at almost all points of~$E$~\cite[Section~11.2.2, Theorem~1]{Zorich}.

\begin{definition}[Exhaustion]
An exhaustion of a set $E \subset \mathbb{R}^d$ is a sequence of admissible sets $(E_s)_{s \in \N}$ such that:
\[
E_s \subset E_{s+1} \subset E
\quad \text{for all $s \in \mathbb{N}$},
\]
and
\[
\bigcup_{s=1}^\infty E_s = E.
\]
\end{definition}

The notion of exhaustion defines the improper integral over a set $E \subset \R^d$.
\begin{definition}[Riemann improper integral]
Let $(E_s)_{s \in \N}$ be an exhaustion of the set $E$ and suppose the function $f \colon E \to \R $ is Riemann integrable on the $E_s$ for all $s \in \N$. If the sequence $ \int_{E_s} f(x) \dif x$ converges and has a limit independent of the choice of the exhaustion, then $f$ admits an improper integral over $E$ defined by
\[
\int_E f(x) \dif x = \lim_{s \to \infty} \int_{E_s} f(x) \dif x.
\]
\end{definition}

\begin{proposition}[Comparison test for improper integrals, \cite{Zorich}]\label{prop:comp}
Let $f$ and $g$ be functions defined on $E$ and integrable over exactly the same admissible subsets of $E$, and suppose $\abs{f} \leq g$ on $E$. If the improper integral $\int_E g(x) \dif x$ exists, then the improper integrals $\int_E \abs[\big]{f(x)} \dif x$ and $\int_E f(x) \dif x$ also exist.
\end{proposition}

Using improper integrals, we now establish the condition for the existence of the mean embedding $\mu_p$ in $\HS$ when $\X$ is not necessarily an interval. Note that $\X$ can be unbounded and $K$ can be unbounded.

\begin{theorem}\label{thm:mean_embedding}
Let $\mathcal{H}$ be a RKHS of real-valued functions defined on $\X$, with reproducing kernel~$K$. Assume that $K$ and $p$ are continuous almost everywhere and locally bounded on $\X \times \X$ and $\X$ respectively, i.e.\ $K$ (resp.\ $p$) is bounded on all bounded subsets of $\X \times \X$ (resp.\ $\X$). Assume that the Riemann improper integral $\int_{\X \times \X} \abs[\big]{K(x,y)}p(x)p(y) \dif x \dif y$ exists. Then:
\begin{enumerate}
\item \label{thm:mean_embedding_a} the mean embedding $\mu_p \colon \xinX \mapsto \int_\X K(x,y) p(y) \dif y$ exists and is in $\HS$; moreover, $\norm{\mu_p}^2 = \iint_{\X \times \X} K(x, y) p(x) p(y) \dif x \dif y$;
\item \label{thm:mean_embedding_b} for all function $f \in \HS$, the function $x \mapsto f(x)p(x) $ admits an improper integral on $\X$ and the mapping $\finHS \mapsto \mathbb{E}_{X \sim p}\bracks[\big]{f(X)} = \int_\X f(x) p(x) \dif x$ is continuous; furthermore, the reproducing property holds:
\[
\forall f \in \HS,
\quad \mathbb{E}_{X \sim p}\bracks[\big]{f(X)} = \langle f, \mu_p \rangle.
\]
\end{enumerate}
\end{theorem}

\begin{proof}
Let $(E_s)_{s\in \N}$ be an exhaustion of $\X$. For all $s \in \N$, as $E_s$ is bounded, there exists an interval~$I_s$ in $\R$ such that for all $f \in \HS$,
\[
\int_{E_s} f(x) p(x) \dif x = \int_{I_s} f(x) \mathbb{1}_{E_s}(x) p(x) \dif x.
\]
Let $s \in \N$ and consider for all $n \in \N^*$, a partition $(v^s_{i,n})_{0 \leq i \leq n}$ of the interval $I_s$ composed of $n$ nodes. Suppose that the mesh of this partition $\lambda_n$ defined as $\lambda_n = \max_{1 \leq i \leq n} (v^s_{i,n} - v^s_{i-1,n})$ converges to 0 as $n$ tends to $\infty$. Let $n \in \N^*$ and consider the following operator in $\CCclass$,
\[
L_{s,n} = \sum_{i=1}^n(v^s_{i,n} - v^s_{i-1,n})p(v^s_{i,n}) \mathbb{1}_{E_s}(v^s_{i,n}) T_{v^s_{i,n}}.
\]
Thus, by~\eqref{eq:L_ntilde_expl},
\[
\forall x \in \X,
\quad \widetilde{L_{s,n}}(x) = \sum_{i=1}^n(v^s_{i,n} - v^s_{i-1,n})p(v^s_{i,n}) \mathbb{1}_{E_s}(v^s_{i,n})K(v^s_{i,n}, \! \void).
\]
From now on, as $L_{s,n}f(x)$ and $\widetilde{L_{s,n}}(x)$ do not depend on $x$, we will omit $x$ and simply write $L_{s,n}f$ and $\widetilde{L_{s,n}}$.

Notice that for all $n,m \in \N^*$, and all $x \in \X$,
\[
L^\ell_{s,n} L^r_{s,m} K(x,x) = \sum_{i=1}^n \sum_{j=1}^m (v^s_{i,n} - v^s_{i-1,n})p(v^s_{i,n}) (v^s_{j,m} - v^s_{j-1,m})p(v^s_{j,m}) \mathbb{1}_{E_s \times E_s}(v^s_{i,n},v^s_{j,m}) K(v^s_{i,n},v^s_{j,m}).
\]
From the assumptions on $K$ and $p$, the function $(x,y) \mapsto K(x,y) p(x)p(y)$ is continuous and bounded on the bounded set $E_s \times E_s$. Then, by~\cite[Section~11.2.2, Theorem~1]{Zorich} the Riemann integral $\iint_{E_s \times E_s}K(x, y) p(x)p(y) \dif x \dif y$ exists. As the function $x,y \mapsto K(x,y)p(x)p(y)\mathbb{1}_{E_s \times E_s}(x,y)$ is continuous almost everywhere on the interval $I_s$, the double sum above converges as $n, m$ tend to $+\infty$ to $\int_{I_s \times I_s} K(x,y)p(x)p(y)\mathbb{1}_{E_s \times E_s}(x,y) \dif x \dif y $, which is equal by Definition~\ref{def:admissible_set} to $\int_{E_s \times E_s} K(x,y)p(x)p(y) \dif x \dif y $. Then using Theorem~\ref{THM Property} on the sequence $(L_{s,n})_{n \in \N}$, we conclude that $(\widetilde{L_{s,n}})_{n \in \N}$ converges in $\HS$. Denote by $\widetilde{L_s}$ its limit. Furthermore, for all $f \in \HS$, $L_{s,n}f$ converges pointwise with respect to $n$, and we denote $L_s f = \lim_{n \to \infty} L_{s,n} f$. Finally (still by Theorem~\ref{THM Property}),
\begin{equation}\label{eq:scalarf}
\forall f \in \HS,
\quad L_s f = \langle f, \widetilde{L_s} \rangle.
\end{equation}
For all $x \in \X$, the function $y \mapsto K(x,y)p(y)$ is continuous almost everywhere and bounded on the bounded set $E_s$. Thus, it is Riemann integrable on $E_s$~\cite[Section~11.2.2, Theorem~1]{Zorich}. Then, for all $x \in \X$,
\[
\widetilde{L_s}(x) =
\lim_{n \to \infty} \sum_{i=1}^n(v^s_{i,n} - v^s_{i-1,n})p(v^s_{i,n})\mathbb{1}_{E_s}(v^s_{i,n})K(v^s_{i,n}, x) = \int_{E_s} K(x,y) p(y) \dif y.
\]
Thus, by Lemma~\ref{lemma:RKHSandPointwiseConvergence},
\begin{equation}\label{eq:Lntilde_def}
\widetilde{L_s} = \int_{E_s} K(\void,y) p(y) \dif y.
\end{equation}
From~\eqref{eq:scalarf}, we obtain that for all $f \in \HS $,
\[
L_sf = \lim_{n \to \infty} \sum_{i=1}^n(v^s_{i,n} - v^s_{i-1,n})p(v^s_{i,n})\mathbb{1}_{E_s}(v^s_{i,n}) f(v^s_{i,n}) = \langle f, \widetilde{L_s} \rangle.
\]
As this limit does not depend on the subdivision $(v^s_{i,n})_{0 \leq i \leq n, \: n \in \N}$, this implies that all functions of~$\HS$ are $p$-integrable on $E_s$ and,
\[
\forall f \in \HS,
\quad \int_{E_s} f(x) p(x) \dif x = \langle f, \widetilde{L_s} \rangle.
\]
In particular, for $t \in \N$, choosing $f = \widetilde{L_t}$ (which is in $\HS$) and using~\eqref{eq:Lntilde_def}, we have
\[
\langle \widetilde{L_s}, \widetilde{L_t} \rangle = \int_{E_s} \parens[\Bigg]{\int_{E_t} K(x,y)p(x) p(y) \dif y} \dif x.
\]
By the same argument used above, the Riemann integral $\int_{E_s \times E_t}K(x,y) p(x)p(y) \dif x \dif y$ exists. By Fubini's theorem for Riemann integrals~\cite[Section~11.4]{Zorich}, we have
\begin{equation}\label{eq:scalarprodImproper}
\langle \widetilde{L_s}, \widetilde{L_t} \rangle = \int_{E_s \times E_t} K(x,y)p(x)p(y) \dif x \dif y.
\end{equation}
Let us show that when $s, t$ tend to $+\infty$, $\langle \widetilde{L_s}, \widetilde{L_t} \rangle$ converges to $\int_{\X \times \X}K(x,y) p(x)p(y) \dif x \dif y$. Let $s, t \in \N$, as $x,y \mapsto \abs[\big]{K(x,y)}p(x)p(y) $ is integrable on $\X \times \X$ and using Proposition~\ref{prop:comp} with $f(x,y) = K(x,y)p(x)p(y) \mathbb{1}_{\X \times \X \backslash E_s \times E_t}(x,y) $ and $g(x,y) = \abs[\big]{K(x,y)}p(x)p(y)$ for all $x,y \in \X$, then $\int_{\X \times \X \backslash E_s \times E_t} K(x,y)p(x)p(y) \dif x \dif y $ and $\int_{\X \times \X \backslash E_s \times E_t} \abs[\big]{K(x,y)}p(x)p(y) \dif x \dif y$ exist. Let us denote $I(s,t) = \int_{E_s \times E_t} K(x,y) p(x) p(y) \dif x \dif y$ and $I = \int_{\X \times \X} K(x,y) p(x) p(y) \dif x \dif y$. We then have,
\[
\begin{split}
\abs[\big]{I(s,t) - I}
	& = \abs*{\int_{\X \times \X \backslash E_s \times E_t} K(x,y) p(x)p(y) \dif x \dif y}
\\	& \leq \int_{\X \times \X \backslash E_s \times E_t} \abs[\big]{K(x,y)} p(x)p(y) \dif x \dif y
\\	& = \int_{\X \times \X} \abs[\big]{K(x,y)} p(x)p(y) \dif x \dif y - \int_{E_s \times E_t} \abs[\big]{K(x,y)} p(x)p(y) \dif x \dif y.
\end{split}
\]
As above, since $E_{\min(s,t)} \times E_{\min(s,t)} \subset E_s \times E_t$, Proposition~\ref{prop:comp} implies that
\[
\int_{E_{\min(s,t)} \times E_{\min(s,t)}} \abs[\big]{K(x,y)} p(x) p(y) \dif x \dif y
\]
exists and
\[
\abs[\big]{I(s,t) - I}
\leq \int_{\X \times \X} \abs[\big]{K(x,y)} p(x) p(y) \dif x \dif y - \int_{E_{\min(s,t)} \times E_{\min(s,t)}} \abs[\big]{K(x,y)} p(x) p(y) \dif x \dif y.
\]
As $\parens{E_{s} \times E_{s}}_{s \in \N}$ is an exhaustion of $\X \times \X$, $\int_{E_{\min(s,t)} \times E_{\min(s,t)}} \abs[\big]{K(x,y)} p(x) p(y) \dif x \dif y$ converges to $\int_{\X \times \X} \abs[\big]{K(x,y)} p(x) p(y) \dif x \dif y $ as $s, t$ tend to $+ \infty$. Thus, $I(s,t) = \langle \widetilde{L_s}, \widetilde{L_t} \rangle $ converges to $I = \int_{\X \times \X} K(x,y)p(x)p(y) \dif x \dif y$ as $s, t$ tend to $+\infty$.

By applying Loève criterion (Proposition~\ref{loève}) to the sequence $(\widetilde{L_s})_{s \in \mathbb{N}}$, we conclude that $(\widetilde{L_s})$ converges in $\HS$. Let $\mu_p$ denote its limit. Notice that for all $x \in \X$, we have
\[
\mu_p(x) = \lim_{s \to \infty} \int_{E_s} K(x,y) p(y) \dif y.
\]
Thus, for all $x \in \X,$ the sequence $\parens[\big]{\int_{E_s} K(x,y) p(y) \dif y}$ converges as $s$ tends to $+ \infty $ and its limit does not depend on the exhaustion $(E_s)$. This implies that $\int_{\X} K(x,y) p(y) \dif y$ is finite for all $x \in \mathcal{X}$, and by Lemma~\ref{lemma:RKHSandPointwiseConvergence}, we deduce that
\[
\mu_p = \int_{\mathcal{X}} K(\void,y) p(y) \dif y.
\]
To conclude the proof of~\eqref{thm:mean_embedding_a}, recall that $I(s,t) \to I$ as $s$ and $t$ tend to $+ \infty$. This implies that $I(s,s) \to I$ when $s$ tends to $+ \infty$. As $\widetilde{L_s} $ converges in $\HS$ to $\mu_p$, we conclude that
\[
\norm{\mu_p}^2=\int_{\X \times \X} K(x, y) p(x)p(y) \dif x \dif y.
\]
It remains to prove~\eqref{thm:mean_embedding_b}. Let $f \in \HS$. From the proof of~\eqref{thm:mean_embedding_a}, we have
\[
\int_{E_s} f(x) p(x) \dif x = \langle f, \widetilde{L_s} \rangle.
\]
As $(\widetilde{L_s})$ converges in $\HS$ to $\mu_p$ when $s$ tends to $+\infty$, we obtain that $\parens[\big]{\int_{E_s} f(x) p(x) \dif x}$ converges to $\langle f, \mu_p \rangle $ as $s$ tends to $+\infty$. As the limit does not depend on the chosen exhaustion, we conclude that $fp$ admits an improper integral on $\X$ and,
\[
\int_\X f(x)p(x) \dif x =
\langle f, \mu_p \rangle.
\]
This implies that the mapping $\finHS \mapsto \int_\X f(x)p(x) \dif x$ is continuous.
\end{proof}

\begin{remark}
Theorem~\ref{thm:mean_embedding} contains two statements. The first one, \eqref{thm:mean_embedding_a}, is that the function ${\mu_p \colon \xinX \mapsto \int_\X K(x,y) p(y) \dif y}$ exists and belongs to $\HS$. This result is the analogue for improper Riemann integrals of that established for Lebesgue integrals (see~\cite[Proposition~2.3]{barp2024targeted} and~\cite{oates2022minimum}). Under the same condition, we further demonstrate in~\eqref{thm:mean_embedding_b} that the reproducing property holds, with a less restrictive condition than in~\cite{Mean_embedding, muandet2017kernel}, as we will show in Proposition~\ref{prop:embed}. Notice that \eqref{thm:mean_embedding_b}~is not simply deduced from~\eqref{thm:mean_embedding_a} by continuity of the scalar product in $\HS$ since the interchange of limit and integral is not guaranteed in general.
\end{remark}

\begin{proposition}\label{prop:embed}
Let $\HS$ be a RKHS of real-valued function defined on $\X$, with reproducing kernel~$K$. Assume that $K$ and $p$ are continuous almost everywhere and locally bounded on $\X \times \X$. If the standard variation $\int_{\X} \sqrt{K(x, x)}p(x) \dif x$ exists, then $\iint_{\X \times \X} \abs[\big]{K(x, y)} p(x) p(y) \dif x \dif y$ is also finite.
\end{proposition}

\begin{proof}
Let $(E_n)$ be an exhaustion of $\X$. Recall that $K(x,y) = \angles[\big]{K(x, \! \void), K(y, \! \void)}_\HS$ for all $x, y \in \X$. Using Cauchy--Schwarz inequality, we then obtain:
\[
\int_{E_n \times E_n} \abs[\big]{K(x, y)} p(x) p(y) \dif x \dif y \leq \int_{E_n \times E_n} \sqrt{K(x, x)} p(x) \sqrt{K(y, y)} p(y) \dif x \dif y.
\]
Since $E_n \times E_n$ is an admissible subset of $\X \times \X$, applying Fubini's theorem for Riemann integrals (cf.~\cite[Section~11.4]{Zorich}) gives:
\[
\begin{split}
\int_{E_n \times E_n} \sqrt{K(x, x)} p(x) \sqrt{K(y, y)} p(y) \dif x \dif y
	& = \int_{E_n} \int_{E_n} \sqrt{K(x, x)} p(x) \sqrt{K(y, y)} p(y) \dif x \dif y
\\	& = \parens[\Bigg]{\int_{E_n} \sqrt{K(x, x)} p(x) \dif x}^2.
\end{split}
\]
Thus, if $\int_{\X} \sqrt{K(x, x)} p(x) \dif x$ is finite, it follows that $\iint_{\X \times \X} \abs[\big]{K(x, y)} p(x) p(y) \dif x \dif y$ is also finite.
\end{proof}

The minimal requirement for the existence of the mean embedding $\mu_p$ is that $\iint_{\X \times \X} \abs[\big]{K(x, y)} p(x) p(y) \dif x \dif y$ is finite. This is weaker than the criterion found in the literature that requires $\int_{\X} \sqrt{K(x, x)}p(x) \dif x$ to be finite, as shown by the previous proposition. Note, however, that the (almost sure) continuity of the kernel is also required in Theorem~\ref{thm:mean_embedding}, which is a mild assumption. Finally, the following example exhibits a kernel $K$ for which $\int_{\X} \sqrt{K(x, x)}p(x) \dif x$ is not finite but such that $\iint_{\X \times \X} \abs[\big]{K(x, y)} p(x) p(y) \dif x \dif y$ is finite.

\begin{example}[Kernel with non-finite standard deviation and well-defined mean embedding]
Let $p \colon x \mapsto \frac{1}{x^2}\mathbb{1}_{x \geq 1}$ and let us consider the following function:
\[
K \colon [1,\infty) \times [1,\infty) \to \mathbb{R},
\quad K(x, y) = xy e^{-\parens{x - y}^2}.
\]
It is a positive semidefinite kernel as it is the covariance function of the process $xZ(x)$ where $Z$~is a stationary process with mean zero and Gaussian covariance function $\exp\parens[\big]{-(x-y)^2}$. The standard deviation is clearly infinite:
\[
\mathbb{E}_{X \sim p}\bracks[\big]{\sqrt{K(X, X)}} = \int_1^{+\infty} \frac{1}{x} \dif x = +\infty.
\]
Let us now show that $I \coloneqq \int_{[1,\infty)^2} \abs[\big]{K(x, y)} p(x) p(y) \dif x \dif y$ exists.

As the function $x,y \mapsto K(x, y) p(x) p(y)$ is positive, it is enough to prove that for only one specific exhaustion $(E_n)$ of the set $[1,\infty)$, the sequence $I_n \coloneqq \iint_{E_n ^2} K(x, y) p(x) p(y) \dif x \dif y$ converges to $I$. Indeed, the positivity guarantees that the result will then be valid for all exhaustions~\cite[Section~11.6, Proposition~1]{Zorich}. Here, we choose $E_n = [1, n]$.

To compute $I_n$, let us split the cubic domain $E_n \times E_n$ into two triangles:
\[
I_n = \int_1^n \int_1^n \frac{1}{xy} e^{-\parens{x - y}^2} \dif x \dif y = \int_1^n \int_y^n \frac{1}{xy} e^{-\parens{x - y}^2} \dif x \dif y + \int_1^n \int_1^y \frac{1}{xy} e^{-\parens{x - y}^2} \dif x \dif y.
\]
Using the symmetry of the integrand, we can see that the two terms are equal, and thus
\[
I_n = 2 \int_1^n \int_y^n \frac{1}{xy} e^{-\parens{x - y}^2} \dif x \dif y.
\]
Now let us apply the change of variables $T \colon \xyinRtwo \mapsto (w = x - y, y)$. As for all $x, y \in \R$,
\[
1 \leq y \leq n \quad \text{and} \quad y \leq x \leq n
\qquad \iff \qquad
0 \leq w \leq n - 1 \quad \text{and} \quad 1 \leq y \leq n - w,
\]
we have
\[
I_n = 2 \int_{0}^{n-1} \int_{1}^{n-w} \frac{1}{y(w + y)} e^{-w^2} \dif y \dif w.
\]
Now, for all $w \in [0, n-1]$, we have
\[
0 \leq \int_{1}^{n-w} \frac{1}{y(w + y)} \dif y
\leq \int_{1}^{n-w} \frac{1}{y^2} \dif y = \bracks[\Bigg]{- \frac{1}{y}}_1^{n-w} = 1 - \frac{1}{n - w} \leq 1.
\]
Therefore,
\[
I_n \leq 2\int_0^{n-1} e^{-w^2} \dif w \leq 2 \int_0^{+\infty} e^{-w^2} \dif w =\sqrt{\pi}.
\]
Finally, the sequence $(I_n)_{n \geq 1}$, which is monotonically increasing, is bounded. Hence it converges as $n \to + \infty$. Therefore, $\iint_{[1, \infty)^2} \abs[\big]{K(x, y)} p(x) p(y) \dif x \dif y$ is finite, even though $\mathbb{E}_{X \sim p}\bracks[\big]{\sqrt{K(X, X)}}$ is not finite.
\end{example}

\printCOI

\printbibliography

\end{document}