
\documentclass[psamsfonts]{amsart}
\usepackage{amsmath}
\usepackage{amssymb}
%GES%\usepackage{harvard}
\usepackage{amscd}
%\usepackage{diagrams}

%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
%% Setup

\newtheorem{thm}{Theorem}[section]
\newtheorem{lem}[thm]{Lemma}
\newtheorem{prop}[thm]{Proposition}
\newtheorem{cor}[thm]{Corollary}
\newtheorem{defn}{Definition}[section]
\newtheorem{example}{Example}[section]
\newtheorem{exercise}{Exercise}[section]
\newtheorem{algorithm}{Algorithm}[section]


\newcommand\be{\begin{equation}}
\newcommand\ee{\end{equation}}
\newcommand\half{\frac{1}{2}}
\newcommand\fhat{\widehat{f}}
\renewcommand\Im{\operatorname{Im}}
\newcommand\phihat{\widehat{\phi}}
\newcommand\CC{{\mathbb C}}
\newcommand\PP{{\mathbb P}}
\newcommand\RR{{\mathbb R}}
\newcommand\ZZ{{\mathbb Z}}

\numberwithin{equation}{section}

%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%



\title[On Spherical Averages of Radial Basis Functions]
{On Spherical Averages of Radial Basis Functions}
\author[B. J. C. Baxter]{B. J. C. Baxter}

\address{School of Economics, Mathematics and Statistics,\\
Birkbeck College, University of London,\\
Malet Street, London WC1E 7HX, England}

\subjclass{Primary: 41A30, 43A70; Secondary: 43A35}
 \keywords{Radial Basis Functions, Spherical Average, Haar Measure, Paley-Wiener}

\email{b.baxter@bbk.ac.uk}

\dedicatory{Dedicated to Arieh Iserles on the occasion of his $60^{th}$
  birthday.}


\begin{document}

\begin{abstract}
A radial basis function (RBF) has the general form
\[
s(x) = \sum_{k=1}^n a_k \phi(x - b_k), \qquad x \in \RR^d,
%\label{abs_sa1}
\]
where the coefficients $a_1, \ldots, a_n$ are real numbers, the
points, or centres,
$b_1, \ldots, b_n$ lie in $\RR^d$, and $\phi : \RR^d \to \RR$ is a
radially symmetric function. Such approximants are highly useful and
enjoy rich theoretical properties; see, for instance,
Buhmann \cite{Buhmann}, 
Fasshauer \cite{Fasshauer}, Light and Cheney \cite{WAL} or Wendland \cite{Wendland}. 
The important special case of {\em polyharmonic splines} results when
$\phi$ is the fundamental solution of the iterated Laplacian operator,
and this class includes the Euclidean norm $\phi(x) = \|x\|$ when $d$
is an odd positive integer, the
thin plate spline $\phi(x) = \|x\|^2 \log \|x\|$ when $d$ is an even
positive integer, and univariate splines. Now B-splines generate a
compactly supported basis for univariate spline spaces, but an
analyticity argument implies that a nontrivial
polyharmonic spline generated by (\ref{sa1}) cannot be compactly
supported when $d > 1$. However, a pioneering paper of Jackson \cite{Jackson}
established that the {\em spherical average} of
a radial basis function generated by the Euclidean norm 
can be compactly supported when the centres and
coefficients satisfy certain moment conditions; Jackson then used
this compactly supported spherical average to construct approximate
identities, with which he was then able to derive some of the earliest
uniform convergence results for a class of radial basis functions. 
Our work extends this
earlier analysis, but our technique is entirely novel, and applies to
all polyharmonic splines. Furthermore, we observe that the technique
provides yet another way to generate compactly supported, radially symmetric,
positive definite functions. Specifically, we find
that the spherical averaging operator commutes with the Fourier
transform operator, and we are then able to identify Fourier
transforms of compactly supported functions using the Paley--Wiener
theorem. Furthermore, the use of Haar measure on compact Lie groups would
not have occurred without frequent exposure to Arieh Iserles' study of
geometric integration.
\end{abstract}

\maketitle




\section{Introduction}
A paper on radial basis functions (RBFs)
might initially strike the reader as being
out of place in this {\em Festschrift}, 
since Arieh Iserles has not worked on RBFs directly. However, his influence on
generations of Cambridge numerical analysis researchers has been
enormous. In particular, the RBF research of Michael Powell's group
has been greatly enhanced by Arieh's breadth of knowledge and
encouragement.
In this note, I demonstrate that the Fourier transform of
a RBF, viewed as a tempered distribution, possesses a certain
natural form, which results from the fundamental fact that the
spherical averaging operator commutes with the Fourier transform
operator. This fact, in itself a nontrivial result, will be established 
using an alternative definition of spherical averaging
obtained via Haar measure on the orthogonal group. We then apply the 
Paley-Wiener theorem, which characterizes the Fourier transforms of
compactly supported distributions as entire functions of exponential
type, from which certain moment conditions emerge quite
naturally. These topics are particularly relevant to this {\em Festschrift},
since it is very much Arieh's influence on my mathematical
development which is here evident, from the theory of Lie groups to
classical complex analysis.


A radial basis function has the general form
\be
s(x) = \sum_{k=1}^n a_k \phi(x - b_k), \qquad x \in \RR^d,
\label{sa1}
\ee
where the coefficients $a_1, \ldots, a_n$ are real numbers, the
points, or centres,
$b_1, \ldots, b_n$ lie in $\RR^d$, and $\phi : \RR^d \to \RR$ is a
radially symmetric function. Such approximants are highly useful and
enjoy rich theoretical properties; see, for instance,
Buhmann \cite{Buhmann} or Light and Cheney \cite{WAL}. The approximation theory community began to
study RBFs during the mid-1980s, following the seminal work of
Micchelli \cite{Micchelli}. In the course
of his fundamental work on the convergence properties of RBFs,
Jackson \cite{Jackson} discovered the following remarkable fact: if
$\phi(x) = \|x\|$, the Euclidean norm, then the {\em spherical average} of
(\ref{sa1}) is compactly supported when the dimension $d$ is odd and
the coefficients and centres satisfy the relations
\be
\sum_{k=1}^n a_k \|b_k\|^{2\ell} = 0, \qquad \ell = 0, 1, \ldots,
(d-1)/2,
\label{sa2}
\ee
the spherical average $As$ being defined by
\be
As(x) = \int_{S^{d-1}} s(\|x\|\theta)\,d\mu_d(\theta), \qquad x \in
\RR^d,
\label{sa3}
\ee
where $\mu_d$ denotes normalized $(d-1)$-dimensional Lebesgue measure
on the unit sphere $S^{d-1}$ in $\RR^d$.
Jackson's method was to expand the integrand of (\ref{sa3}) for large
$\|x\|$ and, in a {\em tour de force} of classical analysis, to
identify the result as a certain hypergeometric function, from which
he was then able to deduce the compact support of $As$ when relations
(\ref{sa2}) hold. Jackson \cite{Jackson} was motivated by the construction of
approximate identities using compactly supported spherical averages of
radial basis functions; an ingenious construction, although it is not
our primary interest in this paper.
In contrast, the new technique presented here 
generalizes to {\em any} function $\phi :
\RR^2 \to \RR$ whose (distributional) Fourier transform is of the form
$\phihat(\xi) = C \|\xi\|^{-2m}$, for some positive integer
$m$. The class of all such functions is called the {\em polyharmonic
splines}, following Madych and Nelson \cite{MadychNelson}, for any such function is the fundamental solution of an
iterated Laplacian operator. In particular, our analysis applies to
the important case of {\em thin plate splines}, for which $\phi(x) = \|x\|^2 \log_e
\|x\|$, for $x \in \RR^2$. We also remark that a minor modification of
our technique provides another way to construct compactly supported,
radially symmetric, positive definite functions
(cf.~\cite{Wendland} for further details of such constructions).



\section{Spherical averaging via the Fourier transform}

Let $f : \RR^d \to \RR$ be any continuous function. Its
{\em spherical average} $Af : \RR^d \to \RR$ is usually defined by the
relation
\be
Af(x) = \int_{S^{d-1}} f(\|x\| \theta)\, d\mu_d(\theta), \qquad x \in
\RR^d,
\label{saf1}
\ee
where $\mu_d$ denotes normalized Haar probability measure on the unit
sphere $S^{d-1} = \{x \in \RR^d : \|x\| = 1\}$; in other words,
$\mu_d$ denotes ordinary $(d-1)$-dimensional Lebesgue measure scaled
by the $(d-1)$-dimensional measure of the unit sphere $S^{d-1}$. 
Spherical averages have long had important applications in the 
theory of partial differential equations, as exemplified by the 
elegant monograph of John \cite{John}.
However, our
calculations will be greatly simplified by use of the following
equivalent definition.

\begin{defn}
Let $f : \RR^d \to \RR$ be any continuous function. The {\em spherical
average} $Af : \RR^d \to \RR$ can also be defined by the equation
\be
Af(x) = \int_{O(d)} f(Ux)\,d\sigma_d(U), \qquad x \in \RR^d,
\label{saf2}
\ee
where $O(d)$ denotes the orthogonal group, that is,
\[
O(d) = \{ V \in \RR^{d \times d} : V^T V = I\},
\]
and $\sigma_d$ denotes the normalized Haar probability measure on $O(d)$.
\end{defn}

\noindent
Haar measure $\sigma_d$ on the compact metric group $O(d)$ is 
the unique measure possessing the invariance property
\be
\sigma_d(Y) = \sigma_d(QY) = \sigma_d(YQ), 
\label{saf2a}\ee
for any Borel subset $Y \subset O(d)$ and any $Q \in O(d)$, and
further properties, together with an illuminating rigorous construction,
may be found in the early chapters of Milman and Schechtman \cite{MilmanSchechtman}.
It may be easily checked that (\ref{saf2a}) implies the equivalence of
(\ref{saf1}) and (\ref{saf2}). 

Now Haar measure is,
perhaps, somewhat unfamiliar to our audience, let us mention some salient
facts for the convenience of the reader. We shall say that $M \in
\RR^{n \times n}$ is a {\em Gaussian random matrix} if its elements
are independent Gaussian random variables with mean zero and unit
variance; see, for instance, Baxter and Iserles \cite{baxterai} or Edelman and Raj Rao \cite{edelmanrao}. 
If we calculate the QR factorization of this Gaussian random matrix
$M = QR$, where $Q \in \RR^{n \times n}$ is orthogonal and $R \in
\RR^{n \times n}$ is upper triangular with positive diagonal entries,
then $Q$ is a (non-Gaussian) random matrix that is uniformly generated
with respect to Haar measure on $O(d)$. Thus, generating $N$ such
orthogonal matrices $Q_1, \ldots, Q_N$ in this way, we can
approximately integrate any continuous function 
$f : O(d) \to \RR$ via the Monte
Carlo sum
\[
\int_{O(d)} f(Q) \sigma_d(Q) 
\approx
N^{-1} \sum_{k=1}^N f(Q_k),
\]
and the integral over the orthogonal group is the limit of this sum as 
$N \to \infty$.


All of the radial basis functions currently studied are continuous
functions of at most polynomial growth, and their analysis makes great use of
the Schwartz theory of tempered distributions. Our notation is fairly
standard and follows that of Friedlander and Joshi \cite{Friedlander} and
Rudin \cite{Rudin}. 
In particular,
we let $S(\RR^d)$ denote the vector space of infinitely differentiable
real-valued functions whose every derivative has supra-algebraic
decay. This vector space becomes a locally convex topological vector
space in the usual way, its dual $S^\prime(\RR^d)$ being the vector
space of {\em tempered distributions}. In particular, every continuous
function $\psi : \RR^d \to \RR$ of
polynomial growth is a tempered distribution, and its action on
$S(\RR^d)$ is given by the integral relation
\be
\langle \psi, f \rangle
=
\int_{\RR^d} \psi(x) f(x)\,dx, \qquad f \in S(\RR^d).
\ee
Further, the Fourier transform $\fhat$ of $f \in S(\RR^d)$ is defined
by
\[
\fhat(\xi) = \int_{\RR^d} f(x) \exp(-i\xi^T x)\,dx, \qquad \xi \in
\RR^d,
\]
and the Fourier transform operator $F : f \mapsto \fhat$ defines a
linear bijection $F : S(\RR^d) \to S(\RR^d)$. The inverse Fourier
transform is then given by the integral
\[
f(x) = (2\pi)^{-d} \int_{\RR^d} \fhat(\xi) \exp(i\xi^T x)\,d\xi,
\qquad x \in \RR^d,
\]
and the Fourier transform is defined on the dual space
$S^\prime(\RR^d)$ by the requirement that
\[
\langle F \psi, f\rangle = \langle \psi, F f \rangle,
\]
for any $f \in S(\RR^d)$ and $\psi \in S^\prime(\RR^d)$.
We refer the reader to Rudin \cite{Rudin} for further exposition.

\begin{thm}
The spherical averaging operator $A$ and the Fourier transform
operator $F$ commute when applied to elements of the space $S(\RR^d)$,
that is we have the commutative diagram
%\begin{diagram}[width=5em]
%S(\RR^d) & \rTo^A & S_R(\RR^d)\\
%\dTo^F &      & \dTo_F\\
%S(\RR^d) & \rTo^A & S_R,\\
%\end{diagram}
\[
\begin{CD}
S(\RR^d) @>A>> S_R(\RR^d)\\
@VFVV @VFVV\\
S(\RR^d) @>A>> S_R(\RR^d)\\
\end{CD}
\]
where $S_R(\RR^d)$ denotes the subspace of radially
symmetric functions in $S(\RR^d)$.
\label{thm2.2}
\end{thm}

\begin{proof}
Applying Fubini's theorem yields the equations
\begin{eqnarray}
 \widehat {Af}(\xi) 
 &=& \int_{\RR^d} \Bigl( \int_{O(d)} f(Ux) \,d\sigma_d(U) \Bigr)
\exp(-i\xi^Tx)\,dx \nonumber\\
 &=& \int_{O(d)} \Bigl( \int_{\RR^d} f(Ux) \exp(-i\xi^Tx)\,dx
\Bigr)\,d\sigma_d(U) \nonumber\\
 &=& \int_{O(d)} \widehat{f}(U\xi)\,d\sigma_d(U) \nonumber\\
 &=& A\widehat{f}(\xi). 
\end{eqnarray}
Thus $F(Af) = A(Ff)$, for any $f \in S(\RR^d)$. Finally, the
well-known fact that the Fourier transform of any radially symmetric
function is itself radially symmetric implies that $F$ maps
$S_R(\RR^d)$ into itself.

\end{proof}

The tempered distributional definition of the spherical averaging operator is defined 
via its action on $S(\RR^d)$, that is
\[
\langle A\psi, f \rangle := \langle \psi, Af \rangle, \qquad f \in
S(\RR^d), \psi \in S^\prime(\RR^d).
\]

\noindent
Now the Fourier transform is also defined on $S^\prime(\RR^d)$ via its
action on $S(\RR^d)$. More precisely, we have
\[
\langle F\psi, g \rangle = \langle \psi, Fg\rangle, 
\qquad g \in S(\RR^d),
\]
for any $\psi \in S^\prime(\RR^d)$. We use the same trick to extend
the definition of our spherical averaging operator $A$ to
$S^\prime(\RR^d)$, that is
\[
\langle A\psi, g \rangle = \langle \psi, Ag\rangle, 
\qquad g \in S(\RR^d).
\]

\begin{thm}
The spherical averaging operator $A$ and the Fourier transform
operator $F$ commute when applied to tempered distributions,
that is, we obtain the commutative diagram
%\begin{diagram}[width=5em]
%S^\prime(\RR^d) & \rTo^A & S^\prime_R(\RR^d)\\
%\dTo^F &      & \dTo_F\\
%S^\prime(\RR^d) & \rTo^A & S^\prime_R(\RR^d),\\
%\end{diagram}
\[
\begin{CD}
S^\prime(\RR^d) @>A>> S_R^\prime(\RR^d)\\
@VFVV @VFVV\\
S^\prime(\RR^d) @>A>> S_R^\prime(\RR^d)\\
\end{CD}
\]

where $S^\prime_R(\RR^d)$ denotes the subspace of
rotation-invariant tempered distributions.
\end{thm}

\begin{proof}
This is merely diagram chasing:
\[
\langle AF\psi, g \rangle
= \langle F\psi, Ag \rangle
= \langle \psi, F(Ag) \rangle
= \langle \psi, A(Fg) \rangle
= \langle A \psi, Fg \rangle
= \langle F(A\psi), g\rangle.
\]
\end{proof}

\section{Spherically averaging radial basis functions}

\begin{thm}
Let $\phi : \RR^d \to \RR$ be a radially symmetric continuous function
of polynomial growth and define
\[
s(x) = \sum_{k=1}^n a_k \phi(x-b_k), \qquad x \in \RR^d.
\]
Then
\be
\widehat{As}(\xi) = \phihat(\xi) \sum_{k=1}^n a_k \Omega_d(\|b_k\|
\|\xi\|), \qquad \xi \in \RR^d,
\label{sarbf_1}
\ee
where $\Omega_d : \RR \to \RR$ is defined by
\be
\Omega_d(t) = \widehat{\mu_d}(t u), \qquad t \in \RR,
\label{3.2}
\ee
for any fixed unit vector $u \in \RR^d$. Thus
$\Omega_d$ is essentially the Fourier transform of the Haar
probability measure on the sphere.
\end{thm}

\begin{proof}
Since $\phi$ is a tempered distribution, we deduce
\[
\widehat{s}(\xi) = \phihat(\xi) \sum_{k=1}^n a_k \exp(-ib_k^T \xi), \qquad
\xi \in \RR^d,
\]
and we recall that $\phihat$ is a radially symmetric
function. Now, the spherical average of the exponential
\[
E_b(\xi) := \exp(ib^T\xi), \qquad \xi \in \RR^d,
\]
is the radially symmetric function defined by the integral
\[
AE_b(\xi) = \int_{S^{d-1}} \exp(i\|\xi\|b^T\theta) d\mu_d(\theta).
\]
Since we may rotate the coordinate system without modifying the value
of this integral, we deduce
\[
AE_b(\xi) = \int_{S^{d-1}} \exp(i\|\xi\| \|b\| u^T\theta) d\mu_d(\theta),
\]
where $u$ can be any unit vector.  Thus we have
\[
AE_b(\xi) = \Omega_d(\|b\| \|\xi\|), \qquad \xi \in \RR^d,
\]
and the spherical average is given by
\[
\widehat{As}(\xi) = \phihat(\xi) \sum_{k=1}^n a_k \Omega_d(\|b_k\| \|\xi\|).
\]
\end{proof}

We shall need the Paley-Wiener theorem in the form given by Rudin \cite{Rudin}
and provide some background material for the reader.


\begin{defn}
A continuous function $f \colon \CC^d \to
\CC$ is said to be an {\it entire function (of several complex variables)}
if the maps 
\[
\{z \mapsto f(a_1, \ldots, a_{j-1}, z, a_{j+1}, \ldots,
a_d) : z \in \CC\}
\]
are entire functions (of one complex variable) for
any complex numbers $a_1, \ldots, a_d$ and $1 \le j \le d$.
\end{defn}

\begin{thm} [Paley--Wiener] If $\mu$ is a signed measure on $\RR^d$ supported
by $\{x\in\RR^d:\|x\| \le r\}$, then its Fourier transform $\widehat\mu$ is
an entire function of exponential type $r$, that is
\[ 
\widehat\mu(z) \le C \exp(r|\Im z|), \qquad z \in \CC^d, 
\]
for some constant $C$. Every entire function of exponential type $r$
arises in this way.
\end{thm}

\begin{proof}
This is Theorem 7.22 of Rudin \cite{Rudin}.
\end{proof}

\noindent
The definition is a special case of Rudin \cite{Rudin}, Definition 7.20, 
and
\[
 \left|\Im z \right| := \left( (\Im z_1)^2 + \cdots + (\Im z_d)^2 \right)^{1/2}, \qquad z =
(z_1, \ldots, z_d) \in \CC^d.
\]

We have used the fact that a compactly supported distribution $\psi$ of
order zero can be 
identified with a signed measure. Indeed, we have the inequality
$|\psi(f)| \le C \|f\|_\infty$ for every continuous function $f \colon K
\to \RR$, where $K$ denotes the compact set supporting $\psi$. Thus $\psi$
is a continuous linear functional on $(C(K), \|\cdot\|_\infty)$,
and the Riesz representation theorem allows us to identify $\psi$
with a measure $\mu$ via the formula
\[
\psi(f) = \int_K f(x) \,d\mu(x), \qquad f \in C(K). 
\]

We observe that $\Omega_d$ is therefore an entire function of
exponential type $1$, because of (\ref{3.2}). Further, since the
Fourier transform of a rotation-invariant measure is radially
symmetric, we deduce that $\Omega_d$ is an even function. Hence
its Taylor series takes the form
\be
\Omega_d(z) = \sum_{\ell=0}^\infty c^{(d)}_\ell z^{2\ell}, \qquad z
\in \CC.
\ee
Further, it can be shown that
\[
\Omega_d(t) = \Gamma(d/2) (2/t)^{(d-2)/2} J_{(d-2)/2}(t);
\]
see equation (1.8) of Schoenberg \cite{IJS38} or 
Section 41 of Donoghue \cite{Donoghue}. Since not all readers will be
fans of Bessel functions, let us mention that using spherical polar
coordinates provides the alternative formula
for $\Omega_d$ is 
\[
\Omega_d(t) = \frac{\widehat{f_d}(t)}{\widehat{f_d}(0)}, \qquad t \in
\RR^d,
\]
where $f_d : \RR \to \RR$ is the compactly supported univariate
function given by
\[
f_d(x) =
\begin{cases}
(1-x^2)^{(d-3)/2} & \text{for $|x| \le 1$,}\\
0 & \text{otherwise.}
\end{cases}
\]

\begin{thm}
Let $\phi : \RR^d \to \RR$ be a polyharmonic spline, that is, a
tempered distribution whose Fourier transform takes the form
\be
\phihat(\xi) = C \|\xi\|^{-2m}, \qquad \xi \in \RR^d \setminus
\{0\},
\ee
for some positive integer $m$. Then the spherical average $As$ of
\be
s(x) = \sum_{k=1}^n a_k \phi(x - b_k), \qquad x \in \RR^d.
\label{tps1}
\ee
is compactly supported if and only if
\be
\sum_{k=1}^n a_k \|b_k\|^{2\ell} = 0, \qquad\hbox{ for } \ell = 0, 1,
\ldots, %(d-1)/2.
m-1.
\label{tps2}
\ee
\end{thm}

\begin{proof}
We have
\begin{eqnarray}
\widehat{As}(\xi) &=& C\|\xi\|^{-2m} \sum_{k=1}^n a_k \Omega_d(\|b_k\|
\|\xi\|)\nonumber\\
&=& C\|\xi\|^{-2m} \sum_{k=1}^n a_k \sum_{\ell=0}^\infty c^{(d)}_\ell
\|b_k\|^{2\ell} \|\xi\|^{2\ell}\nonumber\\
&=& C \|\xi\|^{-2m} \sum_{\ell=0}^\infty c^{(d)}_\ell \|\xi\|^{2\ell}
\Bigl(\sum_{k=1}^n a_k \|b_k\|^{2\ell}\Bigr)\nonumber\\
&=:& C \|\xi\|^{-2m} g(\xi).
\end{eqnarray}
Thus $\widehat{As}$ is an entire function of exponential type $1$ if and only
if the entire function $g$ possesses a zero at the origin of order
$2m$, so cancelling the pole of order $2m$. Therefore we deduce that
$\widehat{As}$ is an entire function of exponential type, and hence $As$
is compactly supported, if and only if the first $2m$
coefficients of the Taylor series of $g$ vanish, that is,
\[
\sum_{k=1}^n a_k \|b_k\|^{2\ell} = 0, \qquad \ell = 0, 1, \ldots,
m-1.
\]
\end{proof}

Finally, we observe that an easy modification of the above
construction yields another way to construct compactly supported
positive definite functions. Indeed, if we define $s : \RR^d \to \RR$
by \eqref{tps1} and \eqref{tps2}, then the convolution of its
spherical average with itself, that is,
\[
\psi(x) = (As)\ast(As)(x), \qquad \hbox{ for } x \in \RR^d,
\]
is a compactly supported radially symmetric function. Further, it is
also a positive definite function, since its Fourier transform
satisfies $\widehat{\psi}(\xi) = |\widehat{\phi}(\xi)|^2$, for all
$\xi \in \RR^d$. Furthermore, if we add the extra moment condition
\be
\sum_{k=1}^n a_k \|b_k\|^{2m} \ne 0, 
\label{tps3}
\ee
then we deduce that $\widehat{\psi}(0) > 0$, which implies that it is
a strictly positive definite function.

\section{Applications}

We shall now address the important special cases of the Euclidean norm
$\phi(x) = \|x\|$, for $x \in \RR^d$ and $d$ odd, and the thin plate
spline $\phi(x) = \|x\|^2 \log_e \|x\|$, for $x \in \RR^d$ and $d$
even, for which we first establish that these are, indeed,
polyharmonic radial basis functions. 
It is usual to refer the reader to standard texts such as
Jones \cite{Jones} for the explicit formulae giving the Fourier
transforms of these tempered distributions. However, we prefer to
sketch a simple derivation based on the Schoenberg theory of positive
definite functions for completeness.
Further details are given in the excellent textbooks
Fasshauer \cite{Fasshauer} and Wendland \cite{Wendland}. 

The
following integrals occur in almost all calculations of this form.


\begin{lem}
Define
\be
I(\lambda, \mu) = \int_0^\infty e^{-\lambda/t} t^{-\mu}\,dt,
\qquad\hbox{ for } \lambda > 0, \mu > 1.
\label{s1}
\ee
Then
\be
I(\lambda, \mu) = \frac{\Gamma(\mu - 1)}{\lambda^{\mu - 1}}.
\label{s2}
\ee
\end{lem}

\begin{proof}
We simply set $\tau=\lambda/t$; the convergence of the integrals is elementary.
\end{proof}

\begin{lem} We have
\be
\sum_{j,k=1}^n y_j y_k e^{-t\|x_j-x_k\|^2}
=
(2\pi)^{-d} \int_{\RR^d} \Bigl| \sum_{k=1}^n y_k e^{ix_k^T\xi}\Bigr|^2
(\pi/t)^{d/2} e^{-\|\xi\|^2/4t}\,dt.
\label{s3}
\ee
\end{lem}

\begin{proof}
Simply apply the Fourier inversion theorem.
\end{proof}

\subsection{The Euclidean norm}

The Euclidean norm is given by the dimension-in\-de\-pen\-dent formula
\be
\phi(x) = \int_0^\infty \Bigl(1 - e^{-t\|x\|^2}\Bigr) (4\pi)^{-1/2}
t^{-3/2}\,dt.
\label{e1}
\ee
We obtain this expression by setting $\tau=t\|x\|^2$ in the Gamma function
integral
\[
\Gamma(-1/2) = \int_0^\infty \Bigl(e^{-t}-1\Bigr)t^{-3/2}\,dt,
\]
which is an analytic continuation of the usual integral relation; see,
for instance, Whittaker and Watson \cite{WW}, Section 12.21.

If $y_1, \ldots, y_n \in \RR$ satisfy $\sum_{j=1}^n y_j = 0$, then
(\ref{e1}) implies the relation
\be
\sum_{j,k=1}^n y_j y_k \|x_j - x_k\|
=
-\int_0^\infty \Bigl(\sum_{j,k=1}^n y_j y_k e^{-t\|x_j-x_k\|^2}\Bigr)
(4\pi)^{-1/2}t^{-3/2}\,dt. 
\label{e2}
\ee
Applying the Fourier inversion theorem to the Gaussian quadratic form
and swapping the order of integration (which is valid by Fubini's
theorem), we obtain
\be
\sum_{j,k=1}^n y_j y_k \|x_j - x_k\|
=
(2\pi)^{-d} \int_{\RR^d} \Bigl| \sum_{k=1}^n y_k e^{ix_k^T\xi}\Bigr|^2
\psi_d(\xi)\,d\xi,
\label{e3}
\ee
where
\begin{eqnarray}
\psi_d(\xi)
&=& -(4\pi)^{-1/2}\int_0^\infty e^{-\|\xi\|^2/4t} (\pi/t)^{d/2} t^{-3/2} \,dt\nonumber\\
&=& -(4\pi)^{-1/2} \pi^{d/2} I(\|\xi\|^2/4, (d+3)/2).
\label{e4}
\end{eqnarray}
Substituting (\ref{s2}) in (\ref{e4}) yields
\be
\psi_d(\xi) = -\Bigl(\frac{2^d \pi^{(d-1)/2}
\Gamma(\frac{d+1}{2})}{\|\xi\|^{d+1}}\Bigr).
\label{e5}
\ee
All of this is entirely classical, which is useful when explaining
Fourier transform arguments to audiences suspicious of distribution
theory\begin{footnote}{The integral formula (\ref{phihat1}) is also derived in
Baxter \cite{Bax94} in a more general context.}\end{footnote}.
Of course, (\ref{e5}) is the Fourier transform of the
Euclidean norm in $d$-dimensional space, that is,
\be
\phihat(\xi) = 
-\Bigl(\frac{2^d \pi^{(d-1)/2}
\Gamma(\frac{d+1}{2})}{\|\xi\|^{d+1}}\Bigr).
\label{phihat1}
\ee
We observe that $\phihat$ extends to an analytic function on
$\CC^d\setminus\{0\}$ when $d$ is odd, and $\phihat$ would therefore be an
exponential function of exponential type were it not for its pole at
the origin, and this is the crux of our analysis. Hence Theorem 3.3 
implies that, when $d$ is odd, the spherical average of
\[
s(x) = \sum_{k=1}^n a_k \|x-b_k\|, \qquad x \in \RR^d,
\]
is compactly supported if and only if
\[
\sum_{k=1}^n a_k \|b_k\|^{2\ell} = 0, \qquad 
\hbox{ for } \ell = 0, 1, \ldots, (d-1)/2.
\]
Jackson's construction had, as its end, the construction of an
approximate identity $As$, i.e. a spherical average that was compactly
supported with nonzero integral. The last condition therefore requires
the further moment condition
\[
\sum_{k=1}^n a_k \|b_k\|^{d+1} \ne 0.
\]


\subsection{The thin plate spline}

We shall find it convenient to 
use the slightly nonstandard definition $\phi(x)=\|x\|^2 \log
\|x\|^2 $ for the thin plate spline. However, 
since $\phi(x) = 2 \|x\|^2 \log \|x\|$, there should be no confusion. 
The integral representation is
\be
\phi(x)
=
\|x\|^2 - 1
+
\int_0^\infty
\Bigl( e^{-t\|x\|^2} - e^{-t} + t(\|x\|^2 - 1) e^{-t}\Bigr)
t^{-2}\,dt.
\label{t1}
\ee
We obtain (\ref{t1}) by setting $f(t) = t \log t$, using the formula 
\[
f^{\prime\prime}(t)
= \frac{1}{t}
= \int_0^\infty e^{-\alpha t}\,d\alpha
\]
and integrating twice. Then $\phi(x) = f(\|x\|^2)$.


If $y_1, \ldots, y_n \in \RR$ and $x_1, \ldots, x_n \in \RR^d$ satisfy 
\[
\sum_{j=1}^n y_j = 0
\quad\hbox{ and }\quad
\sum_{j=1}^n y_j x_j = 0,
\]
then (\ref{t1}) implies the quadratic form relation
\be
\sum_{j,k=1}^n y_j y_k \phi(x_j - x_k)
=
\int_0^\infty \Bigl(\sum_{j,k=1}^n y_j y_k e^{-t\|x_j - x_k\|^2}\Bigr)
t^{-2}\,dt.
\label{t2}
\ee
Just as before, we obtain
\be
\sum_{j,k=1}^n y_j y_k \phi(x_j - x_k)
=
(2\pi)^{-d} \int_{\RR^d} \Bigl|\sum_{k=1}^n y_k e^{ix_k^T\xi}\Bigr|^2
\psi_d(\xi)\,d\xi,
\label{t3}
\ee
where
\be
\psi_d(\xi)
=
\int_0^\infty (\pi/t)^{d/2} e^{-\|\xi\|^2/4t} t^{-2}\,dt
= \pi^{d/2} I(\|\xi\|^2/4, 2 + d/2).
\label{t4}
\ee
Thus
\be
\psi_d(\xi)
= \frac{2^{d+2} \pi^{d/2} \Gamma(1 + d/2)}{\|\xi\|^{d+2}}.
\label{t5}
\ee
As for the Euclidean norm, this is the Fourier transform of $\phi$ in $d$-dimensional space.
For example
\be
\psi_2(\xi) = \frac{16\pi}{\|\xi\|^4}.
\ee
We observe that, when the dimension $d$ is even, $\phihat$ would again
be an entire function of exponential type were it not for the pole at
the origin. Hence Theorem 3.3 implies that, for even dimension $d$, 
\[
s(x) = \sum_{k=1}^n a_k \|x-b_k\|^2 \log_e \|x-b_k\|, \qquad x \in
\RR^d,
\]
has a compactly supported spherical average if and only if the moment
conditions
\[
\sum_{k=1}^n a_k \|b_k\|^{2\ell} = 0
\]
are satisfied for $\ell \le d/2$.


\section*{Acknowledgments}
I am particularly grateful to the anonymous referees for their comments
and suggestions.


\begin{thebibliography}{99}

\bibitem{Bax94}
B.~J.~C.~Baxter (1994), ``Norm estimates for inverses of Toeplitz
distance matrices'',
{\em J. Approx. Theory\/}\ {\bf 79}:222--242. 

\bibitem{baxterai}
B.~J.~C.~Baxter and A.~Iserles (2003), 
``On the foundations of computational mathematics'', 
in Handbook of Numerical Analysis XI (P.G. Ciarlet and F. Cucker, eds), 
North-Holland, Amsterdam, 3--34.

\bibitem{Buhmann}
M.~D.~Buhmann (2003), {\em Radial Basis Functions: Theory and
Implementations}, Cambridge Monographs on Applied and Computational
Mathematics, Cambridge University Press.

\bibitem{Donoghue}
W.~F.~Donoghue, Jr., {\em Distributions and Fourier Transforms}, Academic Press, New York.

\bibitem{edelmanrao}
A.~ Edelman and N.~Raj Rao (2005),
``Random matrix theory'',
{\em Acta Numerica}: {\bf 14}: 233--297.

\bibitem{Fasshauer}
G.~Fasshauer (2007), {\em Meshfree Approximation Methods with Matlab},
World Scientific Publishing.

\bibitem{Friedlander}
F.~G.~Friedlander and M.~Joshi (1999), {\em Introduction to the Theory
of Distributions}, Cambridge University Press.

\bibitem{Jackson}
I.~R.~H.~Jackson (1988), ``Convergence Properties of Radial Basis
Functions'', {\em Constr. Approx.\/}\ {\bf 4}: 243--264.

\bibitem{John}
F.~John (1955), {\em Plane waves and spherical means applied to partial differential equations}, 
Interscience, New York.

\bibitem{Jones}
D.~S.~Jones (1982), {\em The Theory of Generalised Functions}, 
Cambridge University Press.

\bibitem{WAL}
W.~A.~Light and E.~W.~Cheney (2000), {\em A Course in Approximation
Theory}, Brooks/Cole.

\bibitem{MadychNelson}
W.~R.~Madych and S.~A.~Nelson (1990),
``Polyharmonic cardinal splines'', {\em J. Approx. Theory\/}\ {\bf 60}: 141--156.

\bibitem{Micchelli}
C.~A.~Micchelli (1986), ``Interpolation of scattered data:
distance matrices and conditionally positive functions'', {\em Constr.
Approx.\/} {\bf 2}: 11-22.

\bibitem{MilmanSchechtman}
V.~Milman and G.~Schechtman (1986), 
{\em Asymptotic theory of finite dimensional normed spaces},
Lecture Notes in Mathematics 1200, Springer.

\bibitem{Poggio}
T.~Poggio and S.~Smale (2003), ``The Mathematics of Learning: Dealing
with Data'', Notices of the AMS, May 2003: 537--544.

\bibitem{MJDP}
M.~J.~D.~Powell (1992), ``The Theory of Radial Basis Functions in
1990'', in {\em Advances in Numerical Analysis}, vol. II,
ed. W.~A.~Light, Oxford University Press, pp. 105--210. 

\bibitem{Rudin}
W.~Rudin (1991), {\em Functional Analysis}, McGraw-Hill.

\bibitem{IJS38}
I.~J.~Schoenberg (1938), ``Metric spaces and completely
monotone functions'',  {\em Ann. of Math.\/} {\bf 39}: 811-841.

\bibitem{Wendland}
H.~Wendland (2004), {\em Scattered Data Approximation}, 
Cambridge University Press.


\bibitem{WW}
E.~T.~Whittaker and G.~N.~Watson (1927), {\em A Course of Modern Analysis}, 
Cambridge University Press.

\bibitem{Widder}
D.~V.~Widder (1946), {\em The Laplace Transform}, Princeton University Press.
       
\end{thebibliography}

\end{document}
