Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
43 changes: 37 additions & 6 deletions paper.bib
Original file line number Diff line number Diff line change
Expand Up @@ -30,13 +30,48 @@ @book{golyandina_singular_2020
keywords = {time series, forecasting, singular value decomposition, Multivariate Singular Spectrum Analysis, signal extraction ., signal processing},
}

@article{golyandina_particularities_2020,
title = {Particularities and commonalities of singular spectrum analysis as a method of time series analysis and signal processing},
volume = {12},
copyright = {© 2020 Wiley Periodicals, Inc.},
issn = {1939-0068},
url = {https://onlinelibrary.wiley.com/doi/abs/10.1002/wics.1487},
doi = {10.1002/wics.1487},
language = {en},
number = {4},
urldate = {2025-10-27},
journal = {WIREs Computational Statistics},
author = {Golyandina, Nina},
year = {2020},
keywords = {decomposition, forecasting, signal processing, singular spectrum analysis, time series},
pages = {e1487},
}


@article{hassani_singular_2007,
title = {Singular {Spectrum} {Analysis}: {Methodology} and {Comparison}},
volume = {5},
issn = {1680-743X, 1683-8602},
shorttitle = {Singular {Spectrum} {Analysis}},
url = {https://jds-online.org/journal/JDS/article/1027},
doi = {10.6339/JDS.2007.05(2).396},
language = {en},
number = {2},
urldate = {2025-10-27},
journal = {Journal of Data Science},
author = {Hassani, Hossein},
month = aug,
year = {2007},
pages = {239--257},
}


@article{vautard_singular_1989,
title = {Singular spectrum analysis in nonlinear dynamics, with applications to paleoclimatic time series},
volume = {35},
issn = {0167-2789},
url = {https://www.sciencedirect.com/science/article/pii/0167278989900778},
doi = {10.1016/0167-2789(89)90077-8},
abstract = {We distinguish between two dimensions of a dynamical system given by experimental time series. Statistical dimension gives a theoretical upper bound for the minimal number of degrees of freedom required to describe tje attractor up to the accuracy of the data, taking into account sampling and noise problems. The dynamical dimension is the intrinsic dimension of the attractor and does not depend on the quality of the data. Singular Spectrum Analysis (SSA) provides estimates of the statistical dimension. SSA also describes the main physical phenomena reflected by the data. It gives adaptive spectral filters associated with the dominant oscillations of the system and clarifies the noise characteristics of the data. We apply SSA to four paleoclimatic records. The principal climatic oscillations, and the regime changes in their amplitude are detected. About 10 degrees of freedom are statistically significant in the data. Large noise and insufficient sample length do not allow reliable estimates of the dynamical dimension.},
number = {3},
urldate = {2024-12-21},
journal = {Physica D: Nonlinear Phenomena},
Expand All @@ -52,7 +87,6 @@ @article{broomhead_extracting_1986
issn = {0167-2789},
url = {https://www.sciencedirect.com/science/article/pii/016727898690031X},
doi = {10.1016/0167-2789(86)90031-X},
abstract = {We consider the notion of qualitative information and the practicalities of extracting it from experimental data. Our approach, based on a theorem of Takens, draws on ideas from the generalized theory of information known as singular system analysis due to Bertero, Pike and co-workers. We illustrate our technique with numerical data from the chaotic regime of the Lorenz model.},
number = {2},
urldate = {2024-12-21},
journal = {Physica D: Nonlinear Phenomena},
Expand All @@ -68,7 +102,6 @@ @article{allen_monte_1996
shorttitle = {Monte {Carlo} {SSA}},
url = {https://journals.ametsoc.org/view/journals/clim/9/12/1520-0442_1996_009_3373_mcsdio_2_0_co_2.xml},
doi = {10.1175/1520-0442(1996)009<3373:MCSDIO>2.0.CO;2},
abstract = {Singular systems (or singular spectrum) analysis (SSA) was originally proposed for noise reduction in the analysis of experimental data and is now becoming widely used to identify intermittent or modulated oscillations in geophysical and climatic time series. Progress has been hindered by a lack of effective statistical tests to discriminate between potential oscillations and anything but the simplest form of noise, that is, “white” (independent, identically distributed) noise, in which power is independent of frequency. The authors show how the basic formalism of SSA provides a natural test for modulated oscillations against an arbitrary “colored noise” null hypothesis. This test, Monte Carlo SSA, is illustrated using synthetic data in three situations: (i) where there is prior knowledge of the power-spectral characteristics of the noise, a situation expected in some laboratory and engineering applications, or when the “noise” against which the data is being tested consists of the output of an independently specified model, such as a climate model; (ii) where a simple hypothetical noise model is tested, namely, that the data consists only of white or colored noise; and (iii) where a composite hypothetical noise model is tested, assuming some deterministic components have already been found in the data, such as a trend or annual cycle, and it needs to be established whether the remainder may be attributed to noise. The authors examine two historical temperature records and show that the strength of the evidence provided by SSA for interannual and interdecadal climate oscillations in such data has been considerably overestimated. In contrast, multiple inter- and subannual oscillatory components are identified in an extended Southern Oscillation index at a high significance level. The authors explore a number of variations on the Monte Carlo SSA algorithm and note that it is readily applicable to multivariate series, covering standard empirical orthogonal functions and multichannel SSA.},
language = {en},
urldate = {2024-12-21},
author = {Allen, Myles R. and Smith, Leonard A.},
Expand All @@ -83,7 +116,6 @@ @article{schreiber_surrogate_2000
issn = {0167-2789},
url = {https://www.sciencedirect.com/science/article/pii/S0167278900000439},
doi = {10.1016/S0167-2789(00)00043-9},
abstract = {Before we apply nonlinear techniques, e.g. those inspired by chaos theory, to dynamical phenomena occurring in nature, it is necessary to first ask if the use of such advanced techniques is justified by the data. While many processes in nature seem very unlikely a priori to be linear, the possible nonlinear nature might not be evident in specific aspects of their dynamics. The method of surrogate data has become a very popular tool to address such a question. However, while it was meant to provide a statistically rigorous, foolproof framework, some limitations and caveats have shown up in its practical use. In this paper, recent efforts to understand the caveats, avoid the pitfalls, and to overcome some of the limitations, are reviewed and augmented by new material. In particular, we will discuss specific as well as more general approaches to constrained randomisation, providing a full range of examples. New algorithms will be introduced for unevenly sampled and multivariate data and for surrogate spike trains. The main limitation, which lies in the interpretability of the test results, will be illustrated through instructive case studies. We will also discuss some implementational aspects of the realisation of these methods in the TISEAN software package.},
language = {en},
number = {3},
urldate = {2021-10-09},
Expand Down Expand Up @@ -217,6 +249,5 @@ @book{durbin_time_2012
author = {Durbin, James and Koopman, Siem Jan},
month = may,
year = {2012},
doi = {10.1093/acprof:oso/9780199641178.001.0001},
doi = {10.1093/acprof:oso/9780199641178.001.0001},
doi = {10.1093/acprof:oso/9780199641178.001.0001}
}
73 changes: 40 additions & 33 deletions paper.md
Original file line number Diff line number Diff line change
Expand Up @@ -37,7 +37,7 @@ affiliations:
index: 4
ror: 00r8amq78

date: 25 June 2025
date: 27 October 2025
bibliography: paper.bib
---

Expand All @@ -56,43 +56,50 @@ testing.

# Statement of Needs

SSA is a non-parametric method that allows for the analysis and decomposition of
time series into nonlinear trends and pseudo-periodic signatures, without prior
knowledge of their underlying dynamics
[@elsner_singular_1996; @golyandina_singular_2020]. The basic Singular Spectrum
Analysis (SSA) algorithm for univariate time series, as described by
@broomhead_extracting_1986 (BK-SSA) or @vautard_singular_1989 (VG-SSA), applies
to univariate time series. It consists of three major steps
[@golyandina_singular_2020]. The first step is the time-delayed matrix
construction. The second step consists in a Singular Value Decomposition of the
trajectory matrix. The BK-SSA approach is based on a time-delayed trajectory
matrix with dimensions depending on the window parameter and the number of unit
lags. This matrix consists of lagged copies of time series segments of a
specified length, forming a Hankel matrix, i.e., with equal anti-diagonal
values. In contrast, the VG-SSA approach captures time dependencies by
constructing a special type of covariance matrix that has a Toeplitz structure,
meaning that its diagonal values are identical. The eigenvalues of the SVD
depend on the variance captured by each mode, either composed of one (trend) or
two (trend or pseudo-periodic cycles) eigenvectors (or components). In the
third step, the eigenvectors are then grouped, for pseudo-periodic components,
and their contributions to the time series are reconstructed via projection.
SSA is a non-parametric method that provides a low-assumption framework for
exploring, discovering, and decomposing linear or nonlinear or pseudo-periodic
patterns in time series data, in contrast to methods that require strong
_a priori_ hypotheses about signal components
[@elsner_singular_1996; @golyandina_singular_2020].
The SSALib package includes Monte Carlo SSA to support statistical inference and
reduce subjective user guidance. Its Python Application Programming Interface is
designed to streamline the SSA workflow and facilitate time series exploration,
including built-in plotting features.

SSALib is particularly relevant for researchers and practitioners working in
domains where time series analysis is central, i.e., climate and environmental
sciences, geophysics, neuroscience, econometrics, or epidemiology.

# Mathematical Background

The mathematical background of Singular Spectrum Analysis (SSA) has been
primarily developed during the 1980–2000 period
[@golyandina_particularities_2020; @elsner_singular_1996; @golyandina_singular_2020].
The basic Singular Spectrum Analysis (SSA) algorithm for univariate time
series, as described by @broomhead_extracting_1986 (BK-SSA) or
@vautard_singular_1989 (VG-SSA), applies to univariate time series. It consists
of three major steps [@hassani_singular_2007; @golyandina_singular_2020]. The
first step is the time-delayed matrix construction. The second step consists in
a Singular Value Decomposition of the trajectory matrix. The BK-SSA approach is
based on a time-delayed trajectory matrix with dimensions depending on the
window parameter and the number of unit lags. This matrix consists of lagged
copies of time series segments of a specified length, forming a Hankel matrix,
i.e., with equal anti-diagonal values. In contrast, the VG-SSA approach captures
time dependencies by constructing a special type of covariance matrix that has a
Toeplitz structure, meaning that its diagonal values are identical. The
eigenvalues of the SVD depend on the variance captured by each mode, either
composed of one (trend) or two (trend or pseudo-periodic cycles) eigenvectors
(or components). In the third step, the eigenvectors are then grouped, for
pseudo-periodic components, and their contributions to the time series are
reconstructed via projection.

For testing the significance of the retrieved mode, @allen_monte_1996
proposed a Monte-Carlo approach, by comparison of the variance captured by the
eigenvector on the original time series with that captured in many random
autoregressive surrogate time series [@schreiber_surrogate_2000]. Many
extensions have been proposed for
the methods, paving the way for future developments, such as multi-time
series method (M-SSA), SSA-based interpolation and extrapolation, or causality
tests.

As a nonparametric method, SSA provides a low-assumption framework for
exploring, discovering, and decomposing linear or nonlinear patterns in time
series data, in contrast to methods that require strong _a priori_ hypotheses
about signal components. The SSALib package includes Monte Carlo SSA to support
statistical inference and reduce subjective user guidance. Its Python
Application Programming Interface is designed to streamline the SSA workflow and
facilitate time series exploration, including built-in plotting features.
extensions have been proposed for the methods, paving the way for future
developments, such as multi-time series method (M-SSA), SSA-based interpolation
and extrapolation, or causality tests [@golyandina_singular_2020].

# Implementation Details

Expand Down