Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
31 commits
Select commit Hold shift + click to select a range
72f0fb2
modified oaconvolve
NimaSarajpoor Jan 8, 2026
be0035e
update code logic
NimaSarajpoor Jan 8, 2026
907e2e2
minor clean ups
NimaSarajpoor Jan 8, 2026
969f187
major changes to imporve readability
NimaSarajpoor Jan 9, 2026
b366e35
Add param blocksize
NimaSarajpoor Jan 9, 2026
a3d2e44
add temp test for challenger
NimaSarajpoor Jan 9, 2026
22d233b
add func for computing block size
NimaSarajpoor Jan 10, 2026
d6eedfa
remove redundant code
NimaSarajpoor Jan 10, 2026
4e23694
added clearer functions
NimaSarajpoor Jan 11, 2026
01a0e7a
minor change
NimaSarajpoor Jan 11, 2026
dddb708
minor change to help with future refactoring
NimaSarajpoor Jan 11, 2026
41db845
minor change
NimaSarajpoor Jan 11, 2026
99e450b
Added reference for finding optimal block size
NimaSarajpoor Jan 11, 2026
e8fa331
fixed test
NimaSarajpoor Jan 11, 2026
f45f541
revise comment
NimaSarajpoor Jan 12, 2026
39e936c
removed overlap-add explanation. Created PR#36 instead
NimaSarajpoor Jan 13, 2026
f6fed15
renaming private functions to reflect valid convolution
NimaSarajpoor Jan 14, 2026
ccbb651
Merge branch 'main' into oaconvolve
NimaSarajpoor May 17, 2026
1033584
address comments
NimaSarajpoor May 19, 2026
d548d0d
add docstrings and comments
NimaSarajpoor May 19, 2026
c78d67b
improved docstrings and comments
NimaSarajpoor May 20, 2026
fd98840
update comments and docstrings
NimaSarajpoor Aug 1, 2026
0205e15
update comments and docstrings
NimaSarajpoor Aug 1, 2026
bdf8b89
fixed format
NimaSarajpoor Aug 2, 2026
d254ef0
updated imports
NimaSarajpoor Aug 2, 2026
298c5f9
resolved import error and enhanced comment
NimaSarajpoor Aug 10, 2026
3bb3b37
removed redudant code
NimaSarajpoor Aug 10, 2026
d18d010
added a comment
NimaSarajpoor Aug 10, 2026
8506a49
minor changes and comments
NimaSarajpoor Aug 23, 2026
f2e8328
minor changes and fixed coverage
NimaSarajpoor Aug 23, 2026
5d0649c
minor changes
NimaSarajpoor Aug 24, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
225 changes: 218 additions & 7 deletions sdp/challenger_sdp.py
Original file line number Diff line number Diff line change
@@ -1,14 +1,225 @@
import math

import numpy as np
from scipy.fft import next_fast_len
from scipy.special import lambertw


def setup(Q, T):
return
from sdp import pocketfft_r2c_c2r_sdp

# _duccfft replaced _pocketfft in scipy 1.18
try:
from scipy.fft._duccfft.basic import c2r, r2c
except ModuleNotFoundError: # pragma: no cover
from scipy.fft._pocketfft.basic import c2r, r2c


def _compute_block_size(m, n, conv_block_size=None):
Comment thread
NimaSarajpoor marked this conversation as resolved.
"""
Return a block size for the overlap-add method.

Parameters
----------
m : int
Length of the query array Q.

n : int
Length of the time series T.

conv_block_size : int, default None
Block size for the convolution. When `conv_block_size` is None,
it will be automatically set to an optimal value, internally
computed based on the lengths of Q and T.

Returns
-------
conv_block_size : int
Block size for the convolution. Will be at least `m` and at most `n`.
"""
if conv_block_size is None:
# `conv_block_size < n` as, otherwise, there is no
# point in splitting the larger array of length `n`
# `conv_block_size >= 2 * (m-1)` so that
# the vectorized operation can be used later.
# Therefore: `m < n/2 + 1`

# Note:
# A tighter upper bound can be computed
# by considering the range of values returned
# by the `lambertw(..., k=-1)` function for m>=3
if m >= n / 2 + 1:
conv_block_size = n
else:
# To minimize Eq. 3 in
# https://en.wikipedia.org/wiki/Overlap–add_method
# ToDo: Revise `opt_size` by considering RFFT/IRFFT
# instead of FFT/IFFT in the computational cost
overlap = m - 1
Comment thread
NimaSarajpoor marked this conversation as resolved.
opt_size = -overlap * lambertw(-1 / (2 * math.e * overlap), k=-1).real
conv_block_size = next_fast_len(math.ceil(opt_size), real=True)
Comment thread
NimaSarajpoor marked this conversation as resolved.

# ToDo
# The computed (presumed) optimal `conv_block_size` is based on an approximate
# cost function (See Eq. 3 in https://en.wikipedia.org/wiki/Overlap–add_method).
# However, we should plug the obtained value into a "more accurate" cost function
# and compared it with the cost of regular circular convolution to see
# whether overlap-add should be considered or not.

# Each chunk of `T` is padded with `m - 1` zeros to form a convolution block.
# Since a chunk (from `T`) must contain at least one element,
# the minimum block size is `m`. However, to take advantage of vectorized
# operation at a later step, the minimum block size is set to `2 * (m-1)`
conv_block_size = max(conv_block_size, 2 * (m - 1))
Comment thread
NimaSarajpoor marked this conversation as resolved.

# `conv_block_size < n` as, otherwise, there is no
# point in splitting the larger array of length `n`
return min(conv_block_size, n)


def _pocketfft_circular_convolve_block(Q, T, conv_block_size):
m = Q.shape[0]
n = T.shape[0]

# Each block in overlap-add method needs to be padded
# with `m-1` zeros. Therefore, the effective block size
# for T is `conv_block_size - (m-1)`.
T_block_size = conv_block_size - (m - 1)
n_blocks = math.ceil(n / T_block_size)
last_block_start = (n_blocks - 1) * T_block_size

# To compute the circular convolution between the zero-padded Q
# and each zero-padded block of T, the data can be loaded into
# a 2D array with `n_blocks + 1` rows, where the first `n_blocks`
# rows correspond to the blocks of T, and the last row is the
# zero-padded Q.
tmp = np.empty((n_blocks + 1, conv_block_size), dtype=np.float64)
tmp[: n_blocks - 1, :T_block_size] = T[:last_block_start].reshape(
n_blocks - 1, T_block_size
)
tmp[: n_blocks - 1, T_block_size:] = 0.0
tmp[n_blocks - 1, : n - last_block_start] = T[last_block_start:]
tmp[n_blocks - 1, n - last_block_start :] = 0.0

tmp[n_blocks, :m] = Q
tmp[n_blocks, m:] = 0.0

fft_2d = r2c(True, tmp, axis=-1)

return c2r(False, np.multiply(fft_2d[:-1], fft_2d[[-1]]), n=conv_block_size)


def _pocketfft_valid_oaconvolve(Q, T, conv_block_size):
"""
Compute the valid convolution between Q and T using the overlap-add method.
This method performs several circular convolutions between Q and blocks of T,
and then combines the results to obtain the valid convolution between Q and T

Parameters
----------
Q : numpy.ndarray
Query array or subsequence.

def sliding_dot_product(Q, T):
T : numpy.ndarray
Time series or sequence.

conv_block_size : int
Block size for the overlap-add method.
The value cannot be less than len(Q).

Returns
-------
out : numpy.ndarray
The valid convolution between Q and T.

Notes
-----
Each block of the convolution contains part of `T`, padded with `len(Q)-1`
zeros. Therefore, `conv_block_size` must be at least `len(Q)` so that it
can cover at least one element of `T` in each block. However, The current
implementation requires the `conv_block_size` to be at least `2*(len(Q) - 1)`
"""
# performs several circular convolutions between
# zero-padded Q and zero-padded blocks of T
# and returns a 2D array of the results,
# where each row is associated with a block of T
QT_conv_blocks = _pocketfft_circular_convolve_block(Q, T, conv_block_size)

# The subsequences at the boundaries of the blocks
# are shared between adjacent blocks.
# The following logic is needed to reconstruct
# the valid convolution between Q and T
overlap = len(Q) - 1
out = QT_conv_blocks[:, :-overlap]
out[1:, :overlap] += QT_conv_blocks[:-1, -overlap:]

return np.reshape(out, (-1,))[len(Q) - 1 : len(T)]


def _valid_convolve(Q, T, conv_block_size=None):
"""
Compute the valid convolution between Q and T

Parameters
----------
Q : numpy.ndarray
Query array or subsequence.

T : numpy.ndarray
Time series or sequence.

conv_block_size : int, default None
Block size for the overlap-add method. When `conv_block_size`
is None, it will automatically be set to an optimal value,
internally computed based on the lengths of Q and T.

Returns
-------
out : numpy.ndarray
The valid convolution between Q and T.

Notes
-----
The valid convolution between ``Q`` and ``T`` is equivalent to
the sliding dot product between Q[::-1] and T.
"""
m = len(Q)
l = T.shape[0] - m + 1
out = np.empty(l)
for i in range(l):
out[i] = np.dot(Q, T[i : i + m])
n = len(T)
conv_block_size = _compute_block_size(m, n, conv_block_size=conv_block_size)
if conv_block_size >= n:
out = pocketfft_r2c_c2r_sdp._pocketfft_valid_convolve(Q, T)
else:
out = _pocketfft_valid_oaconvolve(Q, T, conv_block_size)

return out


def setup(Q, T):
return


def sliding_dot_product(Q, T, conv_block_size=None):
Comment thread
NimaSarajpoor marked this conversation as resolved.
"""
Compute the sliding dot product between Q and T

Parameters
----------
Q : numpy.ndarray
Query array or subsequence.

T : numpy.ndarray
Time series or sequence.

conv_block_size : int, default None
Block size for the overlap-add method. When `conv_block_size`
is None, it will automatically be set to an optimal value,
internally computed based on the lengths of Q and T.

Returns
-------
out : numpy.ndarray
The sliding dot product between Q and T.
"""
if len(Q) == len(T):
return np.dot(Q, T)
Comment thread
NimaSarajpoor marked this conversation as resolved.
else:
return _valid_convolve(Q[::-1], T, conv_block_size=conv_block_size)
29 changes: 22 additions & 7 deletions sdp/pocketfft_r2c_c2r_sdp.py
Original file line number Diff line number Diff line change
@@ -1,22 +1,37 @@
import numpy as np
from scipy.fft import next_fast_len
from scipy.fft._pocketfft.basic import r2c, c2r


def setup(Q, T):
return
# _duccfft replaced _pocketfft in scipy 1.18
try:
from scipy.fft._duccfft.basic import c2r, r2c
except ModuleNotFoundError: # pragma: no cover
from scipy.fft._pocketfft.basic import c2r, r2c


def sliding_dot_product(Q, T):
def _pocketfft_valid_convolve(Q, T):
Comment thread
NimaSarajpoor marked this conversation as resolved.
"""
Compute the valid convolution between ``Q`` and ``T``
using circular convolution in the frequency domain
"""
n = len(T)
m = len(Q)
next_fast_n = next_fast_len(n, real=True)

tmp = np.empty((2, next_fast_n))
tmp[0, :m] = Q[::-1]
tmp[0, :m] = Q
tmp[0, m:] = 0.0
tmp[1, :n] = T
tmp[1, n:] = 0.0
fft_2d = r2c(True, tmp, axis=-1)

return c2r(False, np.multiply(fft_2d[0], fft_2d[1]), n=next_fast_n)[m - 1 : n]
return c2r(False, np.multiply(fft_2d[0], fft_2d[1]), n=next_fast_n)[
len(Q) - 1 : len(T)
]


def setup(Q, T):
return


def sliding_dot_product(Q, T):
return _pocketfft_valid_convolve(Q[::-1], T)
17 changes: 16 additions & 1 deletion test.py
Original file line number Diff line number Diff line change
Expand Up @@ -119,7 +119,7 @@ def test_sdp(n_T, remainder, comparator):
97,
]
n_Q_power2 = [2, 4, 8, 16, 32, 64]
n_Q_values = n_Q_prime + n_Q_power2 + [n_T]
n_Q_values = n_Q_prime + n_Q_power2 + [n_T - 1, n_T]
n_Q_values = sorted(n_Q for n_Q in set(n_Q_values) if n_Q <= n_T)

modules = utils.import_sdp_mods()
Expand Down Expand Up @@ -210,3 +210,18 @@ def test_pyfftw_sdp_max_n():
np.testing.assert_allclose(comp, ref)

return


def test_oaconvolve_sdp_blocksize():
from sdp.challenger_sdp import sliding_dot_product

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This line needs to be modified if, at a later time, we decide to move the proposal to a new file (module).


T = np.random.rand(2**10)
Q = np.random.rand(2**8)
conv_block_size = 2**9

comp = sliding_dot_product(Q, T, conv_block_size=conv_block_size)
ref = naive_sliding_dot_product(Q, T)

np.testing.assert_allclose(comp, ref)

return
Loading