upscaling 1D array in numpy by mean

Viewed 75

I have a data array such as:

data = np.array([1,3,5,5,2,7,3,5,2,5])

and I want to "upscale" the data using a mean of a given number of samples.

n_steps = 2
upscaled_data = [2, 2, 5, 5, 4.5, 4.5, 4, 4, 3.5, 3.5]

or

n_steps = 3
upscaled_data = [3,3,3,4.7, 4.7, 4.7, 3.33, 3.33, 3.33, 5]

Currently I am doing this:

def upscale_by_mean(data, sample_rate):
    upscaled_data = np.array([])
    for i in range(sample_rate, len(data), sample_rate):
        mean = np.nanmean(data[i-(sample_rate):i])
        val_to_append = np.full(shape=(sample_rate,), fill_value=mean)
        upscaled_data = np.append(upscaled_data, val_to_append)
    
    #The last three lines are just to handle when len(data)/sample_rate is odd
    mean = np.nanmean(data[len(upscaled_data):])
    val_to_append = np.full(shape=(len(data)-len(upscaled_data),), fill_value=mean)
    upscaled_data = np.append(upscaled_data, val_to_append)
    
    return upscaled_data

These above works as expected. However, when I scale it up to an array of 50 million samples, the runtimes become concerning. It seems like there should be a more efficient solution vectorized solution to this.

Edit: An alternative solution that has the same runtime as the answer below is:

def upscale_by_mean(data, sample_rate=5):
    new_data = data[:len(data)-len(data)%sample_rate].reshape(len(data)//sample_rate, sample_rate)
    row_mean= np.nanmean(new_data, axis=1)
    upscaled_data = np.repeat(row_mean, sample_rate)
    if data.size % sample_rate:
        row_mean = np.nanmean(data[-(data.size % sample_rate):])
        val_to_append = np.full(shape=(len(data)-len(upscaled_data),), fill_value=row_mean)
        upscaled_data = np.append(upscaled_data, val_to_append)
    return upscaled_data
1 Answers

Best I can think of is to use np.add.reduceat to vectorize the summing

def _upscale_by_mean(data, sample_rate):
    l = data.size
    ix = np.zeros(l, dtype = int)
    ix[::sample_rate] = 1
    
    vals = np.add.reduceat(data, np.flatnonzero(ix))/sample_rate
    if l % sample_rate:
        vals[-1] = data[-(l % sample_rate):].mean() 
    return vals[ix.cumsum()-1]

Testing:

data = np.random.randint(10, size = 1000)

%timeit _upscale_by_mean(data, 2) #above
26.9 µs ± 756 ns per loop (mean ± std. dev. of 7 runs, 10000 loops each)

%timeit upscale_by_mean(data, 2) #original
9.91 ms ± 308 µs per loop (mean ± std. dev. of 7 runs, 100 loops each)
Related