From 5fa1381f9017031d4a1bf59d5fbaca4909382448 Mon Sep 17 00:00:00 2001 From: Thor Whalen <1906276+thorwhalen@users.noreply.github.com> Date: Tue, 22 Sep 2026 14:05:23 +0000 Subject: [PATCH] fix(audio): 8-bit bias no longer overflows numpy int8 waveforms encode_wav_bytes(np.int8 array, width_bytes=1) worked before #14 and raised OverflowError after it (numpy 2) because the +128 bias was applied to np.int8 scalars. Samples are now widened with operator.index before the shift, which also keeps non-integer samples failing loudly instead of truncating. Co-Authored-By: Claude Opus 5 --- recode/audio.py | 23 ++++++++++++++++++----- test_recode.py | 19 +++++++++++++++++++ 2 files changed, 37 insertions(+), 5 deletions(-) diff --git a/recode/audio.py b/recode/audio.py index 7d2dabb..43660f5 100644 --- a/recode/audio.py +++ b/recode/audio.py @@ -36,6 +36,7 @@ """ +import operator import struct import warnings import wave @@ -164,13 +165,25 @@ def _shift_samples(frames, by: int): [-128, 0, 127] >>> _shift_samples([(0, 255), (128, 64)], -128) [(-128, 127), (0, -64)] + + Samples are widened to Python `int` before the shift, so a numpy `int8` waveform -- + the natural dtype for 8-bit audio -- does not overflow on its way to 0..255, and a + non-integer sample still fails loudly (as the struct codec always made it) instead + of being truncated: + + >>> _shift_samples([-128, 0, 127], 128) + [0, 128, 255] + >>> _shift_samples([0.5], 128) + Traceback (most recent call last): + ... + TypeError: 'float' object cannot be interpreted as an integer """ + + def shift(sample): + return operator.index(sample) + by + return [ - ( - tuple(sample + by for sample in frame) - if isinstance(frame, Iterable) - else frame + by - ) + tuple(map(shift, frame)) if isinstance(frame, Iterable) else shift(frame) for frame in frames ] diff --git a/test_recode.py b/test_recode.py index de81e3d..ef688a7 100644 --- a/test_recode.py +++ b/test_recode.py @@ -770,3 +770,22 @@ def test_a_non_pcm_extensible_subformat_still_raises(): assert non_pcm != raw # guard: the fixture really did carry the PCM GUID with pytest.raises(_wave.Error): decode_wav_bytes(non_pcm) + + +def test_8_bit_encode_accepts_a_numpy_int8_waveform(): + """int8 is the natural dtype of 8-bit audio; biasing it by 128 must not overflow. + + Before the unsigned-8-bit fix, an int8 array went straight to the struct codec and + encoded fine; the bias then added a Python 128 to each np.int8 sample, which numpy 2 + refuses (OverflowError) and numpy 1 silently wraps. + """ + np = pytest.importorskip("numpy") + from recode.audio import encode_wav_bytes + + mono = np.array([-128, 0, 127], dtype=np.int8) + raw = encode_wav_bytes(mono, _SR, width_bytes=1) + assert decode_wav_bytes(raw) == ([-128, 0, 127], _SR) + + stereo = np.array([[-128, 1], [0, 127]], dtype=np.int8) + raw = encode_wav_bytes(stereo, _SR, width_bytes=1, n_channels=2) + assert decode_wav_bytes(raw) == ([(-128, 1), (0, 127)], _SR)