From ba927b9de355613c5ef4749dc7fedca49bcb4e84 Mon Sep 17 00:00:00 2001 From: JakimPL Date: Mon, 5 Oct 2026 13:00:41 +0200 Subject: [PATCH 01/12] Lowered: the constant-Q floor to the triangle's lowest note --- CHANGELOG.md | 1 + docs/concepts/reconstruction.md | 7 ++- docs/development/bugs-and-todos.md | 3 + docs/formats/instruction-libraries.md | 9 +-- .../logic/instruction/library.py | 20 ++++++- src/sampletones_core/constants/algorithm.py | 2 +- src/sampletones_core/constants/general.py | 3 + src/sampletones_core/constants/spectrum.py | 5 +- src/sampletones_core/fft/cqt/transform.py | 41 +++++++++---- src/sampletones_core/fft/features/base.py | 19 ++---- src/sampletones_core/fft/features/cqt.py | 7 +-- src/sampletones_core/fft/features/windowed.py | 8 ++- src/sampletones_core/fft/fragment/fragment.py | 6 +- src/sampletones_core/fft/spectrum/fft.py | 7 ++- src/sampletones_core/fft/spectrum/spectrum.py | 19 +++--- .../generators/implementation/triangle.py | 3 +- src/sampletones_core/library/fragment.py | 4 +- src/sampletones_core/timers/timer.py | 10 ++++ tests/suite/fragments.py | 1 - .../logic/instruction/test_library_logic.py | 58 ++++++++++++++++++- .../fft/cqt/test_transform.py | 35 +++++++++++ .../sampletones_core/fft/features/test_cqt.py | 40 +++++++++++++ tests/unit/sampletones_core/fft/test_axis.py | 26 ++++++++- .../fft/test_instantaneous.py | 29 ++++++++++ .../reconstructor/stems/test_frame.py | 1 - .../reconstructor/stems/test_models.py | 1 - 26 files changed, 296 insertions(+), 69 deletions(-) create mode 100644 tests/unit/sampletones_core/fft/features/test_cqt.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 966f0beb6..f98a2b5b4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,7 @@ * Added stems conversion: to mix several recordings into one reconstruction. * Changed drive to reach for louder instructions while a recording converts; reconvert anything converted at a drive other than `1.00`. * Improved clarity of reconstructions. +* Fixed the lowest triangle notes wobbling in pitch after a conversion; libraries are built again for it. * Optimized the size of reconstructions. * Bumped the reconstruction data-version to `2.2` with backward compatibility for `2.1`. * Bumped the library data-version to `2.1`. diff --git a/docs/concepts/reconstruction.md b/docs/concepts/reconstruction.md index 8ede1f37e..eff67174f 100644 --- a/docs/concepts/reconstruction.md +++ b/docs/concepts/reconstruction.md @@ -96,7 +96,7 @@ rate: |----------|---------------------------|-----------------------------------------|---------------------------| | `fft` | linear | uniform, `Δf ≈ sample_rate / N ≈ 27 Hz` | one short window (~37 ms) | | `logfft` | logarithmic, floored at `Δf` | the FFT's `Δf`, on a musical axis | one short window (~37 ms) | -| `cqt` | logarithmic (constant-Q) | constant *relative* (fine low end) | long for low notes (~300 ms) | +| `cqt` | logarithmic (constant-Q) | constant *relative* (fine low end) | long for low notes (~600 ms) | They sit at different points of the **time–frequency trade-off** (the Gabor limit: sharper frequency resolution requires a longer time window, and vice versa): @@ -113,7 +113,10 @@ sharper frequency resolution requires a longer time window, and vice versa): sharply in time. - **CQT** (constant-Q transform) places bins geometrically and gives every musical interval the same number of bins, so it resolves low pitches finely. - It is the default. + It is the default. Its lowest bin sits at the lowest note the chip sounds: the + triangle's, an octave below the pulse's, since the triangle steps through its wave + at half the pulse's rate. A bass line on the triangle is therefore read from its + fundamental, which is where the pitch of a bent note is read from too. The price is time support: its low-frequency basis functions are long (hundreds of milliseconds), so brief events are smeared in time at the low end. _SampleToNES_ computes the CQT **once over the whole signal** with a hop of one frame, so each diff --git a/docs/development/bugs-and-todos.md b/docs/development/bugs-and-todos.md index 72dc40e34..c82c29567 100644 --- a/docs/development/bugs-and-todos.md +++ b/docs/development/bugs-and-todos.md @@ -67,6 +67,9 @@ dimension the import starts carrying. project sample. An edit to such a document is undoable nowhere ([undo](application/undo.md)), so an edit that silences a channel or lets a recording go is reversible only by reloading the file. * Improve performance of the browser's favorite scan of the entire tree per click +* The `fft` and `logfft` analysis floors. Their window spans two cycles of the pulse's lowest note, so the + triangle's lowest octave reaches them by its harmonics alone. Covering it doubles their window and + coarsens their timing. ## Architecture diff --git a/docs/formats/instruction-libraries.md b/docs/formats/instruction-libraries.md index e9be49e9a..03756e50e 100644 --- a/docs/formats/instruction-libraries.md +++ b/docs/formats/instruction-libraries.md @@ -23,8 +23,9 @@ Each entry contains: * **pitch** (33–119) for pulse and triangle, or **period** (0–15) for noise; * **volume** (0–15) for pulse and noise; * **duty_cycle** (0–3) for pulse, or the **short** (0–1) flag for noise; -* **waveform** — one full period of the rendered wave (the longest noise samples - are trimmed to one second); +* **waveform** — one full period of the rendered wave. The longest noise samples + are trimmed to two seconds, which holds the whole stretch a constant-Q spectrum + is read over; * **spectrum** — the waveform's precomputed frequency content. ### Configuration key @@ -40,14 +41,14 @@ A file holds a deflated [MessagePack](https://msgpack.org/) payload, with the fr [Reconstructions](reconstructions.md#storage-and-export), and has its configuration in the file name: ``` -sr_44100_nf_60_ws_13579_tg_0_sm_cqt_ch_384e710987cb958adf2b214df1267d10.ins +sr_44100_nf_60_ws_27157_tg_0_sm_cqt_ch_384e710987cb958adf2b214df1267d10.ins ``` | Fragment | Meaning | | --- | --- | | `sr_44100` | sample rate 44100 Hz | | `nf_60` | NES frequency 60 Hz | -| `ws_13579` | FFT window size (samples) | +| `ws_27157` | FFT window size (samples) | | `tg_0` | transformation gamma 0 | | `sm_cqt` | spectrum method (`fft` / `logfft` / `cqt`) | | `ch_384e…` | a hash of the library configuration section | diff --git a/src/sampletones_application/logic/instruction/library.py b/src/sampletones_application/logic/instruction/library.py index 604b62cea..e6192536c 100644 --- a/src/sampletones_application/logic/instruction/library.py +++ b/src/sampletones_application/logic/instruction/library.py @@ -71,6 +71,7 @@ def __init__( self._library_manager = library_manager self._is_operation_active = is_operation_active self._eta_estimator: Optional[ETAEstimator] = None + self._replaced_library_path: Optional[Path] = None self._lock_function: Optional[VoidCallback] = None self._unlock_function: Optional[VoidCallback] = None @@ -278,14 +279,19 @@ def rebuild_library(self, library_key: InstructionLibraryKey) -> None: it was built for, the exclusive-operation gate permitting. Those settings become the configuration's before the generation starts, which writes the - library in the place of the one it replaces. + library in the place of the one it replaces. Where this build names that library's file + differently, the replaced file is removed once the new one is written. """ if self._is_operation_active(): logger.warning("A conversion or library generation is already in progress") return self.call(self.on_apply_library_config, library_key, self._library_manager.stored_config(library_key)) + replaced = self._library_manager.get_path(library_key) + rebuilt = self._library_manager.get_path(self._config_manager.key) self.generate_library() + if self._library_manager.is_generating() and replaced != rebuilt: + self._replaced_library_path = replaced def generate_library(self) -> None: if self._library_manager.is_generating(): @@ -511,21 +517,31 @@ def _update_progress_state(self, task_progress: TaskProgress) -> None: ) def _on_generation_completed(self) -> None: - """Closes the generation and reads the catalog again, which lists the library it wrote.""" + """Closes the generation, removes the file a rebuild replaced under another name, and reads + the catalog again, which lists the library it wrote.""" self.call(self.on_generation_completed) self._close_generation() + self._remove_replaced_library() self.refresh_libraries(load_if_needed=False) def _on_generation_error(self, exception: Exception) -> None: self.call(self.on_generation_error, exception) self._close_generation() + self._replaced_library_path = None self.update_status() def _on_generation_canceled(self) -> None: self.call(self.on_generation_canceled) self._close_generation() + self._replaced_library_path = None self.update_status() + def _remove_replaced_library(self) -> None: + """Removes the file a rebuild replaced under another name, where it still stands.""" + replaced, self._replaced_library_path = self._replaced_library_path, None + if replaced is not None and replaced.is_file(): + remove_path(replaced) + def _close_generation(self) -> None: """Lets the creator go along with the tree lock the generation took when it was asked for, which loading a library and rebuilding the tree both yield to.""" diff --git a/src/sampletones_core/constants/algorithm.py b/src/sampletones_core/constants/algorithm.py index 4b646ef82..7eb9eb837 100644 --- a/src/sampletones_core/constants/algorithm.py +++ b/src/sampletones_core/constants/algorithm.py @@ -23,7 +23,7 @@ # Library creation MIN_SAMPLE_LENGTH: Final[float] = 0.05 -MAX_SAMPLE_LENGTH: Final[float] = 1.0 +MAX_SAMPLE_LENGTH: Final[float] = 2.0 LIBRARY_PHASES_PER_SAMPLE: Final[int] = 100 # Calculation methods diff --git a/src/sampletones_core/constants/general.py b/src/sampletones_core/constants/general.py index 188838d63..4de7d6553 100644 --- a/src/sampletones_core/constants/general.py +++ b/src/sampletones_core/constants/general.py @@ -13,7 +13,10 @@ MIN_PLAYED_PITCH: Final[int] = LIMIT_MIN_PITCH PITCH_RANGE: Final[int] = MAX_PITCH - MIN_PITCH +TRIANGLE_PHASE_INCREMENT: Final[float] = 0.5 + MIN_FREQUENCY: Final[float] = APU_CLOCK / (TIMER_CYCLE_DIVIDER * (MAX_TIMER + 1)) +MIN_TRIANGLE_FREQUENCY: Final[float] = MIN_FREQUENCY * TRIANGLE_PHASE_INCREMENT MAX_FREQUENCY: Final[float] = APU_CLOCK / TIMER_CYCLE_DIVIDER NOTE_NAMES: Tuple[str, ...] = ( diff --git a/src/sampletones_core/constants/spectrum.py b/src/sampletones_core/constants/spectrum.py index 1dcd643cf..471e47677 100644 --- a/src/sampletones_core/constants/spectrum.py +++ b/src/sampletones_core/constants/spectrum.py @@ -2,10 +2,11 @@ from sampletones_shared.constants.music import OCTAVE_SEMITONES -from .general import MIN_FREQUENCY, QUIETEST_VOLUME_LEVEL +from .general import MIN_FREQUENCY, MIN_TRIANGLE_FREQUENCY, QUIETEST_VOLUME_LEVEL BINS_PER_OCTAVE: Final[int] = OCTAVE_SEMITONES -CQT_CUTOFF_FREQUENCY: Final[float] = MIN_FREQUENCY +CQT_CUTOFF_FREQUENCY: Final[float] = MIN_TRIANGLE_FREQUENCY +LOG_FFT_CUTOFF_FREQUENCY: Final[float] = MIN_FREQUENCY SPECTRUM_FLOOR: Final[float] = QUIETEST_VOLUME_LEVEL**2 CQT_REFERENCE_CONTEXT_FACTOR: Final[int] = 3 diff --git a/src/sampletones_core/fft/cqt/transform.py b/src/sampletones_core/fft/cqt/transform.py index 7f70f38fb..791a33c81 100644 --- a/src/sampletones_core/fft/cqt/transform.py +++ b/src/sampletones_core/fft/cqt/transform.py @@ -1,4 +1,4 @@ -from typing import Optional +from typing import Final, Optional import numpy as np @@ -13,29 +13,50 @@ from ..utils import calculate_n_bins from .kernel import CQTKernel, build_cqt_kernel +MAX_GATHERED_SAMPLES: Final[int] = 1 << 24 -def _framed_signal(audio: np.ndarray, frame_length: int, hop_length: int) -> Array: - """Stack centered frames of ``audio`` on the compute device, one column per hop. - ``audio`` is zero-padded by half a frame on each side so column ``t`` is centered on sample - ``t * hop_length``; the number of columns is ``1 + len(audio) // hop_length``. - """ +def _padded_signal(audio: np.ndarray, frame_length: int) -> Array: + """``audio`` on the compute device, zero-padded by half a frame on each side, which centers column + ``t`` on sample ``t * hop_length``.""" left = frame_length // 2 right = frame_length - left device_audio: Array = xp.asarray(audio, dtype=xp.complex64) padded: Array = xp.concatenate( [xp.zeros(left, dtype=xp.complex64), device_audio, xp.zeros(right, dtype=xp.complex64)] ) - frame_count = 1 + len(audio) // hop_length - starts = xp.arange(frame_count) * hop_length + return padded + + +def _framed_columns( + padded: Array, + frame_length: int, + hop_length: int, + first: int, + count: int, +) -> Array: + """Stack ``count`` centered frames of the padded signal, one column per hop from column ``first``.""" + starts = (xp.arange(count) + first) * hop_length indices = starts[:, None] + xp.arange(frame_length)[None, :] framed: Array = padded[indices].T return framed def _transform(audio: np.ndarray, kernel: CQTKernel, hop_length: int) -> np.ndarray: - frames = _framed_signal(audio, kernel.frame_length, hop_length) - coefficients: Array = kernel.matrix @ frames + """Correlates the kernel with every frame of ``audio``, ``1 + len(audio) // hop_length`` columns. + + Every frame spans the lowest bin's whole wavelet, so the frames are gathered a block of columns + at a time, each block holding at most ``MAX_GATHERED_SAMPLES`` samples. That keeps the memory a + recording takes to one block, however long the recording runs. + """ + padded = _padded_signal(audio, kernel.frame_length) + frame_count = 1 + len(audio) // hop_length + block = max(1, MAX_GATHERED_SAMPLES // kernel.frame_length) + columns = [ + kernel.matrix @ _framed_columns(padded, kernel.frame_length, hop_length, first, min(block, frame_count - first)) + for first in range(0, frame_count, block) + ] + coefficients: Array = xp.concatenate(columns, axis=1) return to_numpy(coefficients) diff --git a/src/sampletones_core/fft/features/base.py b/src/sampletones_core/fft/features/base.py index 88c1ca5ec..815fbbd7b 100644 --- a/src/sampletones_core/fft/features/base.py +++ b/src/sampletones_core/fft/features/base.py @@ -41,33 +41,26 @@ def sample_rate(self) -> int: def extract(self, audio: np.ndarray) -> List[Fragment]: """ - Build one `Fragment` per frame of `audio` (its central slice, its analysis - window, and its spectral feature). + Build one `Fragment` per whole frame of `audio`: its slice and its spectral feature. """ frame_length = self.window.frame_length count = audio.shape[0] // frame_length if count == 0: return [] - windowed_frames = [self.window.get_windowed_frame(audio, frame_id * frame_length) for frame_id in range(count)] - features = self._frame_features(audio, windowed_frames) + features = self._frame_features(audio, count) return [ Fragment( - audio=self.window.get_frame_from_window(windowed_audio), + audio=audio[frame_id * frame_length : (frame_id + 1) * frame_length], feature=feature, - windowed_audio=windowed_audio, config=self.config, ) - for windowed_audio, feature in zip(windowed_frames, features) + for frame_id, feature in enumerate(features) ] @abstractmethod - def _frame_features( - self, - audio: np.ndarray, - windowed_frames: List[np.ndarray], - ) -> List[Histogram]: - """Per-frame features; `windowed_frames` are the frame-centered analysis windows.""" + def _frame_features(self, audio: np.ndarray, count: int) -> List[Histogram]: + """The features of the first `count` frames of `audio`, each read around its own frame.""" @abstractmethod def reference_feature(self, sample: CyclicArray) -> Histogram: diff --git a/src/sampletones_core/fft/features/cqt.py b/src/sampletones_core/fft/features/cqt.py index de4f37a38..081377902 100644 --- a/src/sampletones_core/fft/features/cqt.py +++ b/src/sampletones_core/fft/features/cqt.py @@ -19,13 +19,12 @@ class CQTFeatureExtractor(FeatureExtractor): """ Whole-signal extraction for the `cqt` method. The constant-Q window spans several frames, so the transform runs once over the whole signal, yielding one column per - frame centered on its hop position. Each frame is aligned to its own time position, - and `windowed_frames` supplies the frame count. + frame centered on its hop position. Each frame is aligned to its own time position. """ - def _frame_features(self, audio: np.ndarray, windowed_frames: List[np.ndarray]) -> List[Histogram]: + def _frame_features(self, audio: np.ndarray, count: int) -> List[Histogram]: spectra = calculate_cqt_spectrum_columns(audio, self.sample_rate, self.window.frame_length) - return [self.transformer.forward(spectrum) for spectrum in spectra[: len(windowed_frames)]] + return [self.transformer.forward(spectrum) for spectrum in spectra[:count]] def reference_feature(self, sample: CyclicArray) -> Histogram: buffer = sample.get_fragment(0, CQT_REFERENCE_CONTEXT_FACTOR * self.window.size) diff --git a/src/sampletones_core/fft/features/windowed.py b/src/sampletones_core/fft/features/windowed.py index ddc33c28a..abeebcefb 100644 --- a/src/sampletones_core/fft/features/windowed.py +++ b/src/sampletones_core/fft/features/windowed.py @@ -20,8 +20,12 @@ class WindowedFeatureExtractor(FeatureExtractor): so one class serves both.) """ - def _frame_features(self, audio: np.ndarray, windowed_frames: List[np.ndarray]) -> List[Histogram]: - return [self._windowed_feature(windowed_audio) for windowed_audio in windowed_frames] + def _frame_features(self, audio: np.ndarray, count: int) -> List[Histogram]: + frame_length = self.window.frame_length + return [ + self._windowed_feature(self.window.get_windowed_frame(audio, frame_id * frame_length)) + for frame_id in range(count) + ] def reference_feature(self, sample: CyclicArray) -> Histogram: """The feature of the spectrum the sample averages to over the phases a frame starts on. diff --git a/src/sampletones_core/fft/fragment/fragment.py b/src/sampletones_core/fft/fragment/fragment.py index 2c8e83e01..efccdabf4 100644 --- a/src/sampletones_core/fft/fragment/fragment.py +++ b/src/sampletones_core/fft/fragment/fragment.py @@ -10,12 +10,10 @@ @dataclass(frozen=True) class Fragment: """ - A single analysis frame: its central time-domain slice, the larger analysis - window it was taken from, and its spectral feature. A data holder — features are - produced by a `FeatureExtractor`. + A single analysis frame: its time-domain slice and its spectral feature. A data + holder — features are produced by a `FeatureExtractor`. """ audio: Array feature: Histogram - windowed_audio: Array config: Config diff --git a/src/sampletones_core/fft/spectrum/fft.py b/src/sampletones_core/fft/spectrum/fft.py index ebdc8d540..0ec9547e7 100644 --- a/src/sampletones_core/fft/spectrum/fft.py +++ b/src/sampletones_core/fft/spectrum/fft.py @@ -5,7 +5,7 @@ from sampletones_core.audio import validate_audio_array from sampletones_core.constants.spectrum import ( BINS_PER_OCTAVE, - CQT_CUTOFF_FREQUENCY, + LOG_FFT_CUTOFF_FREQUENCY, ) from sampletones_core.structures.histogram import Histogram @@ -48,7 +48,7 @@ def calculate_log_spaced_fft_spectrum( audio: np.ndarray, sample_rate: int, fft_size: Optional[int] = None, - cutoff: float = CQT_CUTOFF_FREQUENCY, + cutoff: float = LOG_FFT_CUTOFF_FREQUENCY, bins_per_octave: int = BINS_PER_OCTAVE, ) -> Histogram: """ @@ -57,7 +57,8 @@ def calculate_log_spaced_fft_spectrum( Computes the linear FFT spectrum and rebins it onto a logarithmic axis whose bin widths respect the FFT resolution: bins widen to the linear spacing where a musical interval falls below it, so low tones stay compact and adjacent bins - aggregate independent FFT measurements at every frequency. + aggregate independent FFT measurements at every frequency. The axis starts at the + pulse's lowest note, the lowest frequency the FFT window spans two cycles of. Args: audio: Input audio as array. diff --git a/src/sampletones_core/fft/spectrum/spectrum.py b/src/sampletones_core/fft/spectrum/spectrum.py index 1f6b4fe19..b11ee4e93 100644 --- a/src/sampletones_core/fft/spectrum/spectrum.py +++ b/src/sampletones_core/fft/spectrum/spectrum.py @@ -2,10 +2,7 @@ import numpy as np -from sampletones_core.constants.spectrum import ( - BINS_PER_OCTAVE, - CQT_CUTOFF_FREQUENCY, -) +from sampletones_core.constants.spectrum import BINS_PER_OCTAVE from sampletones_core.structures.histogram import Histogram from .cqt import calculate_cqt_spectrum @@ -18,20 +15,20 @@ def calculate_spectrum( audio: np.ndarray, sample_rate: int, fft_size: Optional[int] = None, - cutoff: float = CQT_CUTOFF_FREQUENCY, bins_per_octave: int = BINS_PER_OCTAVE, n_bins: Optional[int] = None, ) -> Histogram: """ Compute the spectrum of the given audio data, for given FFT size and sample rate, - depending on the selected spectrum calculation method. + depending on the selected spectrum calculation method. Each method starts its axis at its + own floor: the constant-Q transform at the triangle's lowest note, and the log-spaced FFT at + the lowest frequency its window spans two cycles of. Args: method: Spectrum calculation method. audio: Input audio data. sample_rate: Sample rate of the audio data. fft_size: Size of the FFT to be used. If None, uses the length of the audio array. - cutoff: Cutoff frequency. All frequencies below this value will be discarded. bins_per_octave: Number of bins per octave. Only used if n_bins is None. n_bins: Number of constant-Q components. Only used by the constant-Q method. @@ -55,16 +52,14 @@ def calculate_spectrum( audio, sample_rate, fft_size, - cutoff, - bins_per_octave, + bins_per_octave=bins_per_octave, ) case SpectrumMethod.CQT: spectrum = calculate_cqt_spectrum( audio, sample_rate, - cutoff, - bins_per_octave, - n_bins, + bins_per_octave=bins_per_octave, + n_bins=n_bins, ) case _: raise ValueError(f"Unsupported spectrum method: {method}") diff --git a/src/sampletones_core/generators/implementation/triangle.py b/src/sampletones_core/generators/implementation/triangle.py index 508624b36..ade6fae75 100644 --- a/src/sampletones_core/generators/implementation/triangle.py +++ b/src/sampletones_core/generators/implementation/triangle.py @@ -8,6 +8,7 @@ MIN_PITCH, MIXER_TRIANGLE, TRIANGLE_OFFSET, + TRIANGLE_PHASE_INCREMENT, ) from sampletones_core.instructions import ( InstructionTypeUnion, @@ -30,7 +31,7 @@ def __init__( sample_rate=config.library.sample_rate, nes_frequency=config.library.nes_frequency, reset_phase=config.generation.reset_phase, - phase_increment=0.5, + phase_increment=TRIANGLE_PHASE_INCREMENT, ) def __call__( diff --git a/src/sampletones_core/library/fragment.py b/src/sampletones_core/library/fragment.py index 4a69b52ce..57b3a0ea2 100644 --- a/src/sampletones_core/library/fragment.py +++ b/src/sampletones_core/library/fragment.py @@ -73,12 +73,10 @@ def instruction(self) -> InstructionT: return instruction def get_fragment(self, shift: int, config: Config, window: Window) -> Fragment: - windowed_audio = self.sample.get_windowed_fragment(shift, window) - audio = window.get_frame_from_window(windowed_audio) + audio = window.get_frame_from_window(self.sample.get_windowed_fragment(shift, window)) return Fragment( audio=audio, feature=self.feature, - windowed_audio=windowed_audio, config=config, ) diff --git a/src/sampletones_core/timers/timer.py b/src/sampletones_core/timers/timer.py index fb7850891..5a2ba1337 100644 --- a/src/sampletones_core/timers/timer.py +++ b/src/sampletones_core/timers/timer.py @@ -53,6 +53,16 @@ def calculate_base_length(self, minimum_length: int) -> int: return round(cycles * cycle_length) def generate_sample(self) -> CyclicArray: + """The library sample of the waveform the timer runs, a stretch that loops. + + A sample spans whole cycles, at least ``MIN_SAMPLE_LENGTH`` seconds, so it loops without a + seam. A cycle longer than ``MAX_SAMPLE_LENGTH``, the slow noise periods in the long mode, keeps + its middle ``MAX_SAMPLE_LENGTH`` seconds and loops with a seam there. That length covers the + stretch a constant-Q reference feature reads, so the reference stays clear of the seam. + + Returns: + CyclicArray: The sample, at the frequency the timer runs. + """ min_sample_length = round(MIN_SAMPLE_LENGTH * self.sample_rate) max_sample_length = round(MAX_SAMPLE_LENGTH * self.sample_rate) base_length = self.calculate_base_length(min_sample_length) diff --git a/tests/suite/fragments.py b/tests/suite/fragments.py index ad8608d61..afa5c640f 100644 --- a/tests/suite/fragments.py +++ b/tests/suite/fragments.py @@ -18,6 +18,5 @@ def amplified(extractor: FeatureExtractor, fragment: Fragment, gain: float) -> F return Fragment( audio=fragment.audio * gain, feature=feature.astype(precision), - windowed_audio=fragment.windowed_audio * gain, config=fragment.config, ) diff --git a/tests/unit/sampletones_application/logic/instruction/test_library_logic.py b/tests/unit/sampletones_application/logic/instruction/test_library_logic.py index 29261b8d3..7a9b98056 100644 --- a/tests/unit/sampletones_application/logic/instruction/test_library_logic.py +++ b/tests/unit/sampletones_application/logic/instruction/test_library_logic.py @@ -53,6 +53,8 @@ EARLIER_VERSION: Final[str] = "2.0" OTHER_GAMMA: Final[int] = 50 OTHER_TUNING: Final[float] = 432.0 +EARLIER_WINDOW_SIZE: Final[int] = 13579 +DERIVED_WINDOW_SIZE: Final[int] = 0 TEXTS: Final[Dict[str, str]] = { INCOMPATIBLE_VERSION_KEY: "got {} expected {}", @@ -465,10 +467,12 @@ def _write_library( directory: Path, library_config: InstructionsLibraryConfig, library_data_version: str, + window_size: int = DERIVED_WINDOW_SIZE, ) -> InstructionLibraryKey: """Writes an empty library built for ``library_config`` under ``directory``, stated at - ``library_data_version``.""" - key = InstructionLibraryKey.create(library_config, Window.from_config(library_config)) + ``library_data_version``, its window ``window_size`` samples long where that is given and the one + this build derives otherwise.""" + key = InstructionLibraryKey.create(library_config, Window.from_config(library_config, custom_size=window_size)) library = InstructionLibraryData.create(Config().model_copy(update={"library": library_config}), {}) stated = library.metadata.model_copy(update={"library_data_version": library_data_version}) directory.mkdir(parents=True, exist_ok=True) @@ -487,6 +491,12 @@ def _opened_library(catalog: Catalog) -> InstructionLibraryKey: return key +def _rebuild(catalog: Catalog) -> None: + """Takes the start report the generation a rebuild asked for sends.""" + catalog.creator.callbacks["on_start"]() + catalog.queue.drain() + + def _complete(catalog: Catalog) -> None: catalog.write_library() @@ -900,6 +910,50 @@ def test_the_settings_it_was_built_for_are_applied_and_the_generation_starts(sel True, ) + def test_a_library_rebuilt_under_another_name_leaves_only_the_new_file(self, catalog: Catalog) -> None: + """A library an earlier build wrote for another window carries that window in its name, and the + rebuild writes the library this build derives in its place.""" + directory = catalog.manager.library_directory + key = _write_library(directory, _other_settings(catalog), EARLIER_VERSION, EARLIER_WINDOW_SIZE) + replaced = catalog.manager.get_path(key) + + catalog.logic.rebuild_library(key) + _rebuild(catalog) + catalog.write_library() + catalog.queue.drain() + + assert (replaced.exists(), catalog.manager.get_path(catalog.config_manager.key).exists()) == (False, True) + + def test_a_library_rebuilt_under_its_own_name_stays(self, catalog: Catalog) -> None: + key = _write_library(catalog.manager.library_directory, _other_settings(catalog), EARLIER_VERSION) + + catalog.logic.rebuild_library(key) + _rebuild(catalog) + catalog.write_library() + catalog.queue.drain() + + assert (catalog.config_manager.key, catalog.manager.library_state(key)) == (key, LibraryState.CURRENT) + + @pytest.mark.parametrize("ending", [_cancel, _fail], ids=["canceled", "failed"]) + def test_a_rebuild_that_ends_unwritten_keeps_the_file_through_later_generations( + self, + catalog: Catalog, + ending: Any, + ) -> None: + directory = catalog.manager.library_directory + key = _write_library(directory, _other_settings(catalog), EARLIER_VERSION, EARLIER_WINDOW_SIZE) + replaced = catalog.manager.get_path(key) + + catalog.logic.rebuild_library(key) + _rebuild(catalog) + ending(catalog) + catalog.queue.drain() + catalog.start_generation() + catalog.write_library() + catalog.queue.drain() + + assert replaced.exists() + def test_a_rebuild_during_another_operation_changes_nothing(self, catalog: Catalog) -> None: settings = catalog.config_manager.config key = _write_library(catalog.manager.library_directory, _other_settings(catalog), EARLIER_VERSION) diff --git a/tests/unit/sampletones_core/fft/cqt/test_transform.py b/tests/unit/sampletones_core/fft/cqt/test_transform.py index 64d94cbe5..16ab2ee88 100644 --- a/tests/unit/sampletones_core/fft/cqt/test_transform.py +++ b/tests/unit/sampletones_core/fft/cqt/test_transform.py @@ -1,14 +1,22 @@ +from dataclasses import dataclass + import numpy as np import pytest +from sampletones_core.constants.enums import CQTWindow from sampletones_core.constants.spectrum import BINS_PER_OCTAVE, CQT_CUTOFF_FREQUENCY +from sampletones_core.fft.cqt import transform as transform_module from sampletones_core.fft.cqt.frequencies import calculate_cqt_frequencies +from sampletones_core.fft.cqt.kernel import build_cqt_kernel from sampletones_core.fft.cqt.transform import calculate_cqt, calculate_cqt_frames from sampletones_core.fft.spectrum.cqt import calculate_cqt_spectrum_columns from sampletones_core.fft.utils import calculate_n_bins +from tests.suite.base import BaseTestSuite +from tests.suite.case import BaseRegularTestCase SAMPLE_RATE = 22050 HOP_LENGTH = 512 +ROUNDING_SHARE = 1e-5 def _bin_count() -> int: @@ -45,6 +53,33 @@ def test_single_frame_returns_one_column(self) -> None: assert cqt.shape == (_bin_count(), 1) +class TestGatheringInBlocks(BaseTestSuite): + """The frames are gathered a block of columns at a time, and every column comes out as one pass + over the whole recording gives it, within the float32 rounding of a product that sums each frame + in another order.""" + + @dataclass(frozen=True, kw_only=True) + class TestCase(BaseRegularTestCase): + columns_per_block: int + + test_cases = ( + TestCase(label="one column a block", columns_per_block=1), + TestCase(label="blocks leaving a shorter last one", columns_per_block=7), + ) + + @pytest.mark.parametrize("test_case", test_cases, ids=lambda test_case: test_case.label) + def test_the_columns_equal_one_pass(self, test_case: TestCase, monkeypatch: pytest.MonkeyPatch) -> None: + signal = _tone(440.0, 2 * SAMPLE_RATE) + whole = calculate_cqt_frames(signal, SAMPLE_RATE, HOP_LENGTH) + kernel = build_cqt_kernel(SAMPLE_RATE, _bin_count(), CQT_CUTOFF_FREQUENCY, BINS_PER_OCTAVE, CQTWindow.HANN) + monkeypatch.setattr(transform_module, "MAX_GATHERED_SAMPLES", test_case.columns_per_block * kernel.frame_length) + + blocked = calculate_cqt_frames(signal, SAMPLE_RATE, HOP_LENGTH) + + assert blocked.shape == whole.shape + assert float(np.abs(blocked - whole).max()) <= ROUNDING_SHARE * float(np.abs(whole).max()) + + class TestToneNormalization: def test_normalized_peak_is_bin_comparable(self) -> None: frequencies = _bin_frequencies() diff --git a/tests/unit/sampletones_core/fft/features/test_cqt.py b/tests/unit/sampletones_core/fft/features/test_cqt.py new file mode 100644 index 000000000..f678982cd --- /dev/null +++ b/tests/unit/sampletones_core/fft/features/test_cqt.py @@ -0,0 +1,40 @@ +from dataclasses import dataclass + +import pytest + +from sampletones_core.constants.enums import SpectrumMethod +from sampletones_core.constants.spectrum import CQT_REFERENCE_CONTEXT_FACTOR +from sampletones_core.fft import Window +from sampletones_core.generators.implementation.noise import NoiseGenerator +from sampletones_core.instructions import NoiseInstruction +from tests.suite.analysis import analyzed_config +from tests.suite.base import BaseTestSuite +from tests.suite.case import BaseRegularTestCase + + +class TestTheReferenceReadsASeamlessStretch(BaseTestSuite): + """A constant-Q reference feature reads a stretch of its sample several windows long. The slowest + long-mode noise outlasts every stored sample and loops with a seam, so the sample holds the whole + stretch the reference reads, at every sample rate.""" + + @dataclass(frozen=True, kw_only=True) + class TestCase(BaseRegularTestCase): + sample_rate: int + + test_cases = ( + TestCase(label="22050 Hz", sample_rate=22050), + TestCase(label="44100 Hz", sample_rate=44100), + TestCase(label="96000 Hz", sample_rate=96000), + ) + + @pytest.mark.parametrize("test_case", test_cases, ids=lambda test_case: test_case.label) + def test_the_slowest_noise_sample_holds_the_reference_context(self, test_case: TestCase) -> None: + config = analyzed_config(SpectrumMethod.CQT, gamma=0) + config = config.model_copy( + update={"library": config.library.model_copy(update={"sample_rate": test_case.sample_rate})} + ) + window = Window.from_config(config) + + sample = NoiseGenerator(config).generate_sample(NoiseInstruction(on=True, period=0, volume=15, short=False)) + + assert len(sample.array) >= CQT_REFERENCE_CONTEXT_FACTOR * window.size diff --git a/tests/unit/sampletones_core/fft/test_axis.py b/tests/unit/sampletones_core/fft/test_axis.py index b94a5553d..6ad2f9981 100644 --- a/tests/unit/sampletones_core/fft/test_axis.py +++ b/tests/unit/sampletones_core/fft/test_axis.py @@ -5,7 +5,8 @@ from sampletones_core.configs import Config from sampletones_core.constants.enums import SpectrumMethod -from sampletones_core.constants.spectrum import BINS_PER_OCTAVE, CQT_CUTOFF_FREQUENCY +from sampletones_core.constants.general import MIN_TRIANGLE_FREQUENCY +from sampletones_core.constants.spectrum import BINS_PER_OCTAVE, CQT_CUTOFF_FREQUENCY, LOG_FFT_CUTOFF_FREQUENCY from sampletones_core.fft import ( FFTTransformer, Window, @@ -13,6 +14,7 @@ calculate_weights_from_edges, ) from sampletones_core.fft.features import get_feature_extractor +from sampletones_core.fft.spectrum.spectrum import calculate_spectrum def _config(method: SpectrumMethod) -> Config: @@ -38,6 +40,28 @@ def test_cqt_window_holds_the_lowest_filter(self) -> None: assert library.window_size == max(library.frame_length, expected) +class TestEachMethodsFloor: + """The constant-Q transform reaches the lowest note the chip sounds, the triangle's an octave below + the pulse's. The log-spaced FFT starts where its window spans two cycles, whatever the constant-Q + floor is.""" + + def test_the_constant_q_axis_reaches_the_triangles_lowest_note(self) -> None: + config = _config(SpectrumMethod.CQT) + window = Window.from_config(config) + + spectrum = calculate_spectrum(SpectrumMethod.CQT, _signal(window), config.library.sample_rate) + + assert float(spectrum.edges[0]) <= MIN_TRIANGLE_FREQUENCY + + def test_the_log_spaced_fft_axis_starts_at_its_own_floor(self) -> None: + config = _config(SpectrumMethod.LOG_SPACED_FFT) + window = Window.from_config(config) + + spectrum = calculate_spectrum(SpectrumMethod.LOG_SPACED_FFT, _signal(window), config.library.sample_rate) + + assert float(spectrum.edges[0]) == pytest.approx(LOG_FFT_CUTOFF_FREQUENCY) + + class TestFragmentAxis: def test_fft_feature_edges_reach_nyquist(self) -> None: config = _config(SpectrumMethod.FFT) diff --git a/tests/unit/sampletones_core/fft/test_instantaneous.py b/tests/unit/sampletones_core/fft/test_instantaneous.py index 0d9d73b3f..888003e0e 100644 --- a/tests/unit/sampletones_core/fft/test_instantaneous.py +++ b/tests/unit/sampletones_core/fft/test_instantaneous.py @@ -4,7 +4,10 @@ import numpy as np import pytest +from sampletones_core.configs import Config from sampletones_core.fft.instantaneous import FundamentalReading, InstantaneousPitch +from sampletones_core.generators.implementation.triangle import TriangleGenerator +from sampletones_core.instructions import TriangleInstruction SAMPLE_RATE: Final[int] = 44100 HOP: Final[int] = 735 @@ -15,6 +18,9 @@ PITCHED_CONFIDENCE: Final[float] = 0.2 UNPITCHED_CONFIDENCE: Final[float] = 0.1 NOISE_SEED: Final[int] = 4 +TRIANGLE_FRAMES: Final[int] = 120 +SETTLED_FRAMES: Final[int] = 30 +TRIANGLE_CENT_TOLERANCE: Final[float] = 20.0 def _harmonic(frequency: float, seed: int = 0) -> np.ndarray: @@ -70,6 +76,29 @@ def test_a_reference_the_transform_never_reaches_is_read_as_nothing(self) -> Non assert reader.at(1, SAMPLE_RATE) is None +class TestTheTrianglesLowestOctave: + """The triangle sounds an octave below the pulse at the same divider, so its lowest notes stand + below every pulse note. The transform reaches them, and their fundamental is read where the + channel sounds it on every frame, the frames a recording's edges cut short aside.""" + + @pytest.mark.parametrize("pitch", (33, 37, 41, 44), ids=lambda pitch: f"pitch {pitch}") + def test_every_settled_frame_is_read_where_the_channel_sounds(self, pitch: int) -> None: + generator = TriangleGenerator(Config()) + instruction = TriangleInstruction(on=True, pitch=pitch) + audio = np.concatenate([generator(instruction, save=True) for _ in range(TRIANGLE_FRAMES)]) + sounding = generator.sounds_at(pitch, 0) + reader = InstantaneousPitch(audio, SAMPLE_RATE, HOP) + + readings = [reader.at(frame, sounding) for frame in range(SETTLED_FRAMES, TRIANGLE_FRAMES - SETTLED_FRAMES)] + + assert all(reading is not None for reading in readings) + assert all( + abs(CENTS_PER_OCTAVE * math.log2(reading.frequency / sounding)) < TRIANGLE_CENT_TOLERANCE + for reading in readings + if reading is not None + ) + + class TestHowMuchOfAFrameStandsBehindItsReading: def test_a_pitched_frame_reads_with_confidence(self) -> None: readings = _readings(_harmonic(A4_FREQUENCY), A4_FREQUENCY) diff --git a/tests/unit/sampletones_core/reconstructions/reconstructor/stems/test_frame.py b/tests/unit/sampletones_core/reconstructions/reconstructor/stems/test_frame.py index 61b2dde8d..f11b53699 100644 --- a/tests/unit/sampletones_core/reconstructions/reconstructor/stems/test_frame.py +++ b/tests/unit/sampletones_core/reconstructions/reconstructor/stems/test_frame.py @@ -778,7 +778,6 @@ def _silent(fragment: Fragment) -> Fragment: return Fragment( audio=np.zeros_like(np.asarray(fragment.audio)), feature=Histogram(edges=fragment.feature.edges, values=np.zeros_like(np.asarray(fragment.feature.values))), - windowed_audio=np.zeros_like(np.asarray(fragment.windowed_audio)), config=fragment.config, ) diff --git a/tests/unit/sampletones_core/reconstructions/reconstructor/stems/test_models.py b/tests/unit/sampletones_core/reconstructions/reconstructor/stems/test_models.py index 5050f6533..dbd9a8c0e 100644 --- a/tests/unit/sampletones_core/reconstructions/reconstructor/stems/test_models.py +++ b/tests/unit/sampletones_core/reconstructions/reconstructor/stems/test_models.py @@ -28,7 +28,6 @@ def _fragment(config: Config) -> Fragment: return Fragment( audio=audio, feature=Histogram(edges=np.array([0.0, 1.0], dtype=np.float32), values=np.zeros(1, dtype=np.float32)), - windowed_audio=audio, config=config, ) From 13ff2655f2b79fbb66a2ad0be3a8a149045d8d7b Mon Sep 17 00:00:00 2001 From: JakimPL Date: Mon, 5 Oct 2026 13:03:34 +0200 Subject: [PATCH 02/12] Muted: a pulse below timer 8, as the sweep unit does --- docs/development/bugs-and-todos.md | 4 +++ docs/glossary.md | 3 +- src/sampletones_core/constants/general.py | 1 + .../generators/implementation/pulse.py | 10 ++++++ .../generators/implementation/test_pulse.py | 33 ++++++++++++++++++- 5 files changed, 49 insertions(+), 2 deletions(-) diff --git a/docs/development/bugs-and-todos.md b/docs/development/bugs-and-todos.md index c82c29567..3889d9342 100644 --- a/docs/development/bugs-and-todos.md +++ b/docs/development/bugs-and-todos.md @@ -101,6 +101,10 @@ currently out of line. An entry leaves when the code meets the contract again. ## Bugs +* A silent triangle renders the middle of its wave, where the console holds the step it stopped on. A + faithful hold needs a DC-blocking output stage at every mix (the reconstruction's render, the sequencer + and the song render), since nothing drains a held level today. +* A triangle bent to divider 1 renders aliased, where the console plays it above hearing. * The Sample column shows no sample on a frame's first rows, since its reading starts over at each frame. It offers no transpose or volume there, while playback applies them to the sample the previous frame left sounding. diff --git a/docs/glossary.md b/docs/glossary.md index 4101ad8c9..203170f02 100644 --- a/docs/glossary.md +++ b/docs/glossary.md @@ -20,7 +20,8 @@ and the classes that implement it. ### Pulse (square) A channel that plays a square wave. Its duty cycle is selectable and it has 15 volume levels. The chip -has two independent pulse channels: `pulse1` and `pulse2`. +has two independent pulse channels: `pulse1` and `pulse2`. The chip silences a pulse whose +[divider](#divider) is below 8, so a bend that goes higher than that at the top notes goes quiet. ### Triangle diff --git a/src/sampletones_core/constants/general.py b/src/sampletones_core/constants/general.py index 4de7d6553..9e76698db 100644 --- a/src/sampletones_core/constants/general.py +++ b/src/sampletones_core/constants/general.py @@ -7,6 +7,7 @@ APU_CLOCK: Final[float] = 1789773.0 TIMER_CYCLE_DIVIDER: Final[int] = 16 MIN_TIMER: Final[int] = 1 +MIN_SOUNDING_PULSE_TIMER: Final[int] = 8 MAX_TIMER: Final[int] = 0x7FF MIN_PITCH: Final[int] = 33 MAX_PITCH: Final[int] = 119 diff --git a/src/sampletones_core/generators/implementation/pulse.py b/src/sampletones_core/generators/implementation/pulse.py index 8103cae42..99364772f 100644 --- a/src/sampletones_core/generators/implementation/pulse.py +++ b/src/sampletones_core/generators/implementation/pulse.py @@ -8,6 +8,7 @@ DUTY_CYCLES, MAX_VOLUME, MIN_PITCH, + MIN_SOUNDING_PULSE_TIMER, MIXER_PULSE, ) from sampletones_core.instructions import InstructionTypeUnion, PulseInstruction @@ -18,6 +19,12 @@ class PulseGenerator(TonalGenerator[PulseInstruction]): + """The square channel, one of the chip's two pulses. + + A timer below ``MIN_SOUNDING_PULSE_TIMER`` renders silence while the timer runs on. The sweep unit + mutes the channel there whatever the sweep's own setting, and the waveform keeps stepping. + """ + def __init__( self, config: Config, @@ -52,6 +59,9 @@ def __call__( self.save_state(save, pulse_instruction) + if self.timer.timer < MIN_SOUNDING_PULSE_TIMER: + return np.zeros(self.frame_length, dtype=np.float32) + return output def apply(self, output: np.ndarray, instruction: PulseInstruction) -> np.ndarray: diff --git a/tests/unit/sampletones_core/generators/implementation/test_pulse.py b/tests/unit/sampletones_core/generators/implementation/test_pulse.py index a879ee71c..cada72f62 100644 --- a/tests/unit/sampletones_core/generators/implementation/test_pulse.py +++ b/tests/unit/sampletones_core/generators/implementation/test_pulse.py @@ -3,7 +3,13 @@ from sampletones_core.configs import Config from sampletones_core.constants.enums import ChannelName, GeneratorClassName -from sampletones_core.constants.general import DUTY_CYCLES, MAX_VOLUME, MIXER_PULSE +from sampletones_core.constants.general import ( + DUTY_CYCLES, + MAX_PITCH, + MAX_VOLUME, + MIN_SOUNDING_PULSE_TIMER, + MIXER_PULSE, +) from sampletones_core.generators.implementation.pulse import PulseGenerator from sampletones_core.instructions import NoiseInstruction, PulseInstruction @@ -49,6 +55,31 @@ def test_on_instruction_result_is_float32(self, generator: PulseGenerator) -> No assert result.dtype == np.float32 +class TestTheSweepUnitMutesShortTimers: + """The sweep unit mutes a pulse whose timer stands below eight, and the waveform keeps stepping.""" + + def _bent_to(self, generator: PulseGenerator, timer: int) -> PulseInstruction: + detune = timer - generator.played_timer_table[MAX_PITCH] + return PulseInstruction(on=True, pitch=MAX_PITCH, volume=MAX_VOLUME, duty_cycle=2, detune=detune) + + def test_a_timer_below_the_bound_is_silent(self, generator: PulseGenerator) -> None: + result = generator(self._bent_to(generator, MIN_SOUNDING_PULSE_TIMER - 1)) + + assert np.all(result == 0.0) + + def test_a_timer_at_the_bound_sounds(self, generator: PulseGenerator) -> None: + result = generator(self._bent_to(generator, MIN_SOUNDING_PULSE_TIMER)) + + assert np.any(result != 0.0) + + def test_the_waveform_steps_on_through_a_muted_frame(self, generator: PulseGenerator) -> None: + phase = generator.timer.phase + + generator(self._bent_to(generator, MIN_SOUNDING_PULSE_TIMER - 1), save=True) + + assert generator.timer.phase != phase + + class TestPulseApply: def test_volume_zero_returns_all_zeros(self, generator: PulseGenerator) -> None: phase = np.array([0.1, 0.3, 0.6, 0.8], dtype=np.float32) From 4ae402c5fcad14aa7cdca0ba0c3671f9a3c2802b Mon Sep 17 00:00:00 2001 From: JakimPL Date: Mon, 5 Oct 2026 14:10:07 +0200 Subject: [PATCH 03/12] Updated: documentations --- docs/concepts/reconstruction.md | 7 ++++ docs/development/bugs-and-todos.md | 12 ++++++ src/sampletones_core/fft/instantaneous.py | 17 ++++++++- .../fft/test_instantaneous.py | 38 +++++++++++++++++-- 4 files changed, 69 insertions(+), 5 deletions(-) diff --git a/docs/concepts/reconstruction.md b/docs/concepts/reconstruction.md index eff67174f..baea5ae34 100644 --- a/docs/concepts/reconstruction.md +++ b/docs/concepts/reconstruction.md @@ -117,6 +117,13 @@ sharper frequency resolution requires a longer time window, and vice versa): triangle's, an octave below the pulse's, since the triangle steps through its wave at half the pulse's rate. A bass line on the triangle is therefore read from its fundamental, which is where the pitch of a bent note is read from too. + This floor is a measured choice. Each bin's wavelet depends on its own frequency alone, + so every bin above the pulse's lowest note is the same at either floor, and so is what + the conversion makes of the music there. The lower floor adds the triangle's lowest + octave. Starting at the pulse's lowest note (54.6 Hz), a pulse took a triangle sliding + from 37 Hz at its third harmonic, and a 37 Hz bass under a melody went unplayed (44.1 kHz + audio at a 60 Hz frame rate, converted onto pulse 1, the triangle and noise). The lower + floor costs about a tenth more conversion time and a larger library. The price is time support: its low-frequency basis functions are long (hundreds of milliseconds), so brief events are smeared in time at the low end. _SampleToNES_ computes the CQT **once over the whole signal** with a hop of one frame, so each diff --git a/docs/development/bugs-and-todos.md b/docs/development/bugs-and-todos.md index 3889d9342..85e0bd066 100644 --- a/docs/development/bugs-and-todos.md +++ b/docs/development/bugs-and-todos.md @@ -52,6 +52,12 @@ dimension the import starts carrying. confidence threshold, a change weight and a window) are chosen by hand, and [the calibration](../tools/calibration.md) could measure them. The change weight trades vibrato against jitter. +* A setting for the lowest frequency the analysis reads. The analysis starts at the triangle's lowest + note, about 27.3 Hz, so every note the chip plays is read from its fundamental, and its longest window + spans over half a second. Music that stays above the triangle's lowest octave would convert about a + tenth faster with the floor an octave higher. In lower music, a pulse then plays those notes at their + third harmonic. A floor below the triangle's lowest note needs longer library samples, since each must + last three windows. ### Technical @@ -105,6 +111,12 @@ currently out of line. An entry leaves when the code meets the contract again. faithful hold needs a DC-blocking output stage at every mix (the reconstruction's render, the sequencer and the song render), since nothing drains a held level today. * A triangle bent to divider 1 renders aliased, where the console plays it above hearing. +* A bend on a high note reads on the wrong side. The pitch reading compares a note's phase one frame + apart, which tells frequencies apart within 30 Hz of the note at a 60 Hz frame rate. From about + 1.7 kHz up that is under 30 cents, so a wider vibrato folds over and its notes stay unbent. +* Another voice pulls the pitch reading. A partial of another voice beside one of a note's harmonics + adds to that harmonic's reading: a bass sliding from 82 Hz under a steady 440 Hz pulse reads about + 9 cents off, and within a cent alone. * The Sample column shows no sample on a frame's first rows, since its reading starts over at each frame. It offers no transpose or volume there, while playback applies them to the sample the previous frame left sounding. diff --git a/src/sampletones_core/fft/instantaneous.py b/src/sampletones_core/fft/instantaneous.py index f8ce8c703..623b4f26e 100644 --- a/src/sampletones_core/fft/instantaneous.py +++ b/src/sampletones_core/fft/instantaneous.py @@ -6,6 +6,7 @@ from sampletones_core.constants.spectrum import BINS_PER_OCTAVE, CQT_CUTOFF_FREQUENCY from .cqt.frequencies import calculate_cqt_frequencies +from .cqt.normalization import normalize_cqt_energy from .cqt.transform import calculate_cqt_frames HARMONIC_COUNT: Final[int] = 5 @@ -39,6 +40,10 @@ class InstantaneousPitch: by the energy standing behind it. The harmonics are read in order, each one settled against the fundamental the ones below it agreed on, which is what keeps an upper harmonic on the right side of the whole turn its phase states the reading to within. + + The confidence measures every bin on the scale the features use, its energy divided by the + length of its wavelet. A bass then takes the share of a frame its level gives it in every + register, and a melody above it keeps the rest. """ def __init__( @@ -58,7 +63,13 @@ def __init__( cutoff, bins_per_octave, ) - self._energies: np.ndarray = np.sum(self._magnitudes**2, axis=0) + self._bin_energies: np.ndarray = normalize_cqt_energy( + self._magnitudes**2, + self._frequencies, + sample_rate, + bins_per_octave, + ) + self._column_energies: np.ndarray = np.sum(self._bin_energies, axis=0) self._turn_per_column: np.ndarray = 2.0 * np.pi * self._frequencies * hop_length / sample_rate self._resolution: float = sample_rate / (2.0 * np.pi * hop_length) self._ambiguity: float = sample_rate / hop_length @@ -85,6 +96,7 @@ def at(self, frame: int, reference: float) -> Optional[FundamentalReading]: weighted = 0.0 weight = 0.0 + share = 0.0 running = reference for harmonic in range(1, HARMONIC_COUNT + 1): bin_index = self._bin_for(reference * harmonic) @@ -95,6 +107,7 @@ def at(self, frame: int, reference: float) -> Optional[FundamentalReading]: energy = float(self._magnitudes[bin_index, opening]) ** 2 weighted += energy * partial / harmonic weight += energy + share += float(self._bin_energies[bin_index, opening]) running = weighted / weight if weight <= 0.0: @@ -102,7 +115,7 @@ def at(self, frame: int, reference: float) -> Optional[FundamentalReading]: return FundamentalReading( frequency=weighted / weight, - confidence=weight / max(float(self._energies[opening]), MINIMUM_COLUMN_ENERGY), + confidence=share / max(float(self._column_energies[opening]), MINIMUM_COLUMN_ENERGY), ) def _pair(self, frame: int) -> tuple[Optional[int], Optional[int]]: diff --git a/tests/unit/sampletones_core/fft/test_instantaneous.py b/tests/unit/sampletones_core/fft/test_instantaneous.py index 888003e0e..effd8084d 100644 --- a/tests/unit/sampletones_core/fft/test_instantaneous.py +++ b/tests/unit/sampletones_core/fft/test_instantaneous.py @@ -21,12 +21,16 @@ TRIANGLE_FRAMES: Final[int] = 120 SETTLED_FRAMES: Final[int] = 30 TRIANGLE_CENT_TOLERANCE: Final[float] = 20.0 +BASS_SECONDS: Final[float] = 2.0 +BASS_LEVEL: Final[float] = 0.3 +LOW_BASS_FREQUENCY: Final[float] = 36.71 +SHARE_TOLERANCE: Final[float] = 0.1 -def _harmonic(frequency: float, seed: int = 0) -> np.ndarray: +def _harmonic(frequency: float, seed: int = 0, seconds: float = SECONDS) -> np.ndarray: """A steady tone with a full harmonic series, which is what a pitched frame looks like.""" generator = np.random.default_rng(seed) - count = int(SAMPLE_RATE * SECONDS) + count = int(SAMPLE_RATE * seconds) time = np.arange(count) / SAMPLE_RATE audio = np.zeros(count) for harmonic in range(1, 20): @@ -35,7 +39,22 @@ def _harmonic(frequency: float, seed: int = 0) -> np.ndarray: audio += np.sin(2 * np.pi * frequency * harmonic * time + generator.uniform(0, 2 * np.pi)) / harmonic - return (audio / np.abs(audio).max() * 0.3).astype(np.float32) + scaled: np.ndarray = (audio / np.abs(audio).max() * 0.3).astype(np.float32) + return scaled + + +def _bass(frequency: float, level: float) -> np.ndarray: + """A pure low tone lasting long enough for the lowest bins to settle.""" + time = np.arange(int(SAMPLE_RATE * BASS_SECONDS)) / SAMPLE_RATE + audio: np.ndarray = (level * np.sin(2 * np.pi * frequency * time)).astype(np.float32) + return audio + + +def _melody_share(bass: np.ndarray) -> float: + """The median confidence of a melody note read over a bass.""" + melody = _harmonic(A4_FREQUENCY, seconds=BASS_SECONDS) + readings = _readings(melody + bass, A4_FREQUENCY) + return float(np.median([reading.confidence for reading in readings])) def _readings(audio: np.ndarray, reference: float) -> List[FundamentalReading]: @@ -124,3 +143,16 @@ def test_a_tone_sharing_the_frame_lowers_the_share_without_moving_the_reading(se assert float(np.median([reading.confidence for reading in readings])) < float( np.median([reading.confidence for reading in _readings(alone, A4_FREQUENCY)]) ) + + def test_a_bass_takes_the_same_share_in_every_register(self) -> None: + """A bass an octave lower is no louder, so a melody over it keeps the share it holds.""" + low = _melody_share(_bass(LOW_BASS_FREQUENCY, BASS_LEVEL)) + higher = _melody_share(_bass(2 * LOW_BASS_FREQUENCY, BASS_LEVEL)) + + assert abs(low - higher) < SHARE_TOLERANCE * higher + + def test_a_louder_bass_takes_a_larger_share(self) -> None: + quiet = _melody_share(_bass(LOW_BASS_FREQUENCY, BASS_LEVEL)) + loud = _melody_share(_bass(LOW_BASS_FREQUENCY, 2 * BASS_LEVEL)) + + assert loud < quiet From 851e9a4f91e734c09cd53b8de6eb1e3ba9a9771c Mon Sep 17 00:00:00 2001 From: JakimPL Date: Mon, 5 Oct 2026 14:51:42 +0200 Subject: [PATCH 04/12] Restated: the bend confidence levels on the features' scale --- docs/concepts/reconstruction.md | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/docs/concepts/reconstruction.md b/docs/concepts/reconstruction.md index baea5ae34..dceb2d594 100644 --- a/docs/concepts/reconstruction.md +++ b/docs/concepts/reconstruction.md @@ -337,9 +337,10 @@ fundamental the harmonics below it agreed on. This places the note **within a te whole range. The reading also says how much of the frame stands behind it: the share of the column's energy its -harmonics hold. A pitched frame reads around 0.5, a frame sharing the channel with another tone around -0.3, and noise around 0.04. One threshold therefore separates the frames worth bending from the frames -with no pitch to read. +harmonics hold, with every bin measured on the scale the features use. On that scale a bass takes the +share its level gives it in every register, so a melody over a low bass keeps its reading. A pitched frame +reads around 0.6, a frame sharing the channel with another tone around 0.3, and noise around 0.02. One +threshold therefore separates the frames worth bending from the frames with no pitch to read. ### 6.2 Landing the note, and holding it From 9f9a3f842d38b3c16d1466f7cabcd818e6eb940b Mon Sep 17 00:00:00 2001 From: JakimPL Date: Mon, 5 Oct 2026 15:09:13 +0200 Subject: [PATCH 05/12] Rested: a triangle below divider 2 at the middle of its wave --- docs/development/bugs-and-todos.md | 1 - docs/glossary.md | 3 +- src/sampletones_core/constants/general.py | 1 + .../generators/implementation/triangle.py | 10 ++++++ .../implementation/test_triangle.py | 32 ++++++++++++++++++- 5 files changed, 44 insertions(+), 3 deletions(-) diff --git a/docs/development/bugs-and-todos.md b/docs/development/bugs-and-todos.md index 85e0bd066..5179e2d77 100644 --- a/docs/development/bugs-and-todos.md +++ b/docs/development/bugs-and-todos.md @@ -110,7 +110,6 @@ currently out of line. An entry leaves when the code meets the contract again. * A silent triangle renders the middle of its wave, where the console holds the step it stopped on. A faithful hold needs a DC-blocking output stage at every mix (the reconstruction's render, the sequencer and the song render), since nothing drains a held level today. -* A triangle bent to divider 1 renders aliased, where the console plays it above hearing. * A bend on a high note reads on the wrong side. The pitch reading compares a note's phase one frame apart, which tells frequencies apart within 30 Hz of the note at a 60 Hz frame rate. From about 1.7 kHz up that is under 30 cents, so a wider vibrato folds over and its notes stay unbent. diff --git a/docs/glossary.md b/docs/glossary.md index 203170f02..a12fc7566 100644 --- a/docs/glossary.md +++ b/docs/glossary.md @@ -29,7 +29,8 @@ A channel that plays a triangle wave of fixed shape and volume. Only its pitch v the APU clock by 32 where the pulse timers divide by 16, and all three read the same period table. A triangle note therefore sounds an octave below the pulse note with the same period, so a triangle instruction of pitch P sounds at pitch P−12. FamiTracker uses the same convention, so an exported note -plays at the pitch _SampleToNES_ played it. +plays at the pitch _SampleToNES_ played it. Below [divider](#divider) 2 the chip's triangle steps above +hearing, and its output rests at the middle of the wave. ### Noise diff --git a/src/sampletones_core/constants/general.py b/src/sampletones_core/constants/general.py index 9e76698db..f8864aa0a 100644 --- a/src/sampletones_core/constants/general.py +++ b/src/sampletones_core/constants/general.py @@ -8,6 +8,7 @@ TIMER_CYCLE_DIVIDER: Final[int] = 16 MIN_TIMER: Final[int] = 1 MIN_SOUNDING_PULSE_TIMER: Final[int] = 8 +MIN_SOUNDING_TRIANGLE_TIMER: Final[int] = 2 MAX_TIMER: Final[int] = 0x7FF MIN_PITCH: Final[int] = 33 MAX_PITCH: Final[int] = 119 diff --git a/src/sampletones_core/generators/implementation/triangle.py b/src/sampletones_core/generators/implementation/triangle.py index ade6fae75..38d5f253d 100644 --- a/src/sampletones_core/generators/implementation/triangle.py +++ b/src/sampletones_core/generators/implementation/triangle.py @@ -6,6 +6,7 @@ from sampletones_core.constants.enums import ChannelName, GeneratorClassName from sampletones_core.constants.general import ( MIN_PITCH, + MIN_SOUNDING_TRIANGLE_TIMER, MIXER_TRIANGLE, TRIANGLE_OFFSET, TRIANGLE_PHASE_INCREMENT, @@ -21,6 +22,12 @@ class TriangleGenerator(TonalGenerator[TriangleInstruction]): + """The triangle channel, a fixed-volume wave an octave below a pulse at the same divider. + + A timer below ``MIN_SOUNDING_TRIANGLE_TIMER`` renders the middle of the wave while the timer runs + on. The sequencer steps above hearing there, and the console's output settles at its mean level. + """ + def __init__( self, config: Config, @@ -55,6 +62,9 @@ def __call__( self.save_state(save, triangle_instruction) + if self.timer.timer < MIN_SOUNDING_TRIANGLE_TIMER: + return np.zeros(self.frame_length, dtype=np.float32) + return output def apply(self, output: np.ndarray, instruction: TriangleInstruction) -> np.ndarray: diff --git a/tests/unit/sampletones_core/generators/implementation/test_triangle.py b/tests/unit/sampletones_core/generators/implementation/test_triangle.py index e51fb15f0..466bdefb2 100644 --- a/tests/unit/sampletones_core/generators/implementation/test_triangle.py +++ b/tests/unit/sampletones_core/generators/implementation/test_triangle.py @@ -3,7 +3,11 @@ from sampletones_core.configs import Config from sampletones_core.constants.enums import ChannelName, GeneratorClassName -from sampletones_core.constants.general import MIXER_TRIANGLE +from sampletones_core.constants.general import ( + MAX_PITCH, + MIN_SOUNDING_TRIANGLE_TIMER, + MIXER_TRIANGLE, +) from sampletones_core.generators.implementation.triangle import TriangleGenerator from sampletones_core.instructions import PulseInstruction, TriangleInstruction @@ -49,6 +53,32 @@ def test_on_instruction_result_is_float32(self, generator: TriangleGenerator) -> assert result.dtype == np.float32 +class TestAnUltrasonicTriangleRests: + """Below divider 2 the triangle steps above hearing, and the console's output rests at the middle of + its wave while the sequencer keeps stepping.""" + + def _bent_to(self, generator: TriangleGenerator, timer: int) -> TriangleInstruction: + detune = timer - generator.played_timer_table[MAX_PITCH] + return TriangleInstruction(on=True, pitch=MAX_PITCH, detune=detune) + + def test_a_timer_below_the_bound_rests(self, generator: TriangleGenerator) -> None: + result = generator(self._bent_to(generator, MIN_SOUNDING_TRIANGLE_TIMER - 1)) + + assert np.all(result == 0.0) + + def test_a_timer_at_the_bound_sounds(self, generator: TriangleGenerator) -> None: + result = generator(self._bent_to(generator, MIN_SOUNDING_TRIANGLE_TIMER)) + + assert np.any(result != 0.0) + + def test_the_wave_steps_on_through_a_resting_frame(self, generator: TriangleGenerator) -> None: + phase = generator.timer.phase + + generator(self._bent_to(generator, MIN_SOUNDING_TRIANGLE_TIMER - 1), save=True) + + assert generator.timer.phase != phase + + class TestTriangleApply: def test_output_bounded_within_mixer_level(self, generator: TriangleGenerator) -> None: phase = np.linspace(0.0, 1.0, 100, endpoint=False, dtype=np.float32) From 62727b9e0880ce0117c63d5097d1a3ef7fc85a8f Mon Sep 17 00:00:00 2001 From: JakimPL Date: Mon, 5 Oct 2026 15:18:43 +0200 Subject: [PATCH 06/12] Left out: a harmonic another voice sounds a bin away --- docs/concepts/reconstruction.md | 5 ++-- docs/development/bugs-and-todos.md | 6 ++-- src/sampletones_core/fft/instantaneous.py | 12 ++++++-- .../fft/test_instantaneous.py | 30 +++++++++++++++++++ 4 files changed, 46 insertions(+), 7 deletions(-) diff --git a/docs/concepts/reconstruction.md b/docs/concepts/reconstruction.md index dceb2d594..214da8e71 100644 --- a/docs/concepts/reconstruction.md +++ b/docs/concepts/reconstruction.md @@ -333,8 +333,9 @@ and the spectrum discards. A partial standing between two bin centers still adva rate. Comparing that advance across two columns against the rate the bin itself turns at gives the partial's frequency far more finely than the bins are spaced. The reading takes the first few harmonics of the note the decoder chose. It weights each by the energy behind it and settles each against the -fundamental the harmonics below it agreed on. This places the note **within a tenth of a cent** across the -whole range. +fundamental the harmonics below it agreed on. A harmonic counts where its partial stands within half a +semitone of where that fundamental puts it, the room a note owns, so a partial another voice sounds a bin +away stays out of the reading. This places the note **within a tenth of a cent** across the whole range. The reading also says how much of the frame stands behind it: the share of the column's energy its harmonics hold, with every bin measured on the scale the features use. On that scale a bass takes the diff --git a/docs/development/bugs-and-todos.md b/docs/development/bugs-and-todos.md index 5179e2d77..5b2216e59 100644 --- a/docs/development/bugs-and-todos.md +++ b/docs/development/bugs-and-todos.md @@ -113,9 +113,9 @@ currently out of line. An entry leaves when the code meets the contract again. * A bend on a high note reads on the wrong side. The pitch reading compares a note's phase one frame apart, which tells frequencies apart within 30 Hz of the note at a 60 Hz frame rate. From about 1.7 kHz up that is under 30 cents, so a wider vibrato folds over and its notes stay unbent. -* Another voice pulls the pitch reading. A partial of another voice beside one of a note's harmonics - adds to that harmonic's reading: a bass sliding from 82 Hz under a steady 440 Hz pulse reads about - 9 cents off, and within a cent alone. +* Another voice inside a harmonic's bin pulls the pitch reading. A partial of another voice within half a + semitone of one of a note's harmonics shares that harmonic's bin and adds to its reading: a 65 Hz bass + under a steady 330 Hz pulse reads about 4 cents off, and within a cent alone. * The Sample column shows no sample on a frame's first rows, since its reading starts over at each frame. It offers no transpose or volume there, while playback applies them to the sample the previous frame left sounding. diff --git a/src/sampletones_core/fft/instantaneous.py b/src/sampletones_core/fft/instantaneous.py index 623b4f26e..66ac0e4b2 100644 --- a/src/sampletones_core/fft/instantaneous.py +++ b/src/sampletones_core/fft/instantaneous.py @@ -4,12 +4,14 @@ import numpy as np from sampletones_core.constants.spectrum import BINS_PER_OCTAVE, CQT_CUTOFF_FREQUENCY +from sampletones_shared.constants.music import OCTAVE_SEMITONES from .cqt.frequencies import calculate_cqt_frequencies from .cqt.normalization import normalize_cqt_energy from .cqt.transform import calculate_cqt_frames HARMONIC_COUNT: Final[int] = 5 +HARMONIC_ROOM: Final[float] = 2.0 ** (1.0 / (2 * OCTAVE_SEMITONES)) MINIMUM_COLUMN_ENERGY: Final[float] = 1e-20 @@ -39,7 +41,9 @@ class InstantaneousPitch: bins carry the note's harmonics, and each harmonic's estimate is divided back down and weighted by the energy standing behind it. The harmonics are read in order, each one settled against the fundamental the ones below it agreed on, which is what keeps an upper harmonic on the right side - of the whole turn its phase states the reading to within. + of the whole turn its phase states the reading to within. A later harmonic counts where its partial + stands within half a semitone of where that fundamental puts it, the room a note owns, so a + partial another voice sounds beside it stays out of the reading. The confidence measures every bin on the scale the features use, its energy divided by the length of its wavelet. A bass then takes the share of a frame its level gives it in every @@ -103,7 +107,11 @@ def at(self, frame: int, reference: float) -> Optional[FundamentalReading]: if bin_index is None: continue - partial = self._partial_frequency(bin_index, opening, closing, running * harmonic) + expected = running * harmonic + partial = self._partial_frequency(bin_index, opening, closing, expected) + if weight > 0.0 and not expected / HARMONIC_ROOM <= partial <= expected * HARMONIC_ROOM: + continue + energy = float(self._magnitudes[bin_index, opening]) ** 2 weighted += energy * partial / harmonic weight += energy diff --git a/tests/unit/sampletones_core/fft/test_instantaneous.py b/tests/unit/sampletones_core/fft/test_instantaneous.py index effd8084d..51f1edd1f 100644 --- a/tests/unit/sampletones_core/fft/test_instantaneous.py +++ b/tests/unit/sampletones_core/fft/test_instantaneous.py @@ -25,6 +25,9 @@ BASS_LEVEL: Final[float] = 0.3 LOW_BASS_FREQUENCY: Final[float] = 36.71 SHARE_TOLERANCE: Final[float] = 0.1 +BASS_FREQUENCY: Final[float] = 65.41 +BESIDE_FIFTH_HARMONIC: Final[float] = 349.23 +ON_FOURTH_HARMONIC: Final[float] = 4 * BASS_FREQUENCY def _harmonic(frequency: float, seed: int = 0, seconds: float = SECONDS) -> np.ndarray: @@ -50,6 +53,20 @@ def _bass(frequency: float, level: float) -> np.ndarray: return audio +def _triangle(frequency: float) -> np.ndarray: + """A triangle wave, whose odd harmonics leave every other harmonic bin to the voices around it.""" + phase = frequency * np.arange(int(SAMPLE_RATE * BASS_SECONDS)) / SAMPLE_RATE + audio: np.ndarray = (BASS_LEVEL * (1.0 - 4.0 * np.abs((phase % 1.0) - 0.5))).astype(np.float32) + return audio + + +def _bass_cents_under(melody_frequency: float) -> float: + """Where a triangle bass is read, in cents from where it stands, under a melody note.""" + melody = _harmonic(melody_frequency, seconds=BASS_SECONDS) + readings = _readings(_triangle(BASS_FREQUENCY) + melody, BASS_FREQUENCY) + return _median_cents(readings, BASS_FREQUENCY) + + def _melody_share(bass: np.ndarray) -> float: """The median confidence of a melody note read over a bass.""" melody = _harmonic(A4_FREQUENCY, seconds=BASS_SECONDS) @@ -118,6 +135,19 @@ def test_every_settled_frame_is_read_where_the_channel_sounds(self, pitch: int) ) +class TestAnotherVoiceBesideTheHarmonics: + """A harmonic reads where its partial stands, so a partial another voice sounds a bin away from one + of a note's harmonics would pull the note's reading toward it.""" + + def test_a_voice_beside_a_harmonic_leaves_the_reading_in_place(self) -> None: + assert abs(_bass_cents_under(BESIDE_FIFTH_HARMONIC)) < CENT_TOLERANCE + + def test_a_voice_on_a_harmonic_agrees_with_the_note(self) -> None: + """The triangle has no fourth harmonic, so the melody's fundamental fills that bin alone, at the + frequency the note puts it.""" + assert abs(_bass_cents_under(ON_FOURTH_HARMONIC)) < CENT_TOLERANCE + + class TestHowMuchOfAFrameStandsBehindItsReading: def test_a_pitched_frame_reads_with_confidence(self) -> None: readings = _readings(_harmonic(A4_FREQUENCY), A4_FREQUENCY) From b4c450cb9f2d1d28d07b6ffb2d7848974c29d730 Mon Sep 17 00:00:00 2001 From: JakimPL Date: Mon, 5 Oct 2026 17:07:42 +0200 Subject: [PATCH 07/12] Guided: a high note's reading by a short-lag phase --- docs/concepts/reconstruction.md | 10 ++- docs/development/bugs-and-todos.md | 3 - src/sampletones_core/fft/instantaneous.py | 66 +++++++++++++++++++ .../fft/test_instantaneous.py | 58 ++++++++++++++++ 4 files changed, 131 insertions(+), 6 deletions(-) diff --git a/docs/concepts/reconstruction.md b/docs/concepts/reconstruction.md index 214da8e71..41aef0574 100644 --- a/docs/concepts/reconstruction.md +++ b/docs/concepts/reconstruction.md @@ -335,7 +335,11 @@ partial's frequency far more finely than the bins are spaced. The reading takes the note the decoder chose. It weights each by the energy behind it and settles each against the fundamental the harmonics below it agreed on. A harmonic counts where its partial stands within half a semitone of where that fundamental puts it, the room a note owns, so a partial another voice sounds a bin -away stays out of the reading. This places the note **within a tenth of a cent** across the whole range. +away stays out of the reading. The advance across two columns repeats every `sample_rate / hop` hertz +(60 Hz at the defaults), which from about 1 kHz up is narrower than a note's room. There a second reading, +of the advance over a sixteenth of a hop, names the repeat the first harmonic stands on, and the advance +across two columns keeps its precision inside it. This places the note **within a tenth of a cent** across +the whole range. The reading also says how much of the frame stands behind it: the share of the column's energy its harmonics hold, with every bin measured on the scale the features use. On that scale a bass takes the @@ -361,8 +365,8 @@ dividers. ### 6.3 What it costs, and what it leaves alone The refinement enumerates no candidate and rescores nothing. It leaves the library, the per-frame matching -and the decoder's lattice exactly as they were. It adds one transform per recording and a small walk per -channel. +and the decoder's lattice exactly as they were. It adds one transform per recording, two short ones over +the bins from about 1 kHz up for the second reading, and a small walk per channel. The transform's cost depends on the machine. On a CUDA build it is too small to measure. On a CPU build it is a tenth or more of a short conversion, because the reading needs a handful of bins per frame and the diff --git a/docs/development/bugs-and-todos.md b/docs/development/bugs-and-todos.md index 5b2216e59..25ca6b08d 100644 --- a/docs/development/bugs-and-todos.md +++ b/docs/development/bugs-and-todos.md @@ -110,9 +110,6 @@ currently out of line. An entry leaves when the code meets the contract again. * A silent triangle renders the middle of its wave, where the console holds the step it stopped on. A faithful hold needs a DC-blocking output stage at every mix (the reconstruction's render, the sequencer and the song render), since nothing drains a held level today. -* A bend on a high note reads on the wrong side. The pitch reading compares a note's phase one frame - apart, which tells frequencies apart within 30 Hz of the note at a 60 Hz frame rate. From about - 1.7 kHz up that is under 30 cents, so a wider vibrato folds over and its notes stay unbent. * Another voice inside a harmonic's bin pulls the pitch reading. A partial of another voice within half a semitone of one of a note's harmonics shares that harmonic's bin and adds to its reading: a 65 Hz bass under a steady 330 Hz pulse reads about 4 cents off, and within a cent alone. diff --git a/src/sampletones_core/fft/instantaneous.py b/src/sampletones_core/fft/instantaneous.py index 66ac0e4b2..9d74c6866 100644 --- a/src/sampletones_core/fft/instantaneous.py +++ b/src/sampletones_core/fft/instantaneous.py @@ -12,7 +12,9 @@ HARMONIC_COUNT: Final[int] = 5 HARMONIC_ROOM: Final[float] = 2.0 ** (1.0 / (2 * OCTAVE_SEMITONES)) +NEIGHBOR_NOTE: Final[float] = 2.0 ** (1.0 / OCTAVE_SEMITONES) MINIMUM_COLUMN_ENERGY: Final[float] = 1e-20 +SHORT_LAG_DIVISOR: Final[int] = 16 @dataclass(frozen=True) @@ -45,6 +47,10 @@ class InstantaneousPitch: stands within half a semitone of where that fundamental puts it, the room a note owns, so a partial another voice sounds beside it stays out of the reading. + The first harmonic read has the note alone to be settled against. From about 1 kHz up a note's + room is wider than the whole turn the reading repeats over, so there a second reading, over a lag + a fraction of a hop long, names the turn the partial stands on. + The confidence measures every bin on the scale the features use, its energy divided by the length of its wavelet. A bass then takes the share of a frame its level gives it in every register, and a melody above it keeps the rest. @@ -77,6 +83,8 @@ def __init__( self._turn_per_column: np.ndarray = 2.0 * np.pi * self._frequencies * hop_length / sample_rate self._resolution: float = sample_rate / (2.0 * np.pi * hop_length) self._ambiguity: float = sample_rate / hop_length + self._short_lag_first_bin: int = int(np.searchsorted(self._frequencies, self._fold_frequency())) + self._short_lag_partials: np.ndarray = self._read_short_lag(audio, sample_rate, hop_length, bins_per_octave) @property def columns(self) -> int: @@ -108,6 +116,9 @@ def at(self, frame: int, reference: float) -> Optional[FundamentalReading]: continue expected = running * harmonic + if weight <= 0.0: + expected = self._first_guide(bin_index, opening, expected) + partial = self._partial_frequency(bin_index, opening, closing, expected) if weight > 0.0 and not expected / HARMONIC_ROOM <= partial <= expected * HARMONIC_ROOM: continue @@ -146,6 +157,61 @@ def _bin_for(self, frequency: float) -> Optional[int]: return int(np.argmin(np.abs(self._frequencies - frequency))) + def _fold_frequency(self) -> float: + """The frequency from which half the span a reading repeats over is narrower than a note's room.""" + return self._ambiguity / 2.0 / (HARMONIC_ROOM - 1.0) + + def _read_short_lag( + self, + audio: np.ndarray, + sample_rate: int, + hop_length: int, + bins_per_octave: int, + ) -> np.ndarray: + """Each bin's partial from the fold frequency up, read over a short lag at the middle of every hop. + + The lag is ``hop / SHORT_LAG_DIVISOR`` long, so this reading repeats that many times wider than + the one-hop reading and spans a note's room up to the top of the range. It sits at the middle of + the hop the one-hop reading spans, so a moving pitch shows in both alike. Both of its columns come + from one transform, so the kernel's centering cancels out of the phase they turn. + + Returns: + np.ndarray: One frequency in Hz per bin from ``_short_lag_first_bin`` up and per hop. + """ + first = self._short_lag_first_bin + if first >= len(self._frequencies): + return np.empty((0, 0)) + + lag = hop_length // SHORT_LAG_DIVISOR + middle = hop_length // 2 + cutoff = float(self._frequencies[first]) + n_bins = len(self._frequencies) - first + opening = calculate_cqt_frames(audio[middle:], sample_rate, hop_length, cutoff, n_bins, bins_per_octave) + lagged = calculate_cqt_frames(audio[middle + lag :], sample_rate, hop_length, cutoff, n_bins, bins_per_octave) + count = min(opening.shape[1], lagged.shape[1]) + frequencies = self._frequencies[first:, None] + turned = np.angle(lagged[:, :count]) - np.angle(opening[:, :count]) + deviation = np.angle(np.exp(1j * (turned - 2.0 * np.pi * frequencies * lag / sample_rate))) + partials: np.ndarray = frequencies + deviation * sample_rate / (2.0 * np.pi * lag) + return partials + + def _first_guide(self, bin_index: int, column: int, expected: float) -> float: + """Where the first harmonic a frame reads is looked for. + + The short-lag reading guides it where the bin has one and that reading stands between the + note's neighbors. The decoder chose the note nearest the sound, so the partial lies there, even + where the note's own divider stands off equal temperament. The note itself guides it elsewhere. + """ + row = bin_index - self._short_lag_first_bin + if row < 0 or column >= self._short_lag_partials.shape[1]: + return expected + + guide = float(self._short_lag_partials[row, column]) + if expected / NEIGHBOR_NOTE < guide < expected * NEIGHBOR_NOTE: + return guide + + return expected + def _partial_frequency( self, bin_index: int, diff --git a/tests/unit/sampletones_core/fft/test_instantaneous.py b/tests/unit/sampletones_core/fft/test_instantaneous.py index 51f1edd1f..041ca1ddf 100644 --- a/tests/unit/sampletones_core/fft/test_instantaneous.py +++ b/tests/unit/sampletones_core/fft/test_instantaneous.py @@ -25,6 +25,15 @@ BASS_LEVEL: Final[float] = 0.3 LOW_BASS_FREQUENCY: Final[float] = 36.71 SHARE_TOLERANCE: Final[float] = 0.1 +HIGH_FREQUENCIES: Final[tuple[float, ...]] = (1760.0, 3520.0) +FAR_INSIDE_CENTS: Final[tuple[float, ...]] = (-40.0, 40.0) +PULSE_A6_DIVIDER_FREQUENCY: Final[float] = 1747.8 +PAST_HALF_A_SEMITONE_CENTS: Final[float] = 52.0 +VIBRATO_FREQUENCY: Final[float] = 1760.0 +VIBRATO_DEPTH_CENTS: Final[float] = 35.0 +VIBRATO_RATE: Final[float] = 5.5 +VIBRATO_SECONDS: Final[float] = 2.0 +VIBRATO_CENT_TOLERANCE: Final[float] = 10.0 BASS_FREQUENCY: Final[float] = 65.41 BESIDE_FIFTH_HARMONIC: Final[float] = 349.23 ON_FOURTH_HARMONIC: Final[float] = 4 * BASS_FREQUENCY @@ -112,6 +121,55 @@ def test_a_reference_the_transform_never_reaches_is_read_as_nothing(self) -> Non assert reader.at(1, SAMPLE_RATE) is None +def _vibrato() -> tuple[np.ndarray, np.ndarray]: + """A square-wave vibrato and the frequency it sounds at between each pair of columns.""" + time = np.arange(int(SAMPLE_RATE * VIBRATO_SECONDS)) / SAMPLE_RATE + depth = VIBRATO_DEPTH_CENTS / CENTS_PER_OCTAVE + frequency = VIBRATO_FREQUENCY * 2 ** (depth * np.sin(2 * np.pi * VIBRATO_RATE * time)) + phase = np.cumsum(frequency) / SAMPLE_RATE + audio = (0.3 * np.where(phase % 1.0 < 0.5, 1.0, -1.0)).astype(np.float32) + frames = len(time) // HOP + sounding = frequency[: frames * HOP].reshape(frames, HOP).mean(axis=1) + return audio, sounding + + +class TestTheTopOfTheRange: + """A reading repeats every ``sample_rate / hop`` hertz, a span narrower than a note's room from about + 1 kHz up. A tone far inside its room there is read on its own side.""" + + @pytest.mark.parametrize("frequency", HIGH_FREQUENCIES, ids=lambda frequency: f"{frequency:g} Hz") + @pytest.mark.parametrize("cents", FAR_INSIDE_CENTS, ids=lambda cents: f"{cents:+.0f}c") + def test_a_tone_far_inside_its_room_is_read_where_it_stands(self, frequency: float, cents: float) -> None: + truth = frequency * 2 ** (cents / CENTS_PER_OCTAVE) + + readings = _readings(_harmonic(truth), frequency) + + assert abs(_median_cents(readings, frequency) - cents) < CENT_TOLERANCE + + def test_a_tone_nearest_a_note_whose_divider_stands_flat_is_read_where_it_stands(self) -> None: + """The pulse's A6 divider sounds 12 cents flat, so a tone 40 cents sharp of A6 stands 52 cents + from the note the decoder names, which is still the nearest one.""" + cents = PAST_HALF_A_SEMITONE_CENTS + truth = PULSE_A6_DIVIDER_FREQUENCY * 2 ** (cents / CENTS_PER_OCTAVE) + + readings = _readings(_harmonic(truth), PULSE_A6_DIVIDER_FREQUENCY) + + assert abs(_median_cents(readings, PULSE_A6_DIVIDER_FREQUENCY) - cents) < CENT_TOLERANCE + + def test_a_vibrato_is_read_on_its_own_side_at_every_frame(self) -> None: + audio, sounding = _vibrato() + reader = InstantaneousPitch(audio, SAMPLE_RATE, HOP) + + errors = [ + abs(CENTS_PER_OCTAVE * math.log2(reading.frequency / sounding[frame])) + for frame in range(SETTLED_FRAMES, len(sounding) - SETTLED_FRAMES) + if (reading := reader.at(frame, VIBRATO_FREQUENCY)) is not None + ] + + assert errors + assert max(errors) < VIBRATO_CENT_TOLERANCE + + class TestTheTrianglesLowestOctave: """The triangle sounds an octave below the pulse at the same divider, so its lowest notes stand below every pulse note. The transform reaches them, and their fundamental is read where the From 957366aac697b25d16d69f64bdf9835770cc446e Mon Sep 17 00:00:00 2001 From: JakimPL Date: Mon, 5 Oct 2026 18:34:38 +0200 Subject: [PATCH 08/12] Settled: each note's bend on its own --- docs/concepts/reconstruction.md | 3 +- .../reconstructor/refinement/refiner.py | 39 +++++++++++--- .../reconstructor/refinement/test_refiner.py | 53 ++++++++++++++++++- 3 files changed, 86 insertions(+), 9 deletions(-) diff --git a/docs/concepts/reconstruction.md b/docs/concepts/reconstruction.md index 41aef0574..a89c93c27 100644 --- a/docs/concepts/reconstruction.md +++ b/docs/concepts/reconstruction.md @@ -360,7 +360,8 @@ chases. The per-frame proposals are therefore settled by a change-penalized walk Viterbi decoder uses to settle a note contour. The cost of a bend is how far it is from that frame's reading, plus a toll on changing at all. The states a frame may take are the bends its neighborhood proposed, together with no bend. That keeps the walk to a handful of states even where a note owns tens of -dividers. +dividers. A bend counts divider steps from its own note, so each note's frames are settled on their own: a +frame that reads nothing keeps a bend its own note read, and a new note starts from its own reading. ### 6.3 What it costs, and what it leaves alone diff --git a/src/sampletones_core/reconstructions/reconstructor/refinement/refiner.py b/src/sampletones_core/reconstructions/reconstructor/refinement/refiner.py index 09e224f66..99b8ba9a8 100644 --- a/src/sampletones_core/reconstructions/reconstructor/refinement/refiner.py +++ b/src/sampletones_core/reconstructions/reconstructor/refinement/refiner.py @@ -41,7 +41,8 @@ class PitchRefiner: A frame whose sound is not pitched enough for that reading to mean anything makes no proposal, and the run of bends is then settled against a toll on changing, so a stream holds a tuning - rather than chasing one. + rather than chasing one. A bend counts divider steps from its own note, so each note's frames are + settled on their own, and a frame that reads nothing keeps a bend its own note read. Which recordings are carried, and on which channels, each stem entry states for itself, so a channel a stem leaves alone keeps the note the matching chose. @@ -124,11 +125,16 @@ def _refined( self._proposal(generator, channel_name, candidate, frame, stem_ids, readers) for frame, candidate in enumerate(stream) ] - bends = smoothed( - proposals, - window=settings.window, - change_weight=settings.change_weight, - ) + bends: List[int] = [] + for run in _note_runs(stream): + bends.extend( + smoothed( + [proposals[frame] for frame in run], + window=settings.window, + change_weight=settings.change_weight, + ) + ) + return [self._bent(candidate, bend) for candidate, bend in zip(stream, bends)] def _proposal( @@ -180,3 +186,24 @@ def _bent(candidate: ScoredCandidate, bend: int) -> ScoredCandidate: return candidate return replace(candidate, instruction=instruction.model_copy(update={"detune": bend})) + + +def _note_runs(stream: List[ScoredCandidate]) -> List[range]: + """The stream's frames cut wherever the note they sound changes, a rest standing as a note of its own.""" + runs: List[range] = [] + start = 0 + for frame in range(1, len(stream) + 1): + if frame == len(stream) or _note(stream[frame]) != _note(stream[start]): + runs.append(range(start, frame)) + start = frame + + return runs + + +def _note(candidate: ScoredCandidate) -> Optional[int]: + """The note one frame sounds, which a bend counts its divider steps from, and nothing where it rests.""" + instruction = candidate.instruction + if not isinstance(instruction, (PulseInstruction, TriangleInstruction)) or not instruction.on: + return None + + return instruction.pitch diff --git a/tests/unit/sampletones_core/reconstructions/reconstructor/refinement/test_refiner.py b/tests/unit/sampletones_core/reconstructions/reconstructor/refinement/test_refiner.py index ccd540d94..dc11e3c05 100644 --- a/tests/unit/sampletones_core/reconstructions/reconstructor/refinement/test_refiner.py +++ b/tests/unit/sampletones_core/reconstructions/reconstructor/refinement/test_refiner.py @@ -1,5 +1,5 @@ from dataclasses import dataclass -from typing import Dict, Final, List, Tuple +from typing import Dict, Final, List, Optional, Tuple import numpy as np import pytest @@ -8,7 +8,8 @@ from sampletones_core.configs.generation import GenerationConfig from sampletones_core.constants.algorithm import RESTING_STEM_ID from sampletones_core.constants.enums import ChannelName -from sampletones_core.generators import get_generators_by_channels +from sampletones_core.fft.instantaneous import FundamentalReading +from sampletones_core.generators import PulseGenerator, get_generators_by_channels from sampletones_core.instructions import PulseInstruction, TriangleInstruction from sampletones_core.reconstructions.reconstructor.contribution import Contribution from sampletones_core.reconstructions.reconstructor.matching import ScoredCandidate @@ -23,6 +24,8 @@ TONES: Final[List[ChannelName]] = [ChannelName.PULSE1, ChannelName.TRIANGLE] PITCH: Final[int] = 60 +NEXT_PITCH: Final[int] = 62 +SILENT_FRAMES: Final[int] = 2 VOLUME: Final[int] = 12 FRAMES: Final[int] = 4 BEND: Final[int] = 3 @@ -125,6 +128,52 @@ def test_only_the_channels_the_stem_names_are_carried( assert not any(bends), f"{channel_name} was left out and moved anyway" +class TestEachNoteHoldsItsOwnBend: + """A bend counts divider steps from its own note, so a note's frames are settled on their own.""" + + @staticmethod + def _two_notes(config: Config, monkeypatch: pytest.MonkeyPatch) -> List[int]: + """The bends of a note that reads nothing followed by one that reads ``BEND``.""" + sounding = PulseGenerator(config).sounds_at(NEXT_PITCH, BEND) + + class SilentThenBent: + def __init__(self, recording: np.ndarray, sample_rate: int, hop_length: int) -> None: + del recording, sample_rate, hop_length + + def at(self, frame: int, reference: float) -> Optional[FundamentalReading]: + del reference + return FundamentalReading(frequency=sounding, confidence=1.0) if frame >= SILENT_FRAMES else None + + monkeypatch.setattr(refiner, "InstantaneousPitch", SilentThenBent) + pitches = [PITCH] * SILENT_FRAMES + [NEXT_PITCH] * FRAMES + stream = [ + ScoredCandidate( + instruction=PulseInstruction(on=True, pitch=pitch, volume=VOLUME, duty_cycle=2), + cost=0.0, + contribution=Contribution.silence(1, 1), + ) + for pitch in pitches + ] + stems = _stems(StemEntry(id=STEM_A, settings=StemSettings(channels=TONES, bends=list(TONES)))) + refined = PitchRefiner(config=config, channels=get_generators_by_channels(config, TONES), stems=stems) + streams = refined.refine( + {ChannelName.PULSE1: stream}, + {ChannelName.PULSE1: [STEM_A] * len(pitches)}, + {STEM_A: _recording(config, ChannelName.PULSE1)}, + ) + return _bends(streams, ChannelName.PULSE1) + + def test_a_note_that_reads_nothing_stays_at_its_note(self, config: Config, monkeypatch: pytest.MonkeyPatch) -> None: + bends = self._two_notes(config, monkeypatch) + + assert bends[:SILENT_FRAMES] == [0] * SILENT_FRAMES + + def test_the_next_note_takes_the_bend_it_reads(self, config: Config, monkeypatch: pytest.MonkeyPatch) -> None: + bends = self._two_notes(config, monkeypatch) + + assert bends[SILENT_FRAMES:] == [BEND] * FRAMES + + class TestWhatTheRefinementReadsPerStem: """A reading is taken for the recording a bend is read from, and for no other.""" From 9470c860f53bc5d69741ff4bf56e96cfd4d2c4e9 Mon Sep 17 00:00:00 2001 From: JakimPL Date: Mon, 5 Oct 2026 18:35:03 +0200 Subject: [PATCH 09/12] Lowered: the bend confidence threshold to 0.10 --- src/sampletones_core/configs/generation.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/sampletones_core/configs/generation.yaml b/src/sampletones_core/configs/generation.yaml index b96de3b1b..1434f17f5 100644 --- a/src/sampletones_core/configs/generation.yaml +++ b/src/sampletones_core/configs/generation.yaml @@ -25,6 +25,6 @@ decoder: on_off_weight: 0.2 refinement: - confidence: 0.15 + confidence: 0.10 change_weight: 2.0 window: 4 From 50a4aa17ef68212495cd631bb92433c40af8718e Mon Sep 17 00:00:00 2001 From: JakimPL Date: Mon, 5 Oct 2026 18:35:03 +0200 Subject: [PATCH 10/12] Recorded: why a quiet triangle bass is left out --- docs/concepts/reconstruction.md | 3 +++ 1 file changed, 3 insertions(+) diff --git a/docs/concepts/reconstruction.md b/docs/concepts/reconstruction.md index a89c93c27..e2b5a983b 100644 --- a/docs/concepts/reconstruction.md +++ b/docs/concepts/reconstruction.md @@ -411,6 +411,9 @@ be shown and played on a common scale. and the coefficient is one global scalar. Material whose *useful* content spans a wider range than that cannot be fully captured. A long crescendo and a very quiet passage under a loud one are examples. Content far below the working level falls under the quietest playable note and is rendered as silence. +- **The triangle's fixed level.** The triangle plays at one volume. A bass a few decibels quieter than that + level is left out, and the calibration referees score the result closer to the recording than the same + conversion with the bass played too loud. A bass near that level is played. - **CQT time resolution.** Constant-Q analysis needs long windows at low frequencies, so low-pitched transients are smeared in time under `cqt`. `fft` and `logfft` localize time better at the cost of low-frequency resolution. From f7bedf598457d238885664aee7d8bb7f5226b0e8 Mon Sep 17 00:00:00 2001 From: JakimPL Date: Mon, 5 Oct 2026 19:24:49 +0200 Subject: [PATCH 11/12] Restated: the refinement's change weight as what remains to calibrate --- docs/development/bugs-and-todos.md | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/docs/development/bugs-and-todos.md b/docs/development/bugs-and-todos.md index 25ca6b08d..d581944f2 100644 --- a/docs/development/bugs-and-todos.md +++ b/docs/development/bugs-and-todos.md @@ -48,10 +48,12 @@ dimension the import starts carrying. which is a tenth or more of a short conversion on a CPU build. * Keeping the recordings a stopped folder scan has found, so stopping a long walk keeps the count the reader watched climb. -* Calibrating the pitch refinement. The settings that decide how a pitch reading bends a note (a - confidence threshold, a change weight and a window) are chosen by hand, and - [the calibration](../tools/calibration.md) could measure them. The change weight trades vibrato against - jitter. +* Calibrating the pitch refinement's change weight and window. The confidence threshold was measured on + bending probes; the change weight and the window are chosen by hand, and + [the calibration](../tools/calibration.md) could measure them. The change weight counts divider steps, + which span about 2 cents at 110 Hz and about 27 cents at 1760 Hz. One weight therefore lets noise played + as low notes change its bend often while it holds high vibrato back. Counting it in cents weighs every + register alike, and its value then needs choosing again. * A setting for the lowest frequency the analysis reads. The analysis starts at the triangle's lowest note, about 27.3 Hz, so every note the chip plays is read from its fundamental, and its longest window spans over half a second. Music that stays above the triangle's lowest octave would convert about a From e76e1225deedc2ff0dbe1836d923ba4493ae4c0c Mon Sep 17 00:00:00 2001 From: JakimPL Date: Tue, 6 Oct 2026 12:30:55 +0200 Subject: [PATCH 12/12] Minor improvements --- src/sampletones_core/data/document.py | 2 +- src/sampletones_core/fft/window/cyclic.py | 10 +++++ src/sampletones_core/library/fragment.py | 3 +- .../sampletones_core/fft/window/__init__.py | 0 .../fft/window/test_cyclic.py | 39 +++++++++++++++++++ 5 files changed, 51 insertions(+), 3 deletions(-) create mode 100644 tests/unit/sampletones_core/fft/window/__init__.py create mode 100644 tests/unit/sampletones_core/fft/window/test_cyclic.py diff --git a/src/sampletones_core/data/document.py b/src/sampletones_core/data/document.py index 1bfc3d737..6d816e264 100644 --- a/src/sampletones_core/data/document.py +++ b/src/sampletones_core/data/document.py @@ -5,7 +5,7 @@ from sampletones_shared.types.path import Pathlike DOCUMENT_MAGIC: Final[bytes] = b"\x1f\x8b" -COMPRESSION_LEVEL: Final[int] = 9 +COMPRESSION_LEVEL: Final[int] = 6 STATED_TIMESTAMP: Final[int] = 0 diff --git a/src/sampletones_core/fft/window/cyclic.py b/src/sampletones_core/fft/window/cyclic.py index aebb3ebd5..d82f8c9d5 100644 --- a/src/sampletones_core/fft/window/cyclic.py +++ b/src/sampletones_core/fft/window/cyclic.py @@ -56,6 +56,16 @@ def get_windowed_fragment(self, phase: float, window: Window) -> np.ndarray: windowed_fragment: np.ndarray = fragment * window.envelope return windowed_fragment + def get_frame(self, phase: float, window: Window) -> np.ndarray: + """The frame at the middle of the windowed fragment ``phase`` reads, under the envelope there. + + It reads the frame's own samples, so its cost follows the frame length however wide the + window spans. + """ + envelope = window.get_frame_from_window(window.envelope, copy=False) + frame: np.ndarray = self.get_fragment(phase, window.frame_length) * envelope + return frame + @property def length(self) -> int: return len(self.array) diff --git a/src/sampletones_core/library/fragment.py b/src/sampletones_core/library/fragment.py index 57b3a0ea2..b72faf8d3 100644 --- a/src/sampletones_core/library/fragment.py +++ b/src/sampletones_core/library/fragment.py @@ -73,9 +73,8 @@ def instruction(self) -> InstructionT: return instruction def get_fragment(self, shift: int, config: Config, window: Window) -> Fragment: - audio = window.get_frame_from_window(self.sample.get_windowed_fragment(shift, window)) return Fragment( - audio=audio, + audio=self.sample.get_frame(shift, window), feature=self.feature, config=config, ) diff --git a/tests/unit/sampletones_core/fft/window/__init__.py b/tests/unit/sampletones_core/fft/window/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/tests/unit/sampletones_core/fft/window/test_cyclic.py b/tests/unit/sampletones_core/fft/window/test_cyclic.py new file mode 100644 index 000000000..6b3ceec4b --- /dev/null +++ b/tests/unit/sampletones_core/fft/window/test_cyclic.py @@ -0,0 +1,39 @@ +from typing import Final, Tuple + +import numpy as np +import pytest + +from sampletones_core.constants.enums import SpectrumMethod +from sampletones_core.fft import Window +from sampletones_core.fft.window.cyclic import CyclicArray +from tests.suite.analysis import analyzed_config + +SAMPLE_LENGTH: Final[int] = 500 +SHIFTS: Final[Tuple[int, ...]] = (0, 137, SAMPLE_LENGTH - 1) +TAPER_EXTENSIONS: Final[Tuple[int, ...]] = (0, 1) +SEED: Final[int] = 7 + + +class TestAFrameIsTheMiddleOfItsWindow: + """A sample's frame is the middle of the windowed fragment the same shift reads, under the envelope + there: for the flat constant-Q envelope, the FFT's tapered one, and a taper an odd number of samples + long, whose frame reaches one tapered sample. The sample is shorter than a frame, so the reading + wraps around it.""" + + @pytest.mark.parametrize("shift", SHIFTS) + @pytest.mark.parametrize("extension", TAPER_EXTENSIONS) + @pytest.mark.parametrize("method", (SpectrumMethod.CQT, SpectrumMethod.FFT)) + def test_the_frame_equals_the_middle_of_the_windowed_fragment( + self, + method: SpectrumMethod, + extension: int, + shift: int, + ) -> None: + config = analyzed_config(method, gamma=0) + window = Window.from_config(config, custom_size=Window.from_config(config).size + extension) + audio = np.random.default_rng(SEED).standard_normal(SAMPLE_LENGTH).astype(np.float32) + sample = CyclicArray(array=audio, sample_rate=config.library.sample_rate) + + expected = window.get_frame_from_window(sample.get_windowed_fragment(shift, window)) + + np.testing.assert_array_equal(sample.get_frame(shift, window), expected)