Files
reasampler/tests/test_sampler_core.cpp
T
daniel 589a8e078b Γ-W1-T5: a real Preserve time-stretcher — write rate is duration, tap rate is pitch
Generalizes the correlation-aligned SOLA delay line so the feed and the shift are
independent rates over one ring. Unity is bit-identical to the shipped read, asserted
against a hash baseline captured pre-change.
2026-08-02 13:47:19 -04:00

3323 lines
155 KiB
C++
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Standalone tests for reasampler::sampler_core — no VST3, no REAPER, no test
// framework. Same fast build/run loop as bank_model_tests / peaks_tests: feed known
// inputs, assert the engine's behavior.
//
// Covers (PLAN.md S3 / CONTEXT.md §Phase S pure core):
// 1. polyphonic allocation — N notes -> N voices; note-off releases the right voice.
// 2. voice stealing at the bound — deterministic policy (release-first, then oldest).
// 3. ADSR envelope shape vs a known signal, incl. release-before-sustain.
// 4. repitch ratio correctness across +/-1 octave from root incl. unity, asserted on
// the observed period of a synthesized sine.
// 5. loop-point sustain — held note past sample end loops [start,end) seamlessly;
// zero-length loop and absent-loop behavior.
// 6. keymap: chromatic-from-single-root; zoned ranges with boundary notes; velocity
// -> volume; out-of-zone note -> defined no-play.
//
// The plain-data boundary (no VST3/REAPER types in the core) is enforced STRUCTURALLY
// by the CMake target linking neither SDK — this file includes only sampler_core.h +
// the standard library, which is itself the compile-time proof.
#include "../src/core/instrument/engine/voice_engine.h"
#include <algorithm>
#include <cmath>
#include <cstdint>
#include <cstdio>
#include <cstring>
#include <ctime>
#include <vector>
using namespace reasampler;
using namespace reasampler::instrument::engine;
static int g_fail = 0;
#define CHECK(cond) do { if(!(cond)) { \
std::printf("FAIL line %d: %s\n", __LINE__, #cond); ++g_fail; } } while(0)
static bool approx(double a, double b, double tol) { return std::fabs(a - b) <= tol; }
constexpr double kPi = 3.14159265358979323846;
// A silent (all-1.0) sample so a rendered voice's output tracks the envelope * velocity
// directly (DC of amplitude 1). Root at note 60 by default.
static SampleData dcSample(std::size_t frames, int rootNote = 60) {
SampleData s;
s.frames.assign(frames, 1.0f);
s.rootNote = rootNote;
return s;
}
// A mono sine of `cycles` periods over `frames` frames — used to observe repitch by
// measuring the played-back period.
static SampleData sineSample(std::size_t frames, double cycles, int rootNote = 60) {
SampleData s;
s.frames.resize(frames);
for (std::size_t i = 0; i < frames; ++i) {
s.frames[i] = static_cast<float>(std::sin(2.0 * kPi * cycles *
static_cast<double>(i) / static_cast<double>(frames)));
}
s.rootNote = rootNote;
return s;
}
// An ADSR that stays fully open (level 1) forever while held, so voice output equals
// velocity gain — isolates allocation/repitch/loop tests from envelope shaping.
static AdsrParams flatAdsr() {
AdsrParams a;
a.attackFrames = 0;
a.decayFrames = 0;
a.sustainLevel = 1.0;
a.releaseFrames = 0; // note-off -> instant silence.
return a;
}
// ---------------------------------------------------------------------------
// 6. Full-keyboard response over the one loaded capture.
// ---------------------------------------------------------------------------
static void testEveryKeyPlaysTheLoadedCapture() {
// No key range survives: the loaded capture answers every note in 0..127, repitched
// from its root. Each note-on must take a real voice.
SampleData km = dcSample(100, 60);
VoiceEngine eng(128, km);
for (int n = 0; n <= 127; ++n) {
CHECK(eng.noteOn(n, 100) != VoiceEngine::kNoVoice);
}
CHECK(eng.activeVoiceCount() == 128);
}
static void testUnplayableCaptureRefusesEveryNote() {
// Nothing decoded -> the defined no-play at every key, in both voice modes, rather
// than a voice started on an empty read span.
SampleData empty; // no frames
VoiceEngine poly(4, empty);
CHECK(poly.noteOn(60, 100) == VoiceEngine::kNoVoice);
CHECK(poly.noteOn(0, 100) == VoiceEngine::kNoVoice);
CHECK(poly.activeVoiceCount() == 0);
VoiceEngine mono(4, empty, 0, 0, VoiceMode::Mono);
CHECK(mono.noteOn(60, 100) == VoiceEngine::kNoVoice);
CHECK(mono.activeVoiceCount() == 0);
}
static void testOutOfRangeNotesAreRefusedInMono() {
// The mono held stack keys notes as uint8, so an out-of-range note must be rejected
// BEFORE it can alias onto a real held note.
SampleData km = dcSample(100, 60);
VoiceEngine eng(4, km, 0, 0, VoiceMode::Mono);
CHECK(eng.noteOn(-1, 100) == VoiceEngine::kNoVoice);
CHECK(eng.noteOn(128, 100) == VoiceEngine::kNoVoice);
CHECK(eng.activeVoiceCount() == 0);
}
// ---------------------------------------------------------------------------
// 4. Repitch ratio correctness.
// ---------------------------------------------------------------------------
static void testPitchRatioMath() {
CHECK(approx(pitchRatio(60, 60), 1.0, 1e-9)); // unity at root
CHECK(approx(pitchRatio(72, 60), 2.0, 1e-9)); // +1 octave
CHECK(approx(pitchRatio(48, 60), 0.5, 1e-9)); // -1 octave
CHECK(approx(pitchRatio(61, 60), std::pow(2.0, 1.0 / 12.0), 1e-9)); // +1 semitone
}
// --- S-VIEW-6 key-tracking ratio math (pure), asserted at 0 / 100 / 200% + off-root. ---
static void testKeyTrackedRatioMath() {
// 100% (keyTrack == 1.0) is standard 12-tone-ET and BIT-IDENTICAL to pitchRatio: the argument
// to std::pow is (note-root)*1.0, exact in IEEE-754, so the same call yields the same bits.
for (int note = 0; note <= 127; ++note) {
CHECK(keyTrackedRatio(note, 60, 1.0) == pitchRatio(note, 60)); // exact equality, not approx
}
CHECK(approx(keyTrackedRatio(72, 60, 1.0), 2.0, 1e-9)); // +1 octave tracked normally
CHECK(approx(keyTrackedRatio(48, 60, 1.0), 0.5, 1e-9)); // -1 octave tracked normally
// 0% (keyTrack == 0.0): no tracking. Every key — including off-root ones — plays root pitch.
CHECK(approx(keyTrackedRatio(60, 60, 0.0), 1.0, 1e-12)); // at root: unity (trivially)
CHECK(approx(keyTrackedRatio(72, 60, 0.0), 1.0, 1e-12)); // an octave up STILL plays root pitch
CHECK(approx(keyTrackedRatio(48, 60, 0.0), 1.0, 1e-12)); // an octave down STILL plays root pitch
CHECK(approx(keyTrackedRatio(67, 60, 0.0), 1.0, 1e-12)); // an off-root 5th STILL plays root pitch
// 200% (keyTrack == 2.0): double-rate tracking. The semitone offset is doubled, so a +12 key
// plays as if +24 (two octaves, ratio 4.0), a -12 key as -24 (ratio 0.25).
CHECK(approx(keyTrackedRatio(72, 60, 2.0), 4.0, 1e-9)); // +12 -> +24 semis -> 4.0
CHECK(approx(keyTrackedRatio(48, 60, 2.0), 0.25, 1e-9)); // -12 -> -24 semis -> 0.25
CHECK(approx(keyTrackedRatio(60, 60, 2.0), 1.0, 1e-12)); // root is unity at ANY keyTrack
// Off-root at 200% for a single semitone: +1 semi -> +2 semis -> 2^(2/12).
CHECK(approx(keyTrackedRatio(61, 60, 2.0), std::pow(2.0, 2.0 / 12.0), 1e-9));
// An arbitrary intermediate scalar (50%): +12 key tracks as +6 semis -> 2^(6/12) = sqrt(2).
CHECK(approx(keyTrackedRatio(72, 60, 0.5), std::pow(2.0, 6.0 / 12.0), 1e-9));
}
// Observe repitch on the rendered signal: a voice played an octave above root should
// advance through the sample twice as fast, so a sine's observed period halves. We
// measure the period by counting the interval between positive-going zero crossings.
static double observedPeriodFrames(const std::vector<AudioSample>& out) {
std::vector<std::size_t> upCrossings;
for (std::size_t i = 1; i < out.size(); ++i) {
if (out[i - 1] <= 0.0f && out[i] > 0.0f) upCrossings.push_back(i);
}
if (upCrossings.size() < 2) return 0.0;
// Average spacing between crossings.
double sum = 0.0;
for (std::size_t i = 1; i < upCrossings.size(); ++i) {
sum += static_cast<double>(upCrossings[i] - upCrossings[i - 1]);
}
return sum / static_cast<double>(upCrossings.size() - 1);
}
static void testRepitchObservedPeriod() {
// A sine of 20 cycles over 8000 frames -> native period 400 frames at unity.
const std::size_t frames = 8000;
const double cycles = 20.0;
const double nativePeriod = static_cast<double>(frames) / cycles; // 400
// Unity: played at root, observed period ~= native.
{
SampleData km = (sineSample(frames, cycles, 60));
VoiceEngine eng(4, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, frames);
double p = observedPeriodFrames(out);
CHECK(approx(p, nativePeriod, 2.0));
}
// +1 octave: advances 2x, observed period halves.
{
SampleData km = (sineSample(frames, cycles, 60));
VoiceEngine eng(4, km);
eng.noteOn(72, 127);
std::vector<AudioSample> out;
eng.render(out, frames / 2); // half as many frames covers the whole sample
double p = observedPeriodFrames(out);
CHECK(approx(p, nativePeriod / 2.0, 2.0));
}
// -1 octave: advances 0.5x, observed period doubles.
{
SampleData km = (sineSample(frames, cycles, 60));
VoiceEngine eng(4, km);
eng.noteOn(48, 127);
std::vector<AudioSample> out;
eng.render(out, frames);
double p = observedPeriodFrames(out);
CHECK(approx(p, nativePeriod * 2.0, 4.0));
}
}
// --- S-VIEW-6 keyTrack reaches the VARISPEED engine: observed period tracks the scalar. ---
static void testKeyTrackVarispeedObservedPeriod() {
const std::size_t frames = 8000;
const double cycles = 20.0;
const double nativePeriod = static_cast<double>(frames) / cycles; // 400 at unity
auto periodAt = [&](int note, double keyTrack) -> double {
SampleData s = sineSample(frames, cycles, 60);
s.play.pitchEngine = PitchEngine::Varispeed;
SampleData km = (std::move(s));
// The single zone spans the keyboard from root 60; stamp the key-track scalar on it.
km.keyTrack = keyTrack;
VoiceEngine eng(4, km);
eng.noteOn(note, 127);
std::vector<AudioSample> out;
eng.render(out, frames);
return observedPeriodFrames(out);
};
// note 72 (+1 octave). At 100% it plays an octave up (period halves ~200). At 0% it plays at
// ROOT pitch (period ~native 400 — no tracking). The observed periods must differ by ~2x, which
// proves the scalar drove the Varispeed read rate.
const double at100 = periodAt(72, 1.0);
const double at0 = periodAt(72, 0.0);
CHECK(approx(at100, nativePeriod / 2.0, 3.0)); // 100%: tracked an octave up
CHECK(approx(at0, nativePeriod, 3.0)); // 0%: no tracking, plays root pitch
}
// --- S-VIEW-6 keyTrack reaches the PRESERVE engine: at 0% an off-root note collapses to the root
// shift (unity), producing output identical to playing the root note. Proves keyTrack feeds
// baseRatio_ -> the Preserve shift amount (not merely the Varispeed read rate). ---
static void testKeyTrackPreserveShiftCollapsesAtZero() {
const std::size_t frames = 2000;
const std::size_t window = 512;
const double cycles = 40.0;
auto renderPreserve = [&](int note, double keyTrack) -> std::vector<AudioSample> {
SampleData s = sineSample(frames, cycles, 60);
s.play.pitchEngine = PitchEngine::Preserve; // Gate, no loop -> runs to sample end
SampleData km = (std::move(s));
km.keyTrack = keyTrack;
VoiceEngine eng(1, km, /*preserveCap=*/0, static_cast<std::int64_t>(window));
eng.noteOn(note, 127);
std::vector<AudioSample> out;
eng.render(out, frames);
return out;
};
// An off-root note (+7) at keyTrack 0.0 sets the Preserve shift to the root ratio (1.0) — the
// shifter is pass-through, so the output must be BIT-IDENTICAL to playing the ROOT note (whose
// offset is 0, also shift 1.0). If keyTrack only touched Varispeed, these would differ.
const std::vector<AudioSample> offRootNoTrack = renderPreserve(67, 0.0);
const std::vector<AudioSample> rootRef = renderPreserve(60, 1.0);
CHECK(offRootNoTrack.size() == rootRef.size());
bool identical = offRootNoTrack.size() == rootRef.size();
for (std::size_t i = 0; i < offRootNoTrack.size() && identical; ++i) {
if (offRootNoTrack[i] != rootRef[i]) identical = false;
}
CHECK(identical); // 0% tracking collapses the Preserve shift to unity, exactly like the root
}
// ---------------------------------------------------------------------------
// 3. ADSR envelope shape vs a known signal.
// ---------------------------------------------------------------------------
static void testAdsrShape() {
AdsrParams p;
p.attackFrames = 10;
p.decayFrames = 10;
p.sustainLevel = 0.5;
p.releaseFrames = 10;
AdsrEnvelope env;
env.configure(p);
env.noteOn();
// Attack: 0 -> ramps up. Frame 0 == 0, rising each frame.
double prev = -1.0;
for (int i = 0; i < 10; ++i) {
double v = env.tick();
CHECK(v >= prev); // monotonic non-decreasing through attack
CHECK(v >= 0.0 && v <= 1.0);
prev = v;
}
// Decay: from 1.0 down toward sustain 0.5, monotonic non-increasing.
prev = 2.0;
for (int i = 0; i < 10; ++i) {
double v = env.tick();
CHECK(v <= prev + 1e-9); // non-increasing through decay
CHECK(v >= 0.5 - 1e-9); // never below sustain during decay
prev = v;
}
// Sustain: holds 0.5 indefinitely.
for (int i = 0; i < 100; ++i) {
CHECK(approx(env.tick(), 0.5, 1e-9));
}
CHECK(env.stage() == AdsrEnvelope::Stage::Sustain);
// Release: 0.5 -> 0 over 10 frames, then Finished + latched at 0.
env.noteOff();
prev = 1.0;
for (int i = 0; i < 10; ++i) {
double v = env.tick();
CHECK(v <= prev + 1e-9); // non-increasing through release
prev = v;
}
CHECK(env.finished());
for (int i = 0; i < 10; ++i) CHECK(approx(env.tick(), 0.0, 1e-12));
}
static void testAdsrReleaseBeforeSustain() {
// noteOff during the attack ramp releases from the PARTIAL level, not sustain.
AdsrParams p;
p.attackFrames = 100;
p.decayFrames = 10;
p.sustainLevel = 0.8;
p.releaseFrames = 20;
AdsrEnvelope env;
env.configure(p);
env.noteOn();
// Advance 50 frames into a 100-frame attack -> partial level ~0.5.
double last = 0.0;
for (int i = 0; i < 50; ++i) last = env.tick();
CHECK(last > 0.3 && last < 0.7); // partway up the attack ramp
CHECK(env.stage() == AdsrEnvelope::Stage::Attack);
env.noteOff();
CHECK(env.stage() == AdsrEnvelope::Stage::Release);
// First release frame must be at or below the partial level we left off at —
// NOT jump up to sustain 0.8. This is the release-before-sustain guarantee.
double firstRelease = env.tick();
CHECK(firstRelease <= last + 1e-9);
CHECK(firstRelease < p.sustainLevel); // proves it did not snap to sustain
// Decays to zero.
double prev = firstRelease;
for (int i = 0; i < 20; ++i) {
double v = env.tick();
CHECK(v <= prev + 1e-9);
prev = v;
}
CHECK(env.finished());
}
static void testAdsrZeroAttackDecay() {
// Zero attack + zero decay -> jumps straight to sustain on the first ticks.
AdsrParams p;
p.attackFrames = 0;
p.decayFrames = 0;
p.sustainLevel = 0.7;
p.releaseFrames = 5;
AdsrEnvelope env;
env.configure(p);
env.noteOn();
// Zero attack emits the attack peak (1.0) on frame 0 and immediately transitions
// through the (also zero-length) decay, so by frame 1 the envelope is holding
// sustain. The peak-at-boundary is the documented single-frame edge, not a bug.
CHECK(approx(env.tick(), 1.0, 1e-9)); // frame 0: attack peak
CHECK(approx(env.tick(), 0.7, 1e-9)); // frame 1: sustain
CHECK(approx(env.tick(), 0.7, 1e-9));
CHECK(env.stage() == AdsrEnvelope::Stage::Sustain);
}
// ---------------------------------------------------------------------------
// 1. Polyphonic allocation + note-off routing.
// ---------------------------------------------------------------------------
static void testPolyphonicAllocation() {
SampleData km = (dcSample(1000, 60));
VoiceEngine eng(8, km);
// Four simultaneous notes -> four active voices, each on a distinct voice.
std::size_t v60 = eng.noteOn(60, 100);
std::size_t v64 = eng.noteOn(64, 100);
std::size_t v67 = eng.noteOn(67, 100);
std::size_t v72 = eng.noteOn(72, 100);
CHECK(v60 != VoiceEngine::kNoVoice);
CHECK(eng.activeVoiceCount() == 4);
CHECK(v60 != v64 && v64 != v67 && v67 != v72 && v60 != v72);
// Note-off on 64 releases exactly one voice; with instant release it goes idle
// after the next render frame.
eng.noteOff(64);
std::vector<AudioSample> out;
eng.render(out, 1);
CHECK(eng.activeVoiceCount() == 3);
// The still-held notes keep sounding.
eng.render(out, 1);
CHECK(eng.activeVoiceCount() == 3);
}
static void testNoteOffReleasesNewestSameNote() {
// Prove that noteOff releases the NEWEST (highest startOrder) instance of a
// re-triggered note, leaving the older voice in sustain.
//
// Two voices at distinct velocities so their output is distinguishable:
// "first" (older) -> velocity 64 -> gain ~0.504 (G_old)
// "second" (newer) -> velocity 127 -> gain 1.0 (G_new)
//
// With a DC-1 sample and sustain=1, while both are held:
// render sum == G_old + G_new.
//
// After noteOff (must release newest), the newer voice enters a short release.
// Render past releaseFrames: newer voice finishes; only the older voice remains.
// Sum then equals G_old, and activeVoiceCount drops to 1. If the WRONG voice
// were released, the older would finish and the remaining sum would equal G_new
// (1.0 vs ~0.504) — the velocities make the error distinguishable.
const int velOld = 64;
const int velNew = 127;
const double gainOld = velOld / 127.0; // ~0.504
const double gainNew = velNew / 127.0; // 1.0
SampleData sd = dcSample(100000, 60);
sd.play.adsr = flatAdsr();
sd.play.adsr.releaseFrames = 10; // short but non-zero so voice stays active through release
SampleData km = (sd);
// A LINEAR velocity curve keeps the two velocities distinguishable (velocity/127). The default
// flat y=1 curve (S-VIEW-9 R10-F1) would render both at unity, collapsing the distinction this
// note-off-selection test relies on — so we opt this zone back to the linear response.
km.velocityCurve = VelocityCurve::linear();
VoiceEngine eng(8, km);
std::size_t first = eng.noteOn(60, velOld); // older voice, lower gain
std::size_t second = eng.noteOn(60, velNew); // newer voice, higher gain
CHECK(first != second);
CHECK(eng.activeVoiceCount() == 2);
// While both are held, combined output equals gainOld + gainNew.
{
std::vector<AudioSample> out;
eng.render(out, 1);
CHECK(approx(out[0], gainOld + gainNew, 1e-4));
}
// Release once — must target the NEWEST voice (second).
eng.noteOff(60);
// Render past the release (releaseFrames == 10): newer voice goes Finished.
std::vector<AudioSample> out;
eng.render(out, 20);
// Newer voice must be done; only the older voice remains.
CHECK(eng.activeVoiceCount() == 1);
// Tail frames must equal gainOld (~0.504), NOT gainNew (1.0).
// If the older voice were released instead, the tail would be ~1.0 here.
for (std::size_t i = 15; i < out.size(); ++i) {
CHECK(approx(out[i], gainOld, 1e-4));
}
// A second note-off releases the remaining older voice.
eng.noteOff(60);
eng.render(out, 20);
CHECK(eng.activeVoiceCount() == 0);
}
// ---------------------------------------------------------------------------
// 2. Voice stealing at the bound.
// ---------------------------------------------------------------------------
static void testStealsReleasingVoiceFirst() {
// Long per-zone release so the voice stays active through the release tail. Voice::start reads
// sample.play.adsr (the engine holds no ADSR), so the long release lives on the SampleData.
SampleData s = dcSample(100000, 60);
s.play.adsr = flatAdsr();
s.play.adsr.releaseFrames = 100000; // long release so a released voice stays "active"
SampleData km = (std::move(s));
VoiceEngine eng(2, km);
std::size_t vA = eng.noteOn(60, 100); // startOrder 1
std::size_t vB = eng.noteOn(62, 100); // startOrder 2
CHECK(eng.activeVoiceCount() == 2);
// Release the NEWER voice (62) — it becomes the only releasing voice.
eng.noteOff(62);
std::vector<AudioSample> out;
eng.render(out, 1);
CHECK(eng.activeVoiceCount() == 2); // both still ringing (long release)
// A new note with the pool full must steal the RELEASING voice (vB), not the
// older held voice (vA) — release-first policy.
std::size_t vC = eng.noteOn(64, 100);
CHECK(vC == vB);
CHECK(eng.activeVoiceCount() == 2);
}
static void testStealsOldestWhenNoneReleasing() {
// Long per-zone release — placed on SampleData.play.adsr per the S12 fix.
SampleData s = dcSample(100000, 60);
s.play.adsr = flatAdsr();
s.play.adsr.releaseFrames = 100000;
SampleData km = (std::move(s));
VoiceEngine eng(2, km);
std::size_t vA = eng.noteOn(60, 100); // startOrder 1 (oldest)
std::size_t vB = eng.noteOn(62, 100); // startOrder 2
CHECK(vA != vB);
// No voice released; both held. A new note steals the OLDEST (vA).
std::size_t vC = eng.noteOn(64, 100);
CHECK(vC == vA);
CHECK(eng.activeVoiceCount() == 2);
// The stolen voice now carries note 64; a note-off on 60 (the stolen-away note)
// finds nothing to release.
std::size_t before = eng.activeVoiceCount();
eng.noteOff(60);
std::vector<AudioSample> out;
eng.render(out, 1);
CHECK(eng.activeVoiceCount() == before); // 60 no longer exists; no-op
}
// ---------------------------------------------------------------------------
// 5. Loop-point-aware sustain.
// ---------------------------------------------------------------------------
static void testLoopSustainSeamless() {
// A sample whose [0,20) frames are a distinctive ramp and [20,40) is a flat loop
// region of value 0.5. Held far past the sample end, the voice must keep emitting
// the loop region (0.5) rather than going silent.
SampleData s;
s.frames.resize(40);
for (int i = 0; i < 20; ++i) s.frames[i] = static_cast<float>(i) / 20.0f; // attack
for (int i = 20; i < 40; ++i) s.frames[i] = 0.5f; // loop body
s.rootNote = 60;
s.loop.hasLoop = true;
s.loop.start = 20;
s.loop.end = 40;
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(60, 127); // unity ratio, full velocity
std::vector<AudioSample> out;
eng.render(out, 200); // 5x the sample length
// Voice is still active (looping), not exhausted.
CHECK(eng.activeVoiceCount() == 1);
// Frames well past the loop start must sit at the loop body value 0.5.
for (std::size_t i = 60; i < out.size(); ++i) {
CHECK(approx(out[i], 0.5, 1e-4));
}
}
static void testZeroLengthLoopGoesSilent() {
// A zero-length loop (start == end) is the "no sustain" marker: the note runs off
// the sample end and the voice goes idle, rather than spinning on an empty span.
SampleData s = dcSample(50, 60); // 50 frames of 1.0
s.loop.hasLoop = true;
s.loop.start = 25;
s.loop.end = 25; // zero length
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 100); // past the 50-frame end
// After frame ~50 the voice should have gone idle (no loop to sustain it).
CHECK(eng.activeVoiceCount() == 0);
// Tail frames are silent.
for (std::size_t i = 60; i < out.size(); ++i) CHECK(approx(out[i], 0.0, 1e-6));
}
static void testSingleFrameLoop() {
// A loop of exactly one frame [start, start+1) — the narrowest valid loop.
// The path is correct-by-luck (loopLen = 1.0 divides evenly into any integer
// readPos advance at unity ratio), but Tier-2 tight loops make it load-bearing.
SampleData s;
s.frames.resize(10);
for (int i = 0; i < 10; ++i) s.frames[i] = static_cast<float>(i) * 0.1f;
s.rootNote = 60;
s.loop.hasLoop = true;
s.loop.start = 5;
s.loop.end = 6; // single-frame loop: [5, 6)
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(60, 127); // unity ratio, full velocity
std::vector<AudioSample> out;
eng.render(out, 50); // well past the sample end
// Voice must still be active — the single-frame loop keeps it alive.
CHECK(eng.activeVoiceCount() == 1);
// Every frame from the loop-start onward must be the value of frame 5 (0.5).
for (std::size_t i = 10; i < out.size(); ++i) {
CHECK(approx(out[i], 0.5, 1e-4));
}
}
static void testAbsentLoopGoesSilent() {
// No loop at all: held note runs off the end and goes idle (same as zero-length).
SampleData s = dcSample(50, 60);
// s.loop.hasLoop stays false.
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 100);
CHECK(eng.activeVoiceCount() == 0);
for (std::size_t i = 60; i < out.size(); ++i) CHECK(approx(out[i], 0.0, 1e-6));
}
// ---------------------------------------------------------------------------
// start point (S11): the voice's initial read position is SampleData::startFrame.
// ---------------------------------------------------------------------------
static void testStartFrameOffsetsInitialRead() {
// A per-frame ramp (frame i holds i*0.01) so the first rendered value pinpoints the
// read position. startFrame = 30 -> the first output frame reads frame 30 (0.30).
SampleData s;
s.frames.resize(100);
for (int i = 0; i < 100; ++i) s.frames[i] = static_cast<float>(i) * 0.01f;
s.rootNote = 60;
s.startFrame = 30;
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(60, 127); // unity ratio, full velocity, flat gain
std::vector<AudioSample> out;
eng.render(out, 3);
CHECK(approx(out[0], 0.30, 1e-4)); // starts at frame 30, not 0
CHECK(approx(out[1], 0.31, 1e-4)); // advances by unity ratio
CHECK(approx(out[2], 0.32, 1e-4));
}
static void testStartFrameZeroIsUnchanged() {
// startFrame default 0 is exactly the pre-S11 behavior: read begins at frame 0.
SampleData s;
s.frames.resize(20);
for (int i = 0; i < 20; ++i) s.frames[i] = static_cast<float>(i) * 0.05f;
s.rootNote = 60; // startFrame stays 0
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 1);
CHECK(approx(out[0], 0.0, 1e-6)); // frame 0
}
static void testStartFrameOutOfRangeClampsToZero() {
// A start point at/past the sample end degrades to frame 0 (play from the top), never an
// out-of-bounds read that would start the voice already exhausted.
SampleData s = dcSample(10, 60); // 10 frames of 1.0
s.startFrame = 10; // == frameCount: out of range
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 5);
// Reads from frame 0: the DC sample plays its 1.0 body rather than an immediate idle.
CHECK(eng.activeVoiceCount() == 1);
CHECK(approx(out[0], 1.0, 1e-4));
}
static void testStartFrameWithLoop() {
// Start point and loop compose: begin reading mid-sample, then sustain the loop region.
SampleData s;
s.frames.resize(40);
for (int i = 0; i < 40; ++i) s.frames[i] = static_cast<float>(i) * 0.01f;
for (int i = 20; i < 40; ++i) s.frames[i] = 0.5f; // loop body
s.rootNote = 60;
s.startFrame = 10; // begin at frame 10
s.loop.hasLoop = true;
s.loop.start = 20;
s.loop.end = 40;
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 200);
CHECK(approx(out[0], 0.10, 1e-4)); // started at frame 10
CHECK(eng.activeVoiceCount() == 1); // loop sustains it
for (std::size_t i = 60; i < out.size(); ++i) CHECK(approx(out[i], 0.5, 1e-4));
}
static void testStartAfterLoopEndWrapsIntoLoop() {
// Regression (S11 reviewer finding): if startFrame > loop.end (but still < frameCount),
// the voice's initial read head is past the loop end. The wrap-while in renderFrame must
// pull it back into [loopStart, loopEnd) on the very first frame, so the note sounds from
// somewhere inside the loop rather than running off the sample end silently.
//
// Setup: 100-frame sample; loop is [20, 40); startFrame = 60 (past loop.end = 40).
// loop body is a constant 0.5 so every frame inside it reads 0.5.
// After wrap: readPos starts inside [20, 40), first output frame == 0.5.
// Voice must stay active (loop sustains it) and emit the loop value, NOT go silent.
SampleData s;
s.frames.resize(100, 0.0f);
for (int i = 20; i < 40; ++i) s.frames[i] = 0.5f; // loop body
s.rootNote = 60;
s.startFrame = 60; // > loop.end (40), < frameCount (100)
s.loop.hasLoop = true;
s.loop.start = 20;
s.loop.end = 40;
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(60, 127); // unity ratio, full velocity
std::vector<AudioSample> out;
eng.render(out, 50);
// Voice must still be active — the usable loop keeps it alive indefinitely.
CHECK(eng.activeVoiceCount() == 1);
// Every frame after the first wrap must read 0.5 (the loop body). We skip the very
// first frame because the fractional-position wrap lands somewhere in [20,40) and the
// exact offset depends on how many loop lengths fit into 60; what matters is that the
// voice is alive and emitting the loop value, not 0.0 (pre-loop region).
for (std::size_t i = 5; i < out.size(); ++i) {
CHECK(approx(out[i], 0.5, 1e-4));
}
}
// ---------------------------------------------------------------------------
// Gate loop sustain: sample-exact wrapping, the crossfaded seam, and release out.
// ---------------------------------------------------------------------------
// A sample whose every frame carries its own index scaled down, so an observed output value
// names the exact source frame it came from — which is what makes "sample-exact" assertable
// rather than merely plausible.
static SampleData indexSample(std::size_t frames, int rootNote = 60) {
SampleData s;
s.frames.resize(frames);
for (std::size_t i = 0; i < frames; ++i) {
s.frames[i] = static_cast<float>(i) * 0.001f;
}
s.rootNote = rootNote;
return s;
}
// The source frame an output value names, inverted from indexSample's encoding.
static double sourceFrameOf(double out) { return out * 1000.0; }
static void testLoopReadWrapsSampleExactOverManyCycles() {
// Loop [40, 60) over a 100-frame index sample: the head must walk 40..59 and jump back to
// exactly 40, cycle after cycle, with nothing skipped or repeated at the seam.
SampleData s = indexSample(100);
s.loop.hasLoop = true;
s.loop.start = 40;
s.loop.end = 60;
s.play.adsr = flatAdsr();
VoiceEngine eng(1, s);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 200); // 8 full cycles past the loop entry
CHECK(eng.activeVoiceCount() == 1);
for (std::size_t i = 0; i < out.size(); ++i) {
// Output frame i reads source frame i while i < 60, then wraps by 20 each cycle.
const std::int64_t expected = i < 60 ? static_cast<std::int64_t>(i)
: 40 + ((static_cast<std::int64_t>(i) - 60) % 20);
CHECK(approx(sourceFrameOf(out[i]), static_cast<double>(expected), 1e-3));
}
}
static void testZeroCrossfadeLeavesTheSeamHard() {
// With no fade dialled, the frame at the loop end and the frame after it are the raw
// source frames — the step is the whole point of the default, and the whole thing a
// nonzero fade has to smooth.
SampleData s = indexSample(100);
s.loop.hasLoop = true;
s.loop.start = 40;
s.loop.end = 60;
s.loopCrossfadeFrames = 0;
s.play.adsr = flatAdsr();
VoiceEngine eng(1, s);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 80);
CHECK(approx(sourceFrameOf(out[59]), 59.0, 1e-3));
CHECK(approx(sourceFrameOf(out[60]), 40.0, 1e-3)); // hard jump, no blend
}
// The output at output-frame n, for loop [40,60) over the index sample under fade length xf.
static double loopedFrameValue(std::int64_t xf, std::size_t n) {
SampleData s = indexSample(100);
s.loop.hasLoop = true;
s.loop.start = 40;
s.loop.end = 60;
s.loopCrossfadeFrames = xf;
s.play.adsr = flatAdsr();
VoiceEngine eng(1, s);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 80);
return sourceFrameOf(out[n]);
}
static void testCrossfadeBlendsMonotonelyAcrossTheRegion() {
// Loop [40, 60) with an 8-frame pre-seam fade: over output frames 52..59 the read blends
// from source frame n toward source frame n-20 (the same head one loop length back). The
// region ENTRY is exactly continuous — frame 52 carries weight 0 — and every frame in it
// is the exact linear blend, falling monotonically.
SampleData s = indexSample(100);
s.loop.hasLoop = true;
s.loop.start = 40;
s.loop.end = 60;
s.loopCrossfadeFrames = 8;
s.play.adsr = flatAdsr();
VoiceEngine eng(1, s);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 80);
CHECK(approx(sourceFrameOf(out[51]), 51.0, 1e-3));
CHECK(approx(sourceFrameOf(out[52]), 52.0, 1e-3)); // weight 0 at the region edge
for (int n = 52; n < 60; ++n) {
// Normalized over crossfade - 1 (= 7), not crossfade: weight reaches exactly 1 at
// n == 59 (d == 7 == xf - 1), not merely approaching it.
const double w = static_cast<double>(n - 52) / 7.0;
const double want = static_cast<double>(n) * (1.0 - w) + static_cast<double>(n - 20) * w;
CHECK(approx(sourceFrameOf(out[static_cast<std::size_t>(n)]), want, 1e-3));
}
for (int n = 53; n < 60; ++n) {
CHECK(out[static_cast<std::size_t>(n)] < out[static_cast<std::size_t>(n - 1)]);
}
}
// What the crossfade actually buys, measured rather than asserted by adjective: normalizing
// over crossfade - 1 (loop_span.h) lands the weight at exactly 1 on the last rendered frame
// (end - 1) for every REAL crossfade length (xf >= 2), not merely approaching it — that frame
// is the incoming tap outright, which is exactly the value the wrap hands over. The step across
// the seam is therefore the material's own natural one-frame step (indexSample's slope is 1 raw
// frame per frame), constant for any xf >= 2 — not a residual that merely shrinks with a longer
// fade. The |out[60]-out[59]| metric is otherwise a trap: it conflates the natural step with any
// leftover discontinuity, and at xf == loop length the OLD 1/xf normalization happened to score
// 0 by this same metric — a coincidence of that one ratio, not a property of the fix. Comparing
// against the natural step rather than "small" or "zero" closes that hole.
static void testSeamStepMatchesTheNaturalStepForAnyCrossfade() {
auto seamStep = [](std::int64_t xf) {
return std::fabs(loopedFrameValue(xf, 60) - loopedFrameValue(xf, 59));
};
// No crossfade: the seam is the whole loop length less one frame — the wart the fade fixes.
CHECK(approx(seamStep(0), 19.0, 1e-3));
// xf == 1 has no fractional region to blend — its one frame sits exactly at d == 0, caught
// by crossfadeWeight's own d <= 0 floor before the multiply/ceiling ever runs — so it is
// still the hard seam, not a one-frame fade.
CHECK(approx(seamStep(1), 19.0, 1e-3));
// Any REAL crossfade length: the step is exactly the natural one-frame step, not merely
// small — and constant regardless of the fade length, unlike the old formula's proportional
// shrink.
for (std::int64_t xf : {std::int64_t{4}, std::int64_t{8}, std::int64_t{16},
std::int64_t{20}}) {
CHECK(approx(seamStep(xf), 1.0, 1e-3));
}
}
static void testCrossfadeLengthFollowsItsParameter() {
// A longer fade starts earlier and nowhere else: the region begin is end - crossfade, so
// doubling the parameter doubles the number of blended frames.
auto firstFadedFrame = [](std::int64_t xf) {
SampleData s = indexSample(100);
s.loop.hasLoop = true;
s.loop.start = 40;
s.loop.end = 60;
s.loopCrossfadeFrames = xf;
s.play.adsr = flatAdsr();
VoiceEngine eng(1, s);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 70);
for (int n = 40; n < 60; ++n) {
const double got = sourceFrameOf(out[static_cast<std::size_t>(n)]);
if (!approx(got, static_cast<double>(n), 1e-3)) return n;
}
return 60;
};
CHECK(firstFadedFrame(0) == 60); // never diverges from the raw read
CHECK(firstFadedFrame(4) == 57); // region [56,60); frame 56 carries weight 0
CHECK(firstFadedFrame(8) == 53); // region [52,60)
CHECK(firstFadedFrame(16) == 45); // region [44,60)
}
// The fade cannot read before frame 0, so a loop starting at 0 gets none — silently clamped
// rather than reading out of bounds or refusing the loop outright.
static void testCrossfadeIsSuppressedForALoopAtFrameZero() {
SampleData s = indexSample(100);
s.loop.hasLoop = true;
s.loop.start = 0;
s.loop.end = 20;
s.loopCrossfadeFrames = 16;
s.play.adsr = flatAdsr();
VoiceEngine eng(1, s);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 60);
CHECK(eng.activeVoiceCount() == 1);
for (int n = 0; n < 20; ++n) {
CHECK(approx(sourceFrameOf(out[static_cast<std::size_t>(n)]),
static_cast<double>(n), 1e-3));
}
}
static void testNoteOffDuringLoopSustainRunsTheReleaseAndFreesTheVoice() {
// The loop is the SUSTAIN: held, the voice never ends; released, it must leave the
// sustain level down the release ramp and free itself — not keep cycling forever.
SampleData s = dcSample(60, 60); // flat 1.0 so output IS the envelope
s.loop.hasLoop = true;
s.loop.start = 20;
s.loop.end = 40;
s.loopCrossfadeFrames = 4;
AdsrParams a = flatAdsr();
a.releaseFrames = 32;
s.play.adsr = a;
VoiceEngine eng(1, s);
eng.noteOn(60, 127);
std::vector<AudioSample> held;
eng.render(held, 500); // far past the sample end
CHECK(eng.activeVoiceCount() == 1);
CHECK(approx(held.back(), 1.0, 1e-4)); // still at sustain, still looping
eng.noteOff(60);
std::vector<AudioSample> tail;
eng.render(tail, 16);
// Mid-release: audibly decaying, not yet done.
CHECK(tail.back() < 0.75 && tail.back() > 0.0);
CHECK(eng.activeVoiceCount() == 1);
std::vector<AudioSample> rest;
eng.render(rest, 64);
CHECK(eng.activeVoiceCount() == 0);
CHECK(approx(rest.back(), 0.0, 1e-6));
}
// Preserve's contract is LOOP THE SOURCE, SHIFT THE OUTPUT: the loop points name source
// frames, so a transposed note loops the same source span and keeps sounding at level. If the
// span were treated as an output-domain fact, an up-shifted voice would outrun it, freeze its
// shifter tail and decay. Run with and without a crossfade, since the fade is applied to the
// SOURCE feed ahead of the shifter and the ring prime walks the same blend.
static void testPreserveLoopsTheSourceSpanAtEveryTransposition() {
for (std::int64_t xf : {std::int64_t{0}, std::int64_t{16}}) {
for (int note : {48, 60, 72}) {
SampleData s;
s.frames.resize(200, 0.0f);
for (int i = 60; i < 120; ++i) s.frames[i] = 0.5f; // the loop body + its run-in
s.rootNote = 60;
s.sampleRate = 48000;
s.loop.hasLoop = true;
s.loop.start = 80;
s.loop.end = 120;
s.loopCrossfadeFrames = xf;
s.play.adsr = flatAdsr();
s.play.pitchEngine = PitchEngine::Preserve;
VoiceEngine eng(1, s);
eng.noteOn(note, 127);
std::vector<AudioSample> out;
eng.render(out, 4000); // 20x the sample length
// The source span is finite; only the loop can keep a voice alive this long, and
// it does so at every transposition because the span is a source-frame fact.
CHECK(eng.activeVoiceCount() == 1);
// And it is still delivering the loop body, not a decaying frozen tail. The body
// and its run-in are one constant, so the shifter's splices reproduce it whatever
// the shift ratio and whatever the fade blends.
double sum = 0.0;
for (std::size_t i = out.size() - 200; i < out.size(); ++i) sum += out[i];
CHECK(approx(sum / 200.0, 0.5, 0.05));
}
}
}
// ---------------------------------------------------------------------------
// velocity -> volume.
// ---------------------------------------------------------------------------
// S-VIEW-9 BEHAVIOR CHANGE (R10-F1 Option A): the DEFAULT velocity curve is now flat
// y=1, so EVERY velocity plays at unity — NOT the old linear velocity/127. singleSampleChromatic
// builds a zone with the flat default, so the DC-1 sample renders 1.0 at any velocity.
static void testVelocityDefaultCurveIsFlatUnity() {
SampleData km = (dcSample(100, 60)); // DC 1.0, flat default curve
for (int vel : {1, 64, 100, 127}) {
VoiceEngine eng(1, km);
eng.noteOn(60, vel);
std::vector<AudioSample> out;
eng.render(out, 1);
CHECK(approx(out[0], 1.0, 1e-4)); // flat y=1: any velocity -> unity gain
}
}
// A LINEAR curve on the zone reproduces the pre-r10 velocity/127 ramp exactly — proving the curve
// (not a hardcoded map) drives the gain, and that eval is applied at note-on.
static void testVelocityLinearCurveReproducesRamp() {
SampleData km = (dcSample(100, 60)); // DC 1.0
km.velocityCurve = VelocityCurve::linear();
{
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out; eng.render(out, 1);
CHECK(approx(out[0], 1.0, 1e-4));
}
{
VoiceEngine eng(1, km);
eng.noteOn(60, 64);
std::vector<AudioSample> out; eng.render(out, 1);
CHECK(approx(out[0], 64.0 / 127.0, 1e-4));
}
{
VoiceEngine eng(1, km);
eng.noteOn(60, 1);
std::vector<AudioSample> out; eng.render(out, 1);
CHECK(approx(out[0], 1.0 / 127.0, 1e-4));
}
}
// A shaped curve (a single interior knot) drives the gain through eval — a mid velocity reads the
// curve's shaped value, not the linear one. Proves the whole curve, not just the endpoints, applies.
static void testVelocityShapedCurveDrivesGain() {
SampleData km = (dcSample(100, 60)); // DC 1.0
VelocityCurve curve = VelocityCurve::linear();
curve.addPoint(64.0, 0.9); // pull the mid-velocity response UP to 0.9
km.velocityCurve = curve;
VoiceEngine eng(1, km);
eng.noteOn(60, 64);
std::vector<AudioSample> out; eng.render(out, 1);
// At exactly velocity 64 the curve passes through the knot -> gain 0.9 (well above the linear
// 64/127 ~= 0.504), so the rendered DC value is the shaped 0.9.
CHECK(approx(out[0], 0.9, 1e-4));
}
// Two voices summed: polyphony mixes additively.
static void testPolyphonyMixesAdditively() {
SampleData km = (dcSample(100, 60)); // DC 1.0
VoiceEngine eng(4, km);
eng.noteOn(60, 127); // gain 1.0
eng.noteOn(60, 127); // gain 1.0 (second voice, same note)
std::vector<AudioSample> out;
eng.render(out, 1);
CHECK(approx(out[0], 2.0, 1e-4)); // both voices sum
}
// ---------------------------------------------------------------------------
// 7. Stereo channel dimension (S7).
// ---------------------------------------------------------------------------
// A distinct-per-channel stereo DC sample: L = `l`, R = `r` everywhere. A stereo render
// must keep them distinct; a mono render (channel 0 only) sees L.
static SampleData stereoDcSample(std::size_t frames, float l, float r, int rootNote = 60) {
SampleData s;
s.frames.assign(frames, l);
s.framesR.assign(frames, r);
s.rootNote = rootNote;
return s;
}
static void testChannelCount() {
// Mono: framesR empty -> 1 channel. Stereo: matching-length framesR -> 2.
CHECK(dcSample(10, 60).channelCount() == 1);
CHECK(stereoDcSample(10, 1.0f, -1.0f).channelCount() == 2);
// A mismatched framesR length is treated as mono (a bad pair never half-plays).
SampleData bad = dcSample(10, 60);
bad.framesR.assign(5, 0.5f); // wrong length
CHECK(bad.channelCount() == 1);
}
static void testStereoRenderKeepsChannelsDistinct() {
// A stereo sample (L=1.0, R=-1.0) rendered stereo must emit L and R distinctly, each
// scaled by velocity (full here). If the engine copied L to both channels the R check fails.
SampleData km = (stereoDcSample(100, 1.0f, -1.0f, 60));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> left(8, 0.f), right(8, 0.f);
eng.render(left.data(), right.data(), 8);
for (std::size_t i = 0; i < 8; ++i) {
CHECK(approx(left[i], 1.0, 1e-4)); // channel 0
CHECK(approx(right[i], -1.0, 1e-4)); // channel 1 — distinct, NOT a copy of L
}
}
static void testMonoSamplePlaysDualMonoInStereo() {
// A MONO sample rendered through the stereo path plays dual-mono: both channels equal
// (centered), not silent on the right. The cross-mode "mono source in stereo mode" case.
SampleData km = (dcSample(100, 60)); // mono, DC 1.0
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> left(8, 0.f), right(8, 0.f);
eng.render(left.data(), right.data(), 8);
for (std::size_t i = 0; i < 8; ++i) {
CHECK(approx(left[i], 1.0, 1e-4));
CHECK(approx(right[i], 1.0, 1e-4)); // R == L (dual-mono), not 0
}
}
static void testDualMonoStereoSampleRendersCentered() {
// GA Bug 1 (pure-layer proof): a STEREO sample whose two channels are IDENTICAL (a
// dual-mono capture) must render EXACTLY equal L and R — bitwise, every frame — under
// BOTH pitch engines. Any asymmetry here (a silent L, a channel offset, divergent
// shifter state) would pan the output; hard-panned output from a dual-mono capture
// therefore cannot originate in the engine.
auto renderBoth = [](PitchEngine engine, int note) {
SampleData s = sineSample(600, 12.0, 60);
s.framesR = s.frames; // dual-mono: identical channels
s.play.pitchEngine = engine;
SampleData km = (std::move(s));
VoiceEngine eng(1, km, /*preserveCap=*/0, /*preserveWindowFrames=*/128);
eng.noteOn(note, 127);
std::vector<AudioSample> left(256, 0.f), right(256, 0.f);
eng.render(left.data(), right.data(), 256);
bool sound = false, equal = true;
for (std::size_t i = 0; i < left.size(); ++i) {
if (left[i] != 0.0f) sound = true;
if (left[i] != right[i]) equal = false; // EXACT: dual-mono must be centered
}
CHECK(sound); // the render actually produced signal (a 0==0 pass would be vacuous)
CHECK(equal);
};
renderBoth(PitchEngine::Varispeed, 60);
renderBoth(PitchEngine::Varispeed, 67); // off-root: repitch rides both channels equally
renderBoth(PitchEngine::Preserve, 60); // both shifters run (no unity demotion in MIDI)
renderBoth(PitchEngine::Preserve, 67); // off-root Preserve: per-channel shift, same state
}
static void testMonoRenderUnchangedByStereoData() {
// Regression: the mono render path (renderFrame) reads channel 0 ONLY and is byte-identical
// whether or not a second channel is present. A stereo sample rendered mono == its L channel.
SampleData kmS = (stereoDcSample(100, 0.75f, -0.25f, 60));
VoiceEngine engS(1, kmS);
engS.noteOn(60, 127);
std::vector<AudioSample> mono;
engS.render(mono, 8); // the mono overload
for (std::size_t i = 0; i < 8; ++i) CHECK(approx(mono[i], 0.75, 1e-4)); // == L, ignores R
}
static void testStereoRenderAdvancesLikeMonoRepitch() {
// The stereo path must advance the read head by the SAME per-frame ratio as the mono path,
// so repitch is identical. Play a stereo sine (both channels the same signal) an octave up
// and confirm the observed period halves — the mono repitch assertion, on the stereo path.
const std::size_t frames = 8000;
const double cycles = 20.0;
const double nativePeriod = static_cast<double>(frames) / cycles; // 400
SampleData s;
s.frames.resize(frames);
s.framesR.resize(frames);
for (std::size_t i = 0; i < frames; ++i) {
const float v = static_cast<float>(std::sin(2.0 * kPi * cycles *
static_cast<double>(i) / static_cast<double>(frames)));
s.frames[i] = v;
s.framesR[i] = v;
}
s.rootNote = 60;
SampleData km = (std::move(s));
VoiceEngine eng(4, km);
eng.noteOn(72, 127); // +1 octave
std::vector<AudioSample> left(frames / 2, 0.f), right(frames / 2, 0.f);
eng.render(left.data(), right.data(), frames / 2);
CHECK(approx(observedPeriodFrames(left), nativePeriod / 2.0, 2.0));
CHECK(approx(observedPeriodFrames(right), nativePeriod / 2.0, 2.0)); // R repitches identically
}
static void testStereoRenderSumsVoicesPerChannel() {
// Two voices on a stereo sample sum PER CHANNEL (additive polyphony holds in stereo).
SampleData km = (stereoDcSample(100, 0.5f, -0.5f, 60));
VoiceEngine eng(4, km);
eng.noteOn(60, 127);
eng.noteOn(60, 127); // second voice, same note
std::vector<AudioSample> left(1, 0.f), right(1, 0.f);
eng.render(left.data(), right.data(), 1);
CHECK(approx(left[0], 1.0, 1e-4)); // 0.5 + 0.5
CHECK(approx(right[0], -1.0, 1e-4)); // -0.5 + -0.5
}
static void testStereoRenderNullBufferIsNoOp() {
SampleData km = (stereoDcSample(100, 1.0f, -1.0f, 60));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> buf(4, 0.f);
eng.render(nullptr, buf.data(), 4); // null left -> no-op, no crash
eng.render(buf.data(), nullptr, 4); // null right -> no-op
for (float v : buf) CHECK(approx(v, 0.0, 1e-9)); // untouched
}
static void testStereoStartFrameLoopShareOneReadHead() {
// S7 x S11 compose: a STEREO sample with a startFrame AND a sustain loop must read BOTH
// channels from the SAME single read head — one offset, one loop wrap, applied to L and R
// identically (only the sampled value differs). A per-frame L/R ramp that is a fixed offset
// apart (R = L + 0.5) pins the read position on both channels: if the stereo path ever gave
// L and R independent heads, the constant L->R offset would break at the start jump or the
// loop seam.
SampleData s;
s.frames.resize(40);
s.framesR.resize(40);
for (int i = 0; i < 40; ++i) {
s.frames[i] = static_cast<float>(i) * 0.01f; // L: 0.00 .. 0.39
s.framesR[i] = static_cast<float>(i) * 0.01f + 0.5f; // R: L + 0.5, everywhere
}
s.rootNote = 60;
s.startFrame = 10; // begin BOTH channels at frame 10
s.loop.hasLoop = true;
s.loop.start = 20;
s.loop.end = 30; // loop [20,30): frames 20..29
CHECK(s.channelCount() == 2);
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(60, 127); // unity ratio, full velocity, flat gain
std::vector<AudioSample> left(200, 0.f), right(200, 0.f);
eng.render(left.data(), right.data(), 200);
// First frame: both channels start at frame 10 (L=0.10, R=0.60) — the shared start offset.
CHECK(approx(left[0], 0.10, 1e-4));
CHECK(approx(right[0], 0.60, 1e-4));
// The loop sustains the voice indefinitely.
CHECK(eng.activeVoiceCount() == 1);
// At every rendered frame R - L == 0.5 exactly: both channels read the SAME frame index
// (one read head) through the start jump and every loop wrap. A per-channel head drift would
// break this invariant at the seam.
for (std::size_t i = 0; i < left.size(); ++i) {
CHECK(approx(right[i] - left[i], 0.5, 1e-4));
}
// Once fully inside the loop (start=10 -> reaches loop.start=20 within a handful of unity-ratio
// frames), every L value sits in the loop band [0.20, 0.30): the shared head is sustaining the
// loop region on both channels, never running off the sample end.
for (std::size_t i = 15; i < left.size(); ++i) {
CHECK(left[i] >= 0.20 - 1e-4 && left[i] < 0.30 + 1e-4);
}
}
// ===========================================================================
// S15 — sampling modes (Gate AHDSR hold stage, Trigger %-length + fades, note-off immunity).
// ===========================================================================
// --- AHDSR hold stage vs a known signal. ---
static void testAhdsrHoldStageShape() {
// Gate grows a HOLD stage between Attack and Decay: attack 0->1 (5f), HOLD at 1.0 (8f),
// decay 1->0.5 (5f), sustain 0.5. Assert the hold plateau is exactly 1.0 for holdFrames.
AdsrParams p;
p.attackFrames = 5;
p.holdFrames = 8;
p.decayFrames = 5;
p.sustainLevel = 0.5;
p.releaseFrames = 5;
AdsrEnvelope env;
env.configure(p);
env.noteOn();
for (int i = 0; i < 5; ++i) env.tick(); // consume Attack (ends at 1.0)
// The next holdFrames ticks must all be exactly 1.0 (the plateau), stage == Hold.
for (int i = 0; i < 8; ++i) {
CHECK(env.stage() == AdsrEnvelope::Stage::Hold);
CHECK(approx(env.tick(), 1.0, 1e-9));
}
// Then Decay begins, falling from 1.0 toward sustain 0.5.
CHECK(env.stage() == AdsrEnvelope::Stage::Decay);
double v = env.tick();
CHECK(v <= 1.0 + 1e-9 && v >= 0.5 - 1e-9);
}
// --- hold == 0 is byte-identical to the pre-S15 ADSR (back-compat regression). ---
static void testAhdsrHoldZeroEqualsAdsr() {
// The load-bearing back-compat guarantee: hold=0 reproduces the classic ADSR frame-for-frame.
// Assert against a HAND-COMPUTED expected sequence (not another envelope — that would be
// tautological). attack 4, hold 0, decay 4, sustain 0.5. Expected per-tick output:
// Attack: 0/4, 1/4, 2/4, 3/4 (ticks 0..3, level rising 0 -> 0.75)
// Decay: 1.0, then 1.0+(0.5-1)*t for t=1/4..3/4 (ticks 4..7: 1.0, 0.875, 0.75, 0.625)
// Sustain: 0.5 forever (tick 8+)
AdsrParams p;
p.attackFrames = 4;
p.holdFrames = 0; // the degenerate — must NOT insert an extra unity frame
p.decayFrames = 4;
p.sustainLevel = 0.5;
p.releaseFrames = 4;
AdsrEnvelope env;
env.configure(p);
env.noteOn();
const double expected[] = {0.0, 0.25, 0.5, 0.75, // attack
1.0, 0.875, 0.75, 0.625, // decay (first sample 1.0 at t=0)
0.5, 0.5, 0.5}; // sustain
for (double e : expected) CHECK(approx(env.tick(), e, 1e-9));
CHECK(env.stage() == AdsrEnvelope::Stage::Sustain); // reached sustain at the SAME tick count
}
// A trigger-mode DC sample (all 1.0) so a rendered voice's output tracks the trigger envelope
// * velocity directly. `attack`/`decay` are the AHD's ramp lengths in frames, with Hold taking
// the whole remainder — the shape that replaced the retired fade pair. Varispeed so no shift
// colours the amp.
static SampleData triggerSample(std::size_t frames, double lengthFraction,
std::int64_t attack, std::int64_t decay,
std::int64_t startFrame = 0) {
SampleData s = dcSample(frames, 60);
s.startFrame = startFrame;
s.play.playMode = PlayMode::Trigger;
s.play.pitchEngine = PitchEngine::Varispeed; // isolate amp shape from pitch
s.play.trigger.lengthFraction = lengthFraction;
s.play.trigAhd.attackFrames = attack;
s.play.trigAhd.decayFrames = decay;
s.play.trigAhd.holdFraction = 1.0;
return s;
}
// --- Trigger %-length frame math: plays exactly round(frac*(frames-start)) frames then frees. ---
static void testTriggerLengthFractionFrames() {
// 200-frame sample, start 0, 50% length -> plays 100 frames then the voice frees.
SampleData km = (triggerSample(200, 0.5, 0, 0));
VoiceEngine eng(1, km);
eng.noteOn(60, 127); // unity ratio
std::vector<AudioSample> out;
eng.render(out, 200);
// First 100 frames sound (amp>0 for a no-fade trigger = 1.0), then silence + voice freed.
for (std::size_t i = 0; i < 100; ++i) CHECK(out[i] > 0.5f);
for (std::size_t i = 100; i < 200; ++i) CHECK(approx(out[i], 0.0, 1e-6));
CHECK(eng.activeVoiceCount() == 0); // ran off playEnd
}
// --- Trigger start point: %-length measured from the start offset. ---
static void testTriggerLengthWithStart() {
// 200 frames, start 40, 50% -> span 160, play 80 frames (frames 40..119), then free.
SampleData km = (triggerSample(200, 0.5, 0, 0, /*start=*/40));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 200);
for (std::size_t i = 0; i < 80; ++i) CHECK(out[i] > 0.5f);
for (std::size_t i = 80; i < 200; ++i) CHECK(approx(out[i], 0.0, 1e-6));
CHECK(eng.activeVoiceCount() == 0);
}
// --- Trigger fade-in / fade-out ramp shape (equal-power default). ---
// 100 frames, 100% length, attack 20 / decay 20, holdFraction 1.0 — triggerSample() sets no
// curve exponent, so both stages default to util::kCurveNeutral (1.0): the ramps are LINEAR,
// not the retired fade pair's equal-power sin/cos. Asserted against the closed form rather
// than monotonicity alone — a monotonicity-only check is blind to exactly this shape change.
static void testTriggerAhdFadeShape() {
SampleData km = (triggerSample(100, 1.0, 20, 20));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 120);
for (std::size_t i = 0; i < 20; ++i) {
CHECK(approx(out[i], static_cast<double>(i) / 20.0, 1e-3));
}
// Hold plateau at unity.
for (std::size_t i = 20; i < 80; ++i) CHECK(approx(out[i], 1.0, 1e-3));
for (std::size_t i = 80; i < 100; ++i) {
CHECK(approx(out[i], 1.0 - static_cast<double>(i - 80) / 20.0, 1e-3));
}
// Past playEnd = silence.
for (std::size_t i = 100; i < 120; ++i) CHECK(approx(out[i], 0.0, 1e-6));
}
// --- Trigger edge cases: %=0 (immediate free) and fades overlapping (clamped). ---
static void testTriggerEdgeCases() {
// %=0: zero play length -> voice frees at once, no sound.
{
SampleData km = (triggerSample(100, 0.0, 5, 5));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 50);
for (float v : out) CHECK(approx(v, 0.0, 1e-6));
CHECK(eng.activeVoiceCount() == 0);
}
// AHD attack + decay beyond the play length are fitted by fitAhd, not overflowed (no crash, no negative gain, amp in [0,1]).
{
// 40 frames, 100% -> playLen 40; fadeIn 30 + fadeOut 30 = 60 > 40 -> clamped.
SampleData km = (triggerSample(40, 1.0, 30, 30));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 50);
for (std::size_t i = 0; i < 40; ++i) CHECK(out[i] >= -1e-4 && out[i] <= 1.0 + 1e-4);
CHECK(eng.activeVoiceCount() == 0);
}
// %=100 plays the full post-start span.
{
SampleData km = (triggerSample(60, 1.0, 0, 0));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 80);
for (std::size_t i = 0; i < 60; ++i) CHECK(out[i] > 0.5f);
for (std::size_t i = 60; i < 80; ++i) CHECK(approx(out[i], 0.0, 1e-6));
}
}
// --- Trigger out-of-domain lengthFraction: Voice::start's inline copy of trigger_seam's
// formula (map/trigger_seam.h) must clamp the same way the shared formula does — a
// frac > 1.0 never plays past the post-start span, and a NaN frees immediately rather than
// reaching the round()/static_cast<int64_t> undefined on a non-finite value. Pins the second
// implementation directly; test_trigger_seam.cpp pins the first.
static void testTriggerFracAboveOneClampsToSpan() {
// 150% length on a 200-frame sample -> clamped to the full 200-frame post-start span,
// not 300 frames (which would read off the end of the PCM).
SampleData km = (triggerSample(200, 1.5, 0, 0));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 220);
for (std::size_t i = 0; i < 200; ++i) CHECK(out[i] > 0.5f);
for (std::size_t i = 200; i < 220; ++i) CHECK(approx(out[i], 0.0, 1e-6));
CHECK(eng.activeVoiceCount() == 0); // frees exactly at frameCount
}
static void testTriggerFracNaNFreesImmediately() {
SampleData km = (triggerSample(200, std::nan(""), 5, 5));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 50);
for (float v : out) CHECK(approx(v, 0.0, 1e-6));
CHECK(eng.activeVoiceCount() == 0); // playLen 0 -> frees at start, no sound
}
// --- Trigger ignores note-off (S15): the one-shot plays through regardless. ---
static void testTriggerIgnoresNoteOff() {
SampleData km = (triggerSample(200, 0.5, 0, 0));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 10);
eng.noteOff(60); // must be a NO-OP in Trigger
CHECK(eng.activeVoiceCount() == 1); // still sounding after note-off
eng.render(out, 200);
// It still plays its full 100-frame length (frames 10..99 remain > 0 after the note-off).
for (std::size_t i = 10; i < 100; ++i) CHECK(out[i] > 0.5f);
for (std::size_t i = 100; i < 210; ++i) CHECK(approx(out[i], 0.0, 1e-6));
CHECK(eng.activeVoiceCount() == 0); // frees on its own playEnd, not on note-off
}
// ===========================================================================
// S16 — pitch engine (Preserve duration invariance) + pitch envelope (off = identical).
// ===========================================================================
// Render one note to completion (or `maxFrames`) and return the frame count at which the voice
// went idle (the audible LENGTH). A Gate note with a short release + a finite sample runs off.
static std::size_t soundingLength(VoiceEngine& eng, std::size_t maxFrames) {
std::vector<AudioSample> out;
std::size_t len = 0;
for (std::size_t f = 0; f < maxFrames; ++f) {
eng.render(out, 1);
if (eng.activeVoiceCount() > 0) len = f + 1;
else break;
}
return len;
}
// A Preserve-engine one-shot Trigger sample: under Preserve, the %-length wall-clock is stable
// under transpose (the S15xS16 contract). Trigger + Preserve isolates the length measurement from
// Gate's release tail.
static SampleData preserveTriggerSample(std::size_t frames, double lengthFraction) {
SampleData s = dcSample(frames, 60);
s.play.playMode = PlayMode::Trigger;
s.play.pitchEngine = PitchEngine::Preserve;
s.play.trigger.lengthFraction = lengthFraction;
return s;
}
// --- Preserve duration invariance: same note length across +/-12 semitones. ---
static void testPreserveDurationInvariance() {
// A Preserve Trigger at 100% length of a 1000-frame sample plays ~1000 output frames
// regardless of transpose (duration held). Under Varispeed an octave up would halve it.
const std::size_t frames = 1000;
const std::size_t window = 512; // pre-size the shifters
auto lengthAt = [&](int note) -> std::size_t {
SampleData km = (preserveTriggerSample(frames, 1.0));
VoiceEngine eng(1, km, /*preserveCap=*/0, /*window=*/static_cast<std::int64_t>(window));
eng.noteOn(note, 127);
return soundingLength(eng, 4000);
};
const std::size_t atRoot = lengthAt(60);
const std::size_t atUp = lengthAt(72); // +12
const std::size_t atDown = lengthAt(48); // -12
// All three within a small tolerance of the source length (Preserve holds duration). The
// tolerance covers the shifter's fill/latency edge and the terminal ring-out Preserve ends
// on (voice.h's seedTerminalDeclick — bounded by the declick floor at ~185 frames), not a
// duration scaling, which would be 2x.
const std::size_t kTail = 200;
CHECK(atRoot >= frames - 20 && atRoot <= frames + kTail);
CHECK(atUp >= frames - 20 && atUp <= frames + kTail);
CHECK(atDown >= frames - 20 && atDown <= frames + kTail);
// The decisive assertion: the up/down lengths track the root length (NOT halved/doubled).
CHECK(atUp > frames / 2 + 200); // an octave up did NOT halve the duration (Varispeed would)
CHECK(atDown < frames * 2 - 200); // an octave down did NOT double it
}
// --- Varispeed still couples duration (the contrast to Preserve — regression on the old default). ---
static void testVarispeedStillCouplesDuration() {
// A Varispeed Trigger octave up runs off in ~half the frames (pitch & duration coupled).
auto lengthAt = [&](int note) -> std::size_t {
SampleData s = dcSample(1000, 60);
s.play.playMode = PlayMode::Trigger;
s.play.pitchEngine = PitchEngine::Varispeed;
s.play.trigger.lengthFraction = 1.0;
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(note, 127);
return soundingLength(eng, 4000);
};
const std::size_t atRoot = lengthAt(60);
const std::size_t atUp = lengthAt(72);
CHECK(approx(static_cast<double>(atUp), static_cast<double>(atRoot) / 2.0, 30.0));
}
// --- Pitch envelope OFF == bit-identical to the un-modulated engine (regression). ---
static void testPitchEnvOffBitIdentical() {
// Two Varispeed voices, one with a disabled pitch env, one with no pitch env at all. Their
// rendered output must be BIT-IDENTICAL (pitch-env-off applies zero modulation — the S16
// "identical to pre-S16" guarantee). Uses a sine so any pitch drift would show as phase drift.
const std::size_t n = 4000;
auto renderOne = [&](bool withDisabledEnv) -> std::vector<AudioSample> {
SampleData s = sineSample(n, 20.0, 60);
s.play.pitchEngine = PitchEngine::Varispeed;
if (withDisabledEnv) {
s.play.pitchEnv.enabled = false; // explicitly disabled (offset always 0)
s.play.pitchEnv.peakSemitones = 12.0; // a depth that WOULD matter if enabled
s.play.pitchEnv.shape.attackFrames = 0;
s.play.pitchEnv.shape.decayFrames = 500;
}
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(67, 127); // a transposed note so ratio != 1 (exercises the ratio path)
std::vector<AudioSample> out;
eng.render(out, n);
return out;
};
const std::vector<AudioSample> a = renderOne(false);
const std::vector<AudioSample> b = renderOne(true);
CHECK(a.size() == b.size());
bool identical = a.size() == b.size();
for (std::size_t i = 0; i < a.size() && identical; ++i) {
if (a[i] != b[i]) identical = false;
}
CHECK(identical); // disabled pitch env produces the EXACT same samples (no modulation)
}
// --- Pitch envelope ON biases pitch (Varispeed): a positive-peak zero-attack env starts sharp. ---
static void testPitchEnvOnBendsVarispeed() {
// Zero attack + positive peak = "start high, drop to base": the note begins transposed UP and
// settles. Observe the read advancing FASTER at the start (period shorter early) than late.
const std::size_t n = 8000;
SampleData s = sineSample(n, 40.0, 60);
s.play.pitchEngine = PitchEngine::Varispeed;
s.play.pitchEnv.enabled = true;
s.play.pitchEnv.shape.attackFrames = 0; // start at the peak
s.play.pitchEnv.shape.decayFrames = 3000; // glide to base over 3000 frames
s.play.pitchEnv.peakSemitones = 12.0; // +1 octave at t=0
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(60, 127); // at root -> base ratio 1.0; the env supplies the bend
std::vector<AudioSample> out;
eng.render(out, 4000);
// Early period (heavily transposed up) should be shorter than the late period (settled).
std::vector<AudioSample> early(out.begin(), out.begin() + 800);
std::vector<AudioSample> late(out.begin() + 3200, out.begin() + 4000);
const double pe = observedPeriodFrames(early);
const double pl = observedPeriodFrames(late);
CHECK(pe > 0.0 && pl > 0.0);
CHECK(pe < pl); // pitch dropped over time (period lengthened) -> the AD env bent the pitch
}
// --- Compose: engine x mode x stereo x loop (a Preserve Gate loop in stereo sounds + sustains). ---
static void testPreserveGateStereoLoopComposes() {
// A STEREO sample, GATE mode, PRESERVE engine, with a sustain loop. It must sound on BOTH
// channels and sustain (the loop keeps the voice alive) — S7 x S15 x S16 all composing.
SampleData s;
const std::size_t frames = 400;
s.frames.resize(frames);
s.framesR.resize(frames);
for (std::size_t i = 0; i < frames; ++i) {
const float v = static_cast<float>(std::sin(2.0 * kPi * 8.0 *
static_cast<double>(i) / static_cast<double>(frames)));
s.frames[i] = v;
s.framesR[i] = v * 0.5f; // R is a distinct (half-amplitude) channel
}
s.rootNote = 60;
s.loop.hasLoop = true;
s.loop.start = 100;
s.loop.end = 300;
s.play.playMode = PlayMode::Gate;
s.play.pitchEngine = PitchEngine::Preserve;
CHECK(s.channelCount() == 2);
SampleData km = (std::move(s));
VoiceEngine eng(1, km, 0, 512);
eng.noteOn(67, 127); // transposed up a fifth under Preserve (duration held)
std::vector<AudioSample> left(2000, 0.f), right(2000, 0.f);
eng.render(left.data(), right.data(), 2000);
// The loop sustains the voice well past the sample length (400 frames) -> still active.
CHECK(eng.activeVoiceCount() == 1);
// Both channels carry signal (some frame has non-trivial magnitude on each).
double maxL = 0.0, maxR = 0.0;
for (std::size_t i = 600; i < 2000; ++i) {
if (std::fabs(left[i]) > maxL) maxL = std::fabs(left[i]);
if (std::fabs(right[i]) > maxR) maxR = std::fabs(right[i]);
}
CHECK(maxL > 0.05);
CHECK(maxR > 0.02); // R present (half amplitude), distinct from L -> stereo preserved
// Loop-never-freezes regression: the Preserve voice must NOT spuriously freeze when the
// read tap hits the sample end and the loop wraps it back. A spuriously frozen voice
// stops writing to the ring and the looped tail would go silent past the sample end.
// Check a block far past the sample end (sample = 400 frames; window = 512; well past
// any single-pass tail region) to catch any wrap-before-exhaustion ordering error.
double maxLFar = 0.0;
for (std::size_t i = 1800; i < 2000; ++i) {
if (std::fabs(left[i]) > maxLFar) maxLFar = std::fabs(left[i]);
}
CHECK(maxLFar > 0.05); // still alive at frame 1800 (4.5× the 400-frame sample length)
}
// --- Preserve voice cap: a Preserve note-on past the cap is dropped; Varispeed unaffected. ---
// EVERY engine Preserve voice (root included) runs the shifter and counts toward the cap —
// the FA1 unity demotion is gone (it was scoped to the retired PreviewCard).
static void testPreserveVoiceCap() {
SampleData s = dcSample(2000, 60);
s.play.pitchEngine = PitchEngine::Preserve; // held (Gate, no loop -> runs long enough)
SampleData km = (std::move(s));
// 8 voices total, Preserve cap of 2.
VoiceEngine eng(8, km, /*preserveCap=*/2, /*window=*/256);
CHECK(eng.noteOn(62, 127) != VoiceEngine::kNoVoice); // 1st Preserve voice
CHECK(eng.noteOn(64, 127) != VoiceEngine::kNoVoice); // 2nd Preserve voice (at the cap)
CHECK(eng.noteOn(65, 127) == VoiceEngine::kNoVoice); // 3rd DROPPED by the Preserve cap
CHECK(eng.activeVoiceCount() == 2);
}
// ---------------------------------------------------------------------------
// FA1 postscript — the Preserve unity-Varispeed bypass is REMOVED with the PreviewCard
// (preview is now a plain engine noteOn at the root). The engine keeps the shifter at
// EVERY Preserve note; GA2's primed ring speaks on frame 0 at every ratio, so onset is
// uniformly IMMEDIATE across the keyboard with no demotion path at all.
// ---------------------------------------------------------------------------
// The ENGINE'S root-note Preserve voice keeps the OLA path — and since the GA2 prime fix the
// primed shifter speaks on frame 0 at EVERY ratio (the ring holds the first window of real
// source, not warm-up zeros). Uniform onset across the keyboard now means uniformly IMMEDIATE:
// the root and a transposed neighbor both open at full level on the very first frames.
static void testPreserveUnityEngineVoiceSpeaksImmediately() {
auto earlyAndLate = [](int note, double& early, double& late) {
SampleData s = dcSample(4000, 60);
s.play.pitchEngine = PitchEngine::Preserve;
s.play.adsr = flatAdsr(); // isolate the shifter onset from the amp attack
SampleData km = (std::move(s));
VoiceEngine eng(1, km, /*preserveCap=*/0, /*window=*/512);
eng.noteOn(note, 127);
std::vector<AudioSample> out;
eng.render(out, 1500);
early = 1e9;
for (std::size_t i = 0; i < 8; ++i) {
early = (std::min)(early, static_cast<double>(std::fabs(out[i])));
}
late = 0.0;
for (std::size_t i = 600; i < 1500; ++i) {
late = (std::max)(late, static_cast<double>(std::fabs(out[i])));
}
};
double earlyRoot = 0.0, lateRoot = 0.0, earlyUp = 0.0, lateUp = 0.0;
earlyAndLate(60, earlyRoot, lateRoot); // at root: unity shift — NO demotion in the engine
earlyAndLate(62, earlyUp, lateUp); // +2 st: a real shift, same immediate onset
CHECK(earlyRoot > 0.9); // primed ring: full level from frame 0 (no fill silence)
CHECK(earlyUp > 0.9); // ...uniformly across the keyboard (the GA2 onset-gap fix)
CHECK(lateRoot > 0.9);
CHECK(lateUp > 0.9); // and no gaps later either (splices land in real history)
}
// A TRANSPOSED Preserve note keeps the genuine OLA path — and since the GA2 prime fix that
// path has NO onset cost: the ring is primed with the first window of real source, so a
// transposed voice opens at full level on frame 0 (the DAW "zero-sample gaps in the first few
// ms" regression) and NEVER dips while the source sustains (splices land in real history, not
// warm-up zeros).
static void testPreserveTransposedVoiceSpeaksImmediately() {
SampleData s = dcSample(4000, 60);
s.play.pitchEngine = PitchEngine::Preserve;
s.play.adsr = flatAdsr(); // isolate the shifter onset from the amp attack
SampleData km = (std::move(s));
VoiceEngine eng(1, km, /*preserveCap=*/0, /*window=*/512);
eng.noteOn(62, 127); // +2 semitones: a real shift, NOT demoted
std::vector<AudioSample> out;
eng.render(out, 1500);
// Full level from the very first frame (a DC source through complementary crossfades and
// aligned splices holds 1.0 throughout) — pre-fix the first ~window was fill silence.
double lo = 1e9;
for (std::size_t i = 0; i < 1500; ++i) {
lo = (std::min)(lo, static_cast<double>(std::fabs(out[i])));
}
CHECK(lo > 0.9);
}
// Phase S re-scope consequence: a ROOT-note engine Preserve voice keeps its shifter, so it
// COUNTS toward the Preserve cap like any other (pre-re-scope it was demoted and exempt).
static void testPreserveUnityVoiceCountsTowardCap() {
SampleData s = dcSample(2000, 60);
s.play.pitchEngine = PitchEngine::Preserve;
SampleData km = (std::move(s));
VoiceEngine eng(8, km, /*preserveCap=*/2, /*window=*/256);
CHECK(eng.noteOn(60, 127) != VoiceEngine::kNoVoice); // root: a genuine Preserve voice now
CHECK(eng.noteOn(62, 127) != VoiceEngine::kNoVoice); // 2nd (at the cap)
CHECK(eng.noteOn(64, 127) == VoiceEngine::kNoVoice); // 3rd DROPPED by the Preserve cap
CHECK(eng.activeVoiceCount() == 2);
}
// FA1 bug 3a regression, in the DAW's ACTUAL configuration: the velocity curve must drive the
// gain under the PRESERVE product-default engine with a CONFIGURED shifter window (every prior
// velocity test ran the bare Varispeed core). A linear y=x curve at velocity 1 must be
// near-silent — NOT max volume.
static void testVelocityCurveAppliesUnderPreserve() {
auto steadyLevelAt = [&](int vel) -> double {
SampleData s = dcSample(4000, 60);
s.play.pitchEngine = PitchEngine::Preserve;
SampleData km = (std::move(s));
km.velocityCurve = VelocityCurve::linear();
VoiceEngine eng(1, km, /*preserveCap=*/0, /*window=*/256);
eng.noteOn(62, vel); // transposed: the genuine shifter path (not the unity demotion)
std::vector<AudioSample> out;
eng.render(out, 1000);
return static_cast<double>(out[900]); // steady state: ring is fully DC by frame 256
};
CHECK(approx(steadyLevelAt(127), 1.0, 0.02));
CHECK(approx(steadyLevelAt(64), 64.0 / 127.0, 0.02));
CHECK(steadyLevelAt(1) < 0.02); // y=x at velocity 1: near-silent, the Daniel repro case
}
// --- Per-zone A/D/S/R actually reaches the voice envelope (S12). ---
//
// Every AHDSR field rides on SampleData.play.adsr (frames, resolved from the stored seconds at
// keymap build); the engine holds no instrument-wide ADSR. These two tests assert that path.
// The zone's attackFrames drives the envelope ramp. Strategy: put an explicit 10-frame attack on
// the SampleData.play.adsr. If Voice::start reads the zone ADSR, the DC-1 output will be 0 at frame
// 0 and 1.0 after the 10-frame ramp; a voice that ignored the zone ADSR (instant) would already be
// 1.0 at frame 0. This is the load-bearing proof.
static void testPerZoneAdsrReachesVoiceEnvelope() {
SampleData s = dcSample(500, 60);
// Per-zone attack = 10 frames, zero decay, sustain 1.0, zero release.
s.play.adsr.attackFrames = 10;
s.play.adsr.holdFrames = 0;
s.play.adsr.decayFrames = 0;
s.play.adsr.sustainLevel = 1.0;
s.play.adsr.releaseFrames = 0;
s.play.pitchEngine = PitchEngine::Varispeed; // isolate from pitch engine machinery
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(60, 127); // unity pitch, full velocity -> gain 1.0
std::vector<AudioSample> out;
eng.render(out, 20);
// Frame 0: attack start, envelope near 0. A voice ignoring the zone ADSR would read 1.0 here.
CHECK(approx(out[0], 0.0, 1e-9)); // env still at bottom of ramp
// Frame 9: still ramping (last attack frame, linear ramp reaches 0.9).
CHECK(out[9] < 1.0 - 1e-9);
// Frame 10+: attack complete, sustain at 1.0.
CHECK(approx(out[10], 1.0, 1e-9));
CHECK(approx(out[19], 1.0, 1e-9));
}
// Default-valued zone (AdsrParams all zeros) is behavior-identical to the pre-fix flat path.
// A zero-init AdsrParams (attackFrames=0, decayFrames=0, sustainLevel=1.0, releaseFrames=0) must
// yield an instant-attack/instant-sustain voice — frame 0 immediately at 1.0. This preserves the
// back-compat invariant: an old zone with no A/D/S/R storage sounds the same as before.
static void testZeroAdsrIsInstantSustain() {
SampleData s = dcSample(20, 60);
// Default AdsrParams{}: all zeros, sustainLevel = 1.0 (struct default). No attack ramp.
s.play.adsr = AdsrParams{};
s.play.pitchEngine = PitchEngine::Varispeed;
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 5);
// All frames must be 1.0: zero attack + sustain 1.0 = instantly at full level.
for (std::size_t i = 0; i < out.size(); ++i) CHECK(approx(out[i], 1.0, 1e-9));
}
// ---------------------------------------------------------------------------
// Phase S — parameterized voice count, MONO mode (last-note held stack, Retrigger/Legato),
// and the isolated PREVIEW CARD.
// ---------------------------------------------------------------------------
// A DC sample at `level` with a flat (instant, fully-open) envelope — rendered output equals
// level * velocity gain, so WHICH sample is sounding is directly observable in the mix.
static SampleData dcLevelSample(std::size_t frames, float level, int rootNote) {
SampleData s;
s.frames.assign(frames, level);
s.rootNote = rootNote;
s.play.adsr = flatAdsr();
return s;
}
// The mono tests need to read WHICH NOTE holds the single voice off the rendered value, and
// a DC sample makes pitch inaudible. Velocity is the discriminator: a DC 1.0 capture with a
// curve pinned through two probe velocities renders 0.25 for a kVelLow strike and 0.75 for a
// kVelHigh one (the Hermite spline passes exactly through its control points). Each test
// then presses note 50 soft and note 70 hard, so the level names the sounding note.
static constexpr int kVelLow = 32;
static constexpr int kVelHigh = 96;
static SampleData twoLevelSample() {
SampleData km = dcLevelSample(200000, 1.0f, 60);
km.velocityCurve = VelocityCurve::fromPoints({{0.0, 0.0},
{static_cast<double>(kVelLow), 0.25},
{static_cast<double>(kVelHigh), 0.75},
{127.0, 1.0}},
instrument::engine::CurveDomain::Unipolar);
return km;
}
// The rendered value on the next frame — one-frame probe of "what is sounding right now".
static double probeFrame(VoiceEngine& eng) {
std::vector<AudioSample> out;
eng.render(out, 1);
return static_cast<double>(out[0]);
}
// MONO last-note priority: a new note TAKES the single voice; releasing the top note falls
// back to the most-recent still-held note; releasing the last note gates off. Also: mono uses
// ONE voice regardless of the pool size.
static void testMonoLastNotePriorityAndFallback() {
SampleData km = twoLevelSample();
VoiceEngine eng(4, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger);
CHECK(eng.noteOn(50, kVelLow) == 0); // zone A sounds
CHECK(approx(probeFrame(eng), 0.25, 1e-6));
CHECK(eng.noteOn(70, kVelHigh) == 0); // zone B TAKES the voice (last-note priority)
CHECK(eng.activeVoiceCount() == 1); // mono: one voice even with 4 in the pool
CHECK(approx(probeFrame(eng), 0.75, 1e-6));
eng.noteOff(70); // top released -> FALLBACK to still-held 50
CHECK(approx(probeFrame(eng), 0.25, 1e-6));
eng.noteOff(50); // last finger up -> gate off (release 0 = instant)
CHECK(approx(probeFrame(eng), 0.0, 1e-9));
CHECK(eng.activeVoiceCount() == 0);
}
// Releasing a LOWER held note (not the sounding one) changes nothing audible; the released
// note also leaves the stack, so the final note-off truly empties it.
static void testMonoReleaseOfLowerHeldNoteIsInaudible() {
SampleData km = twoLevelSample();
VoiceEngine eng(4, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger);
eng.noteOn(50, kVelLow);
eng.noteOn(70, kVelHigh); // 70 sounds, 50 held beneath
eng.noteOff(50); // releasing the buried note: inaudible
CHECK(approx(probeFrame(eng), 0.75, 1e-6));
eng.noteOff(70); // 50 already left the stack -> silence, no fallback
CHECK(approx(probeFrame(eng), 0.0, 1e-9));
}
// Re-pressing a HELD note moves it to the top of the stack (it sounds again), and the note
// beneath becomes the fallback.
static void testMonoRepressHeldNoteMovesToTop() {
SampleData km = twoLevelSample();
VoiceEngine eng(4, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger);
eng.noteOn(50, kVelLow);
eng.noteOn(70, kVelHigh);
CHECK(eng.noteOn(50, kVelLow) == 0); // re-press while held: back on top
CHECK(approx(probeFrame(eng), 0.25, 1e-6));
eng.noteOff(50); // falls back to 70 (now the most recent held)
CHECK(approx(probeFrame(eng), 0.75, 1e-6));
eng.noteOff(70);
CHECK(approx(probeFrame(eng), 0.0, 1e-9));
}
// A RETRIGGER fallback re-strikes the fallen-back-to note at ITS ORIGINAL velocity (kept per
// held note on the stack), not the departing note's.
static void testMonoRetriggerFallbackUsesOriginalVelocity() {
SampleData s = dcLevelSample(200000, 1.0f, 60);
SampleData km = (std::move(s));
km.velocityCurve = VelocityCurve::linear(); // gain = velocity/127
VoiceEngine eng(4, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger);
eng.noteOn(60, 32); // soft first note
CHECK(approx(probeFrame(eng), 32.0 / 127.0, 1e-4));
eng.noteOn(64, 127); // loud takeover
CHECK(approx(probeFrame(eng), 1.0, 1e-6));
eng.noteOff(64); // fallback re-strikes 60 at ITS velocity (32)
CHECK(approx(probeFrame(eng), 32.0 / 127.0, 1e-4));
}
// An OUT-OF-RANGE note in mono is a defined no-play: it consumes nothing, never joins the
// stack (so it can never take the voice back on a fallback), and its note-off is inert. The
// stack keys notes as uint8, so an unguarded 200 would alias onto 72 and corrupt it.
static void testMonoOutOfRangeNeverJoinsStack() {
SampleData km = twoLevelSample();
VoiceEngine eng(4, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger);
eng.noteOn(70, kVelHigh);
CHECK(eng.noteOn(200, 127) == VoiceEngine::kNoVoice); // past the MIDI range
CHECK(eng.activeVoiceCount() == 1);
CHECK(approx(probeFrame(eng), 0.75, 1e-6)); // 70 undisturbed
eng.noteOff(200); // inert
CHECK(approx(probeFrame(eng), 0.75, 1e-6));
eng.noteOff(70);
CHECK(approx(probeFrame(eng), 0.0, 1e-9));
}
// RETRIGGER restarts the amplitude envelope on a mono takeover: mid-attack level drops back
// to the ramp's origin when the new note takes the voice.
static void testMonoRetriggerRestartsEnvelope() {
SampleData s = dcLevelSample(200000, 1.0f, 60);
s.play.adsr.attackFrames = 100; // slow linear attack: level at frame i = i/100
SampleData km = (std::move(s));
VoiceEngine eng(2, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 50); // mid-attack: level ~0.49 at frame 49
CHECK(approx(out[49], 0.49, 1e-6));
eng.noteOn(62, 127); // takeover: envelope RESTARTS
CHECK(approx(probeFrame(eng), 0.0, 1e-6)); // back at the attack origin
}
// LEGATO keeps the envelope running through a same-sample takeover: pitch moves, NO re-attack.
static void testMonoLegatoContinuesEnvelope() {
SampleData s = dcLevelSample(200000, 1.0f, 60);
s.play.adsr.attackFrames = 100;
SampleData km = (std::move(s));
VoiceEngine eng(2, km, 0, 0, VoiceMode::Mono, MonoTrigger::Legato);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 50);
CHECK(approx(out[49], 0.49, 1e-6));
eng.noteOn(62, 127); // legato takeover: envelope KEEPS running
CHECK(approx(probeFrame(eng), 0.50, 1e-6)); // frame 50 of the SAME attack ramp
}
// LEGATO retunes without restarting the read head, and the velocity gain stays the FIRST
// note's (a legato phrase is one gesture, one strike). Observed on a ramp sample: values
// continue from the current read position at the NEW pitch ratio; a soft second strike does
// not duck the level.
static void testMonoLegatoRetunesWithoutReadRestart() {
SampleData s;
s.frames.resize(200000);
for (std::size_t i = 0; i < s.frames.size(); ++i) {
s.frames[i] = static_cast<float>(i); // ramp: output value == read position
}
s.rootNote = 60;
s.play.adsr = flatAdsr();
SampleData km = (std::move(s));
km.velocityCurve = VelocityCurve::linear();
VoiceEngine eng(2, km, 0, 0, VoiceMode::Mono, MonoTrigger::Legato);
eng.noteOn(60, 127); // unity: read advances 1/frame, full gain
std::vector<AudioSample> out;
eng.render(out, 10);
CHECK(approx(out[9], 9.0, 1e-4));
eng.noteOn(72, 1); // legato to +1 octave at a WHISPER velocity
CHECK(approx(probeFrame(eng), 10.0, 1e-3)); // read CONTINUES at 10 — no restart, gain kept
CHECK(approx(probeFrame(eng), 12.0, 1e-3)); // and now advances at ratio 2 (the new pitch)
}
// LEGATO takeover ALWAYS glides now: with one loaded capture there is no second PCM stream
// to cross into, so the read head never has to restart mid-phrase. (The retired
// cross-sample-restart branch was the multi-zone case.)
static void testMonoLegatoAlwaysGlidesWithinThePhrase() {
SampleData km = twoLevelSample();
km.play.adsr.attackFrames = 100; // a slow attack would expose any restart
VoiceEngine eng(2, km, 0, 0, VoiceMode::Mono, MonoTrigger::Legato);
eng.noteOn(50, kVelLow);
std::vector<AudioSample> out;
eng.render(out, 50); // mid-attack: level ~0.49 * the 0.25 vel gain
CHECK(approx(out[49], 0.49 * 0.25, 1e-6));
eng.noteOn(70, kVelHigh); // takeover: envelope KEEPS running, no re-attack
// Frame 50 of the SAME attack ramp, still at the FIRST strike's velocity gain (a legato
// phrase is one gesture, one strike) — NOT 0.0 (a restart) and NOT 0.75 (a re-strike).
CHECK(approx(probeFrame(eng), 0.50 * 0.25, 1e-6));
}
// LEGATO after the last note was RELEASED re-attacks: a releasing voice's note has left the
// stack, so the next press is a fresh phrase, not a takeover.
static void testMonoLegatoAfterReleaseReattacks() {
SampleData s = dcLevelSample(200000, 1.0f, 60);
s.play.adsr.attackFrames = 100;
s.play.adsr.releaseFrames = 1000; // long release keeps the voice audibly ringing
SampleData km = (std::move(s));
VoiceEngine eng(2, km, 0, 0, VoiceMode::Mono, MonoTrigger::Legato);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 150); // through the attack: at full level
eng.noteOff(60); // release begins (stack now empty)
out.clear();
eng.render(out, 10);
eng.noteOn(62, 127); // a NEW phrase: re-attacks even in Legato
CHECK(approx(probeFrame(eng), 0.0, 1e-6)); // fresh attack origin, not the ringing level
}
// MONO does not apply the S16 Preserve cap: a single voice runs at most one shifter — a
// Preserve->Preserve takeover must never be dropped by the cap.
static void testMonoIgnoresPreserveCap() {
SampleData s = dcSample(4000, 60);
s.play.pitchEngine = PitchEngine::Preserve;
SampleData km = (std::move(s));
VoiceEngine eng(4, km, /*preserveCap=*/1, /*window=*/256,
VoiceMode::Mono, MonoTrigger::Retrigger);
CHECK(eng.noteOn(62, 127) == 0); // 1st Preserve note: at the cap
CHECK(eng.noteOn(64, 127) == 0); // takeover NOT dropped (poly cap would drop it)
CHECK(eng.activeVoiceCount() == 1);
}
// A ramp sample (output value == read position) so a re-attack (read restarts at 0) is
// directly distinguishable from a legato retune (read continues) on the rendered value.
static SampleData rampSample(std::size_t frames, int rootNote) {
SampleData s;
s.frames.resize(frames);
for (std::size_t i = 0; i < frames; ++i) s.frames[i] = static_cast<float>(i);
s.rootNote = rootNote;
s.play.adsr = flatAdsr();
return s;
}
// --- velocity -> pitch ---------------------------------------------------------
//
// A ramp sample reads its own position, so the value on frame N IS the accumulated read rate:
// the transpose is directly observable rather than inferred from a spectrum.
static void testVelocityPitchIsExactlyOffByDefault() {
// The modulation VALUE at every velocity, not a rendered approximation of it: the default
// bipolar curve must yield the identity ratio exactly, so nothing detunes by a hair.
const PlayParams def;
for (int v = 0; v <= 127; ++v) {
CHECK(velocityPitchRatio(def.pitchVelocityCurve, v) == 1.0);
}
// And at the render: two strikes of very different velocity read the ramp identically.
SampleData s = rampSample(4096, 60);
Voice soft;
Voice hard;
soft.start(60, 1, s);
hard.start(60, 127, s);
for (int i = 0; i < 64; ++i) CHECK(soft.renderFrame() == hard.renderFrame());
}
static void testDrawnVelocityPitchCurveTransposesBothWays() {
using instrument::engine::CurveDomain;
SampleData s = rampSample(4096, 60);
const double full = std::pow(2.0, kVelocityPitchRangeSemitones / 12.0);
// A curve pinned at +1 across the domain: every velocity transposes UP by the full scale.
s.play.pitchVelocityCurve =
VelocityCurve::fromPoints({{0.0, 1.0}, {127.0, 1.0}}, CurveDomain::Bipolar);
Voice up;
up.start(60, 100, s);
CHECK(approx(static_cast<double>(up.renderFrame()), 0.0, 1e-9));
CHECK(approx(static_cast<double>(up.renderFrame()), full, 1e-4));
// Pinned at -1: DOWN by the same scale — the half of the domain the old unipolar curve
// could not express at all.
s.play.pitchVelocityCurve =
VelocityCurve::fromPoints({{0.0, -1.0}, {127.0, -1.0}}, CurveDomain::Bipolar);
Voice down;
down.start(60, 100, s);
down.renderFrame();
CHECK(approx(static_cast<double>(down.renderFrame()), 1.0 / full, 1e-4));
// And it follows the curve: a rising ramp gives a soft hit less transpose than a hard one.
s.play.pitchVelocityCurve =
VelocityCurve::fromPoints({{0.0, 0.0}, {127.0, 1.0}}, CurveDomain::Bipolar);
Voice q;
Voice f;
q.start(60, 20, s);
f.start(60, 120, s);
q.renderFrame();
f.renderFrame();
const double quiet = static_cast<double>(q.renderFrame());
const double loud = static_cast<double>(f.renderFrame());
CHECK(quiet > 1.0);
CHECK(loud > quiet);
CHECK(loud < full); // velocity 120 is short of the +1 endpoint
}
// MAJOR-1 regression: MONO+LEGATO with a TRIGGER zone RE-ATTACKS after the last key is up.
// Trigger ignores note-off (Voice::release() is a no-op, so releasing_ never latches), so a
// legato guard keyed on `active && !releasing` saw a ringing one-shot as "still held" and
// silently RETUNED it in place. The correct predicate is the HELD-STACK depth: with no other
// key down, the next note is a fresh phrase and must restart the read head.
static void testMonoLegatoTriggerReattacksAfterKeyUp() {
SampleData s = rampSample(200000, 60);
s.play.playMode = PlayMode::Trigger; // default TriggerParams: full length, no fades
SampleData km = (std::move(s));
VoiceEngine eng(2, km, 0, 0, VoiceMode::Mono, MonoTrigger::Legato);
eng.noteOn(60, 127); // unity: read advances 1/frame
std::vector<AudioSample> out;
eng.render(out, 10);
eng.noteOff(60); // Trigger ignores the gate: keeps ringing...
CHECK(approx(probeFrame(eng), 10.0, 1e-4)); // ...read head still advancing past 10
eng.noteOn(62, 127); // NO key held -> fresh phrase: RE-ATTACK
CHECK(approx(probeFrame(eng), 0.0, 1e-4)); // read RESTARTED at 0 (a retune would read ~11)
// And it is genuinely playing from the top at the new pitch (ratio 2^(2/12) ~ 1.1225),
// not merely silent: the next frame reads at the advanced position.
CHECK(approx(probeFrame(eng), std::pow(2.0, 2.0 / 12.0), 1e-3));
}
// Companion boundary: with another key STILL physically held, a same-sample Trigger takeover
// under Legato still RETUNES (read continues) — the held-stack predicate matches the old
// behavior everywhere except the ringing-but-unheld case above.
static void testMonoLegatoTriggerHeldKeyStillRetunes() {
SampleData s = rampSample(200000, 60);
s.play.playMode = PlayMode::Trigger;
SampleData km = (std::move(s));
VoiceEngine eng(2, km, 0, 0, VoiceMode::Mono, MonoTrigger::Legato);
eng.noteOn(60, 127);
std::vector<AudioSample> out;
eng.render(out, 10);
eng.noteOn(62, 127); // 60 still held -> legato takeover
CHECK(approx(probeFrame(eng), 10.0, 1e-4)); // read CONTINUES at 10 — no re-attack
}
// MAJOR-2: allNotesOff releases every gated poly voice (flat release -> instant silence).
static void testAllNotesOffReleasesPolyVoices() {
SampleData s = dcLevelSample(200000, 1.0f, 60);
SampleData km = (std::move(s));
VoiceEngine eng(4, km);
eng.noteOn(60, 127);
eng.noteOn(62, 127);
eng.noteOn(64, 127);
CHECK(eng.activeVoiceCount() == 3);
eng.allNotesOff();
CHECK(approx(probeFrame(eng), 0.0, 1e-9)); // all gated off (release 0 = instant)
CHECK(eng.activeVoiceCount() == 0);
}
// MAJOR-2, the STUCK-NOTE path: allNotesOff clears the mono held stack, so a phantom entry
// (simulating a LOST note-off) can never be resurrected by the fallback afterwards.
static void testAllNotesOffClearsMonoHeldStack() {
SampleData km = twoLevelSample();
VoiceEngine eng(4, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger);
eng.noteOn(50, kVelLow); // 50's note-off will never arrive (phantom)
eng.noteOn(70, kVelHigh); // 70 sounds, phantom 50 buried on the stack
eng.allNotesOff(); // PANIC
CHECK(approx(probeFrame(eng), 0.0, 1e-9));
CHECK(eng.activeVoiceCount() == 0);
// The stack is empty: a fresh press + release gates off cleanly, with NO fallback
// restart of the phantom (pre-fix, noteOff(70) here re-struck 50 -> 0.25 forever).
eng.noteOn(70, kVelHigh);
CHECK(approx(probeFrame(eng), 0.75, 1e-6));
eng.noteOff(70);
CHECK(approx(probeFrame(eng), 0.0, 1e-9));
CHECK(eng.activeVoiceCount() == 0);
}
// CC 120 (allSoundsOff) hard-stops a ringing TRIGGER one-shot that would otherwise play to
// its bounded playEnd (minutes on a full-length capture). This is the primary repro: allNotesOff
// (CC 123) is a NO-OP on a Trigger voice — only allSoundsOff provides the actual hard stop.
static void testAllSoundsOffStopsTriggerOneShot() {
// A Trigger sample with a very long play length (all-1 DC, flat velocity). After noteOn the
// voice is active and ringing; allSoundsOff must silence it immediately.
SampleData s = dcLevelSample(200000, 1.0f, 60);
s.play.playMode = PlayMode::Trigger;
s.play.trigger.lengthFraction = 1.0; // full length — would ring for 200000 frames
SampleData km = (std::move(s));
VoiceEngine eng(1, km);
eng.noteOn(60, 127);
CHECK(eng.activeVoiceCount() == 1);
// CC 123 (release) must be a NO-OP on a Trigger voice — the one-shot plays through.
eng.allNotesOff();
CHECK(eng.activeVoiceCount() == 1); // still ringing (Trigger ignores release)
std::vector<AudioSample> out;
eng.render(out, 1);
CHECK(out[0] > 0.5f); // still sounding
// CC 120 (hard-stop) must silence it instantly.
eng.allSoundsOff();
CHECK(eng.activeVoiceCount() == 0); // immediately idle
out.clear();
eng.render(out, 1);
CHECK(approx(out[0], 0.0, 1e-9)); // silent
}
// CC 123 (allNotesOff) still releases Gate voices — the existing release behavior is unchanged.
static void testAllNotesOffStillReleasesGateVoices() {
SampleData s = dcLevelSample(200000, 1.0f, 60);
// Default Gate mode, instant release (releaseFrames 0).
SampleData km = (std::move(s));
VoiceEngine eng(4, km);
eng.noteOn(60, 127);
eng.noteOn(62, 127);
CHECK(eng.activeVoiceCount() == 2);
eng.allNotesOff();
std::vector<AudioSample> out;
eng.render(out, 1);
CHECK(approx(out[0], 0.0, 1e-9)); // Gate with 0-release: instant silence
CHECK(eng.activeVoiceCount() == 0);
}
// MONO LEGATO same-note re-press (one-held-note edge case): with only that note on the
// stack, heldCount_ after the re-push is 1 (not >= 2), so it falls through to re-attack
// rather than retune. This is the correct fresh-phrase behavior documented in the comment.
static void testMonoLegatoSameNoteRepressReattacks() {
SampleData s = rampSample(200000, 60);
SampleData km = (std::move(s));
VoiceEngine eng(2, km, 0, 0, VoiceMode::Mono, MonoTrigger::Legato);
eng.noteOn(60, 127); // first press; read starts at 0
std::vector<AudioSample> out;
eng.render(out, 10); // advance the read head to ~10
// Re-press the SAME note while it is the only held note: heldCount_ after removeHeld+push = 1
// -> does NOT satisfy heldCount_ >= 2 -> re-attack (not a legato retune).
eng.noteOn(60, 127);
CHECK(approx(probeFrame(eng), 0.0, 1e-4)); // read RESTARTED at 0 (re-attack, not retune)
}
// GREEN: out-of-range notes are rejected at BOTH mono entry points. The held stack stores
// uint8, so an unguarded off for note 256 (== 0 mod 256) would alias-evict held note 0 —
// losing its fallback. Note-ons out of [0,127] are a defined no-play.
static void testMonoOutOfRangeNotesRejected() {
SampleData s = dcLevelSample(200000, 1.0f, 60);
SampleData km = (std::move(s));
VoiceEngine eng(2, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger);
CHECK(eng.noteOn(128, 127) == VoiceEngine::kNoVoice);
CHECK(eng.noteOn(-1, 127) == VoiceEngine::kNoVoice);
CHECK(eng.activeVoiceCount() == 0);
eng.noteOn(0, 127); // hold the aliasing target (note 0)
eng.noteOn(62, 127); // 62 takes the voice; 0 held beneath
eng.noteOff(256); // MUST NOT alias-evict held note 0
eng.noteOff(-256); // likewise for the negative wrap
CHECK(approx(probeFrame(eng), 1.0, 1e-6)); // 62 undisturbed
eng.noteOff(62); // falls back to STILL-HELD note 0
CHECK(approx(probeFrame(eng), 1.0, 1e-6)); // (alias-evicted pre-fix -> silence here)
eng.noteOff(0);
CHECK(approx(probeFrame(eng), 0.0, 1e-9));
}
// The user-parameterized polyphony bound: an N-voice engine holds exactly N simultaneous
// notes and steals (never grows) on the N+1th; 0 clamps to the documented 1-voice degenerate.
static void testVoiceCountBoundsPolyphony() {
SampleData s = dcLevelSample(200000, 1.0f, 60);
SampleData km = (std::move(s));
VoiceEngine e3(3, km);
CHECK(e3.maxVoices() == 3);
e3.noteOn(60, 127);
e3.noteOn(62, 127);
e3.noteOn(64, 127);
CHECK(e3.activeVoiceCount() == 3);
e3.noteOn(65, 127); // 4th: steals within the pool
CHECK(e3.activeVoiceCount() == 3);
VoiceEngine e1(1, km);
e1.noteOn(60, 127);
e1.noteOn(62, 127);
CHECK(e1.activeVoiceCount() == 1); // 1-voice pool: every note steals the one voice
VoiceEngine e0(0, km);
CHECK(e0.maxVoices() == 1); // documented degenerate: clamped to 1
}
// GA declick (bug 2): a MONO Retrigger TAKEOVER hard-cuts the sounding tone (read head +
// envelope restart in one frame) — pre-fix the output stepped from the old level to the new
// attack's ~0 in one sample, the audible click. With the engine's takeoverDeclick opt-in
// the boundary frame carries the old level and every later frame moves by a bounded small
// delta while the compensation decays under the new attack.
static void testMonoRetrigTakeoverDeclicksRestart() {
SampleData s = dcSample(200000, 60);
s.play.adsr.attackFrames = 100; // real attack: the new tone starts near 0
s.play.adsr.sustainLevel = 1.0;
s.play.adsr.releaseFrames = 0;
SampleData km = (std::move(s));
VoiceEngine eng(1, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger,
/*takeoverDeclick=*/true);
eng.noteOn(60, 127);
std::vector<AudioSample> pre;
eng.render(pre, 200); // past the attack: sustained at 1.0
CHECK(approx(pre.back(), 1.0, 1e-6));
eng.noteOn(64, 127); // Retrigger takeover: hard restart
std::vector<AudioSample> post;
eng.render(post, 400);
// No step at the boundary: the first post-takeover frame still carries the old level
// (pre-fix it was the new attack's ~0 — a full-scale step).
CHECK(approx(post[0], 1.0, 0.06));
// Bounded slope everywhere across the takeover: max per-frame delta is the declick decay
// step (~0.05) + the attack slope (0.01), never a click-sized jump.
double prev = static_cast<double>(pre.back());
double maxDelta = 0.0;
for (AudioSample v : post) {
const double d = std::fabs(static_cast<double>(v) - prev);
if (d > maxDelta) maxDelta = d;
prev = static_cast<double>(v);
}
CHECK(maxDelta < 0.07);
// The compensation dies out: the tail is the new note's sustain alone.
CHECK(approx(post.back(), 1.0, 1e-3));
}
// Peer restart site (peer-symmetry): the Retrigger FALLBACK on note-off — the most-recent
// still-held note re-strikes the voice — is the same hard cut and gets the same declick.
static void testMonoRetrigFallbackDeclicksRestart() {
SampleData s = dcSample(200000, 60);
s.play.adsr.attackFrames = 100;
s.play.adsr.sustainLevel = 1.0;
s.play.adsr.releaseFrames = 0;
SampleData km = (std::move(s));
VoiceEngine eng(1, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger,
/*takeoverDeclick=*/true);
eng.noteOn(60, 127);
std::vector<AudioSample> a;
eng.render(a, 400); // 60 sustains at 1.0
eng.noteOn(64, 127); // takeover (declicked, settles back to 1.0)
std::vector<AudioSample> b;
eng.render(b, 400);
CHECK(approx(b.back(), 1.0, 1e-3));
eng.noteOff(64); // FALLBACK re-strikes held 60 — hard restart
std::vector<AudioSample> post;
eng.render(post, 400);
CHECK(approx(post[0], 1.0, 0.06)); // boundary carries the old level, no step
double prev = static_cast<double>(b.back());
double maxDelta = 0.0;
for (AudioSample v : post) {
const double d = std::fabs(static_cast<double>(v) - prev);
if (d > maxDelta) maxDelta = d;
prev = static_cast<double>(v);
}
CHECK(maxDelta < 0.07);
CHECK(approx(post.back(), 1.0, 1e-3));
}
// The declick is TAKEOVER-only: a fresh mono start (idle voice — first note of a phrase, or
// a re-press after a full gate-off) must NOT ramp from a stale last output; the attack starts
// at ~0 exactly as before.
static void testMonoDeclickOnlyOnTakeover() {
SampleData s = dcSample(200000, 60);
s.play.adsr.attackFrames = 100;
s.play.adsr.sustainLevel = 1.0;
s.play.adsr.releaseFrames = 0;
SampleData km = (std::move(s));
VoiceEngine eng(1, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger,
/*takeoverDeclick=*/true);
// First note of the phrase: no phantom compensation, attack from ~0.
eng.noteOn(60, 127);
CHECK(probeFrame(eng) < 0.02);
std::vector<AudioSample> a;
eng.render(a, 400); // sustain 1.0 (lastOut is now nonzero)
// Full gate-off (release 0 -> instant idle): the next start is FRESH, not a takeover.
eng.noteOff(60);
std::vector<AudioSample> gap;
eng.render(gap, 4);
CHECK(approx(gap.back(), 0.0, 1e-9));
eng.noteOn(62, 127);
CHECK(probeFrame(eng) < 0.02); // no declick from the stale last output
}
// Peer restart site (peer-symmetry): a POLY at-cap STEAL is the same hard cut as the mono
// retrig takeover — read head + envelope restart on a SOUNDING voice — and gets the same
// declick ramp. Pool of 1 makes the steal deterministic: the second note-on must steal the
// only (sounding) voice, and with the opt-in the boundary carries the old level instead of
// stepping to the new attack's ~0.
static void testPolyStealDeclicksRestart() {
SampleData s = dcSample(200000, 60);
s.play.adsr.attackFrames = 100;
s.play.adsr.sustainLevel = 1.0;
s.play.adsr.releaseFrames = 0;
SampleData km = (std::move(s));
VoiceEngine eng(1, km, 0, 0, VoiceMode::Poly, MonoTrigger::Retrigger,
/*takeoverDeclick=*/true);
eng.noteOn(60, 127);
std::vector<AudioSample> pre;
eng.render(pre, 200); // past the attack: sustained at 1.0
CHECK(approx(pre.back(), 1.0, 1e-6));
CHECK(eng.noteOn(64, 127) != VoiceEngine::kNoVoice); // at cap: steals the sounding voice
std::vector<AudioSample> post;
eng.render(post, 400);
CHECK(approx(post[0], 1.0, 0.06)); // boundary carries the old level, no step
double prev = static_cast<double>(pre.back());
double maxDelta = 0.0;
for (AudioSample v : post) {
const double d = std::fabs(static_cast<double>(v) - prev);
if (d > maxDelta) maxDelta = d;
prev = static_cast<double>(v);
}
CHECK(maxDelta < 0.07);
CHECK(approx(post.back(), 1.0, 1e-3)); // compensation dies out; new note sustains
}
// SAME-BLOCK double takeover: two steals of the same voice with NO frame rendered between
// (a two-note chord arriving at cap in one block). The second start() must re-seed the ramp
// from the same pre-cut output level — if start() zeroed lastOut, the pending ramp would be
// dropped and the click would return on exactly this edge.
static void testSameBlockDoubleTakeoverKeepsDeclickSeed() {
SampleData s = dcSample(200000, 60);
s.play.adsr.attackFrames = 100;
s.play.adsr.sustainLevel = 1.0;
s.play.adsr.releaseFrames = 0;
SampleData km = (std::move(s));
VoiceEngine eng(1, km, 0, 0, VoiceMode::Poly, MonoTrigger::Retrigger,
/*takeoverDeclick=*/true);
eng.noteOn(60, 127);
std::vector<AudioSample> pre;
eng.render(pre, 200); // sustained at 1.0
CHECK(approx(pre.back(), 1.0, 1e-6));
eng.noteOn(64, 127); // steal #1 (no render yet)
eng.noteOn(67, 127); // steal #2, same block
std::vector<AudioSample> post;
eng.render(post, 400);
CHECK(approx(post[0], 1.0, 0.06)); // seed survived the double restart
double prev = static_cast<double>(pre.back());
double maxDelta = 0.0;
for (AudioSample v : post) {
const double d = std::fabs(static_cast<double>(v) - prev);
if (d > maxDelta) maxDelta = d;
prev = static_cast<double>(v);
}
CHECK(maxDelta < 0.07);
CHECK(approx(post.back(), 1.0, 1e-3));
}
// Declick with a ZERO-ATTACK takeover onto the SAME DC level: the difference seed is
// (old level new first raw output) = (1.0 1.0) = 0, so NOTHING is added — output stays
// exactly full scale, never above it. (This is the case the retired rev-1 (1 amp) gate
// existed for: an ADDITIVE old-level ramp under an instant-unity attack summed to +6 dB.
// The difference seed makes the blip structurally impossible without any gate — and without
// the gate's fatal hole that kept the click on every instant-unity restart.)
static void testZeroAttackTakeoverNeverExceedsFullScale() {
SampleData s = dcSample(200000, 60);
s.play.adsr.attackFrames = 0; // zero-attack: amp == 1 on the very first frame
s.play.adsr.sustainLevel = 1.0;
s.play.adsr.releaseFrames = 0;
SampleData km = (std::move(s));
VoiceEngine eng(1, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger,
/*takeoverDeclick=*/true);
eng.noteOn(60, 127);
std::vector<AudioSample> pre;
eng.render(pre, 200); // sustained at 1.0
CHECK(approx(pre.back(), 1.0, 1e-6));
eng.noteOn(64, 127); // zero-attack takeover: amp hits 1 on frame 0
std::vector<AudioSample> post;
eng.render(post, 400);
// Every output frame must stay within [-1, 1]: no +6 dB blip.
for (AudioSample v : post) {
CHECK(v <= 1.0f + 1e-4f && v >= -1.0f - 1e-4f);
}
// The zero-attack note settles at sustain 1.0 immediately.
CHECK(approx(post[0], 1.0, 1e-4));
}
// Shared discontinuity probe for the GA2 click tests: max sample-to-sample delta from the
// last pre-restart frame across the whole post-restart span. A hard cut shows up as a
// click-sized step (~ the old instantaneous level); a properly declicked restart moves by
// the signal's own slope plus the ≤5%-of-seed decay step per frame.
static double maxDeltaAcross(double lastPre, const std::vector<AudioSample>& post) {
double prev = lastPre;
double maxDelta = 0.0;
for (AudioSample v : post) {
const double d = std::fabs(static_cast<double>(v) - prev);
if (d > maxDelta) maxDelta = d;
prev = static_cast<double>(v);
}
return maxDelta;
}
// GA2 — the click that SURVIVED the rev-1 declick (DAW report: "mono retrigger STILL
// CLICKS"): a mono Retrigger takeover of a TRIGGER zone. Trigger with no fade-in is at FULL
// amplitude on frame 0, so the rev-1 compensation — gated by (1 amp) — was zeroed exactly
// here and the restart still hard-cut from the old instantaneous level (~1.0 at the sine
// peak) to the new onset's 0. The difference-seeded declick reproduces the old level on the
// boundary frame and bounds every later delta. A sine (not DC) so the test sees the real
// waveform-value jump the DC-sample rev-1 tests masked.
static void testMonoRetrigTriggerZoneDeclicksRestart() {
SampleData s = sineSample(48000, 100.0, 60); // period 480 frames; slope <= ~0.013/frame
s.play.playMode = PlayMode::Trigger; // default fades: NO fade-in -> amp 1 at frame 0
SampleData km = (std::move(s));
VoiceEngine eng(1, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger,
/*takeoverDeclick=*/true);
eng.noteOn(60, 127);
std::vector<AudioSample> pre;
eng.render(pre, 120); // quarter period: ringing at ~ the sine peak
CHECK(pre.back() > 0.99f); // the cut level is large — a real click pre-fix
eng.noteOn(60, 127); // hammer the same key: Retrigger takeover
std::vector<AudioSample> post;
eng.render(post, 400);
// Boundary continuity: the first post-restart frame reproduces the old level (pre-fix it
// stepped to the new onset's sin(0) == 0 — a full-scale discontinuity).
CHECK(std::fabs(static_cast<double>(post[0]) - static_cast<double>(pre.back())) < 0.01);
// Bounded slope across the whole restart: decay step (<= 0.05 of the seed) + sine slope.
CHECK(maxDeltaAcross(static_cast<double>(pre.back()), post) < 0.08);
}
// GA2 peer: the same gate hole on a GATE zone with ZERO attack (amp == 1 on frame 0 — the
// default AdsrParams, and any user-dialed instant attack). Rev-1's (1 amp) gate zeroed the
// compensation here too; the difference seed closes it identically.
static void testZeroAttackGateRetrigNoStep() {
SampleData s = sineSample(48000, 100.0, 60);
s.play.adsr.attackFrames = 0; // instant-unity attack
s.play.adsr.sustainLevel = 1.0;
s.play.adsr.releaseFrames = 0;
SampleData km = (std::move(s));
VoiceEngine eng(1, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger,
/*takeoverDeclick=*/true);
eng.noteOn(60, 127);
std::vector<AudioSample> pre;
eng.render(pre, 120); // ringing at ~ the sine peak
CHECK(pre.back() > 0.99f);
eng.noteOn(60, 127); // zero-attack Retrigger takeover
std::vector<AudioSample> post;
eng.render(post, 400);
CHECK(std::fabs(static_cast<double>(post[0]) - static_cast<double>(pre.back())) < 0.01);
CHECK(maxDeltaAcross(static_cast<double>(pre.back()), post) < 0.08);
}
// PREVIEW re-audition declick (GA2, re-homed on the engine): the preview is now a plain
// engine noteOn at the root, so re-auditioning while the first preview still rings is a
// POLY AT-CAP STEAL when the pool is saturated — with takeoverDeclick opted in (the shell's
// product default), the restart runs the same difference-seeded ramp: boundary continuity
// + bounded slope. Same physics testPolyStealDeclicksRestart pins; this pins it at the
// preview's exact shape (same note, root, full pool of 1).
static void testPreviewReauditionDeclicksViaEngineSteal() {
SampleData s = sineSample(48000, 100.0, 60); // default ADSR: instant unity (worst case)
SampleData km = (std::move(s));
VoiceEngine eng(1, km, 0, 0, VoiceMode::Poly, MonoTrigger::Retrigger,
/*takeoverDeclick=*/true);
eng.noteOn(60, 127); // preview: root note through the pool
std::vector<AudioSample> pre;
eng.render(pre, 120); // ringing at ~ the sine peak
CHECK(pre.back() > 0.99f);
eng.noteOn(60, 127); // audition again: at-cap steal restart
std::vector<AudioSample> post;
eng.render(post, 400);
CHECK(std::fabs(static_cast<double>(post[0]) - static_cast<double>(pre.back())) < 0.01);
CHECK(maxDeltaAcross(static_cast<double>(pre.back()), post) < 0.08);
}
// GA-VoiceSteal repro (DAW bug): voiceCount 3, a triad note-on'd at the SAME sample time
// (three note-ons in one block, no render between), then a 4th note. The steal must take
// EXACTLY ONE voice (the oldest, none releasing) and leave the other two RINGING — the DAW
// symptom was every tone cutting out. Configured like the live instrument: Preserve engine
// (product default), a real OLA window, sine PCM, default-ish AHDSR (3 ms attack, sustain 1,
// 60 ms release), rendered stereo between events like process() does.
static void testOverCapChordStealsExactlyOne() {
SampleData s = sineSample(96000, 2000.0, 60); // ~2 s at 48k
s.play.adsr.attackFrames = 144; // 3 ms @ 48k
s.play.adsr.sustainLevel = 1.0;
s.play.adsr.releaseFrames = 2880; // 60 ms @ 48k
s.play.pitchEngine = PitchEngine::Preserve;
SampleData km = (std::move(s));
// Mirrors the processor: kPreserveVoiceCap = 8, 50 ms OLA window at 48k = 2400 frames.
VoiceEngine eng(3, km, /*preserveVoiceCap=*/8, /*preserveWindowFrames=*/2400);
// The chord: three note-ons at one sample time (same block, no render between).
CHECK(eng.noteOn(60, 100) != VoiceEngine::kNoVoice);
CHECK(eng.noteOn(64, 100) != VoiceEngine::kNoVoice);
CHECK(eng.noteOn(67, 100) != VoiceEngine::kNoVoice);
CHECK(eng.activeVoiceCount() == 3);
// Ring for a while (stereo, like the negotiated bus) — all three still sounding and finite.
std::vector<AudioSample> l(4800, 0.0f), r(4800, 0.0f);
eng.render(l.data(), r.data(), l.size());
CHECK(eng.activeVoiceCount() == 3);
bool finite = true;
for (AudioSample v : l) { if (!std::isfinite(v)) { finite = false; break; } }
CHECK(finite);
// The 4th note: must steal exactly ONE voice (the oldest = note 60) — never all.
CHECK(eng.noteOn(62, 100) != VoiceEngine::kNoVoice);
CHECK(eng.activeVoiceCount() == 3);
// Note 60 was the stolen one: its note-off finds no voice (count unchanged after the
// release window). Notes 64 and 67 must still hold their voices — each note-off drops
// the count by one once the 60 ms release tail has run out.
std::fill(l.begin(), l.end(), 0.0f); std::fill(r.begin(), r.end(), 0.0f);
eng.noteOff(60);
eng.render(l.data(), r.data(), l.size()); // 4800 frames > 2880 release
CHECK(eng.activeVoiceCount() == 3); // 60 no longer owns a voice: no-op
eng.noteOff(64);
std::fill(l.begin(), l.end(), 0.0f); std::fill(r.begin(), r.end(), 0.0f);
eng.render(l.data(), r.data(), l.size());
CHECK(eng.activeVoiceCount() == 2); // 64 was still ringing — ONE voice released
eng.noteOff(67);
std::fill(l.begin(), l.end(), 0.0f); std::fill(r.begin(), r.end(), 0.0f);
eng.render(l.data(), r.data(), l.size());
CHECK(eng.activeVoiceCount() == 1); // 67 was still ringing too
eng.noteOff(62);
std::fill(l.begin(), l.end(), 0.0f); std::fill(r.begin(), r.end(), 0.0f);
eng.render(l.data(), r.data(), l.size());
CHECK(eng.activeVoiceCount() == 0); // the stolen-into 4th note releases last
}
// PREVIEW OBEYS VOICING (the PreviewCard-isolation reversal): a preview is a plain engine
// noteOn, so it is a REAL pool voice — a full pool STEALS for it (never a parallel voice on
// top), it counts toward activeVoiceCount, and its note-off releases through the normal
// path. This is the processor's mailbox-drain contract, pinned in the pure core.
static void testPreviewNoteObeysVoicing() {
SampleData s = dcLevelSample(200000, 1.0f, 60);
SampleData km = (std::move(s));
VoiceEngine eng(2, km);
eng.noteOn(60, 127);
eng.noteOn(62, 127); // the pool is now FULL
eng.noteOn(64, 127); // the preview note: steals — no third voice
CHECK(eng.activeVoiceCount() == 2);
std::vector<AudioSample> buf(4, 0.0f);
eng.render(buf.data(), buf.size());
CHECK(approx(buf[0], 2.0, 1e-6)); // TWO voices sum — never 3 (no side-car)
eng.noteOff(64); // flat release: the preview gates off
std::vector<AudioSample> buf2(1, 0.0f);
eng.render(buf2.data(), buf2.size());
CHECK(eng.activeVoiceCount() == 1); // the surviving MIDI note still rings
CHECK(approx(buf2[0], 1.0, 1e-6));
}
// PREVIEW JOINS THE MONO HELD STACK: a preview routed through the real note path is a mono
// stack entry like any host note — it TAKES the single voice on press (last-note priority)
// and its release FALLS BACK to the still-held host note instead of cutting to silence.
// Pins the processor's mailbox-drain contract for Mono the way testPreviewNoteObeysVoicing
// pins it for Poly steal.
static void testPreviewNoteJoinsMonoHeldStack() {
SampleData km = twoLevelSample();
VoiceEngine eng(4, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger);
CHECK(eng.noteOn(50, kVelLow) == 0); // the host-MIDI note: zone A sounds
CHECK(approx(probeFrame(eng), 0.25, 1e-6));
CHECK(eng.noteOn(70, kVelHigh) == 0); // the preview press: TAKES the voice
CHECK(eng.activeVoiceCount() == 1); // still mono — the preview is no side-car
CHECK(approx(probeFrame(eng), 0.75, 1e-6));
eng.noteOff(70); // preview release: FALLBACK to the held note
CHECK(approx(probeFrame(eng), 0.25, 1e-6));
eng.noteOff(50); // host note up: gate off (flat release = instant)
CHECK(approx(probeFrame(eng), 0.0, 1e-9));
CHECK(eng.activeVoiceCount() == 0);
}
// PREVIEW NOTE-OFF ROUTES TO THE DRAIN ENGINE: mirror of process()'s dual-engine off
// routing. A preview held across a reload leaves its ringing voice in the DISPLACED
// (draining) snapshot while the fresh live engine has no voice at that pitch. The off is
// sent to BOTH — exactly what the mailbox drain does: the fresh engine must safely no-op,
// the drain engine must release its voice (otherwise the old-snapshot preview would
// sustain until the next reload hard-cut it).
static void testPreviewNoteOffRoutesToDrainEngine() {
SampleData km = twoLevelSample();
VoiceEngine drainEng(2, km); // was live when the preview fired
VoiceEngine liveEng(2, km); // the post-reload fresh snapshot: no voices
CHECK(drainEng.noteOn(70, kVelHigh) != VoiceEngine::kNoVoice);
CHECK(approx(probeFrame(drainEng), 0.75, 1e-6)); // the preview rings in the old snapshot
CHECK(liveEng.activeVoiceCount() == 0);
// The preview release, drained to BOTH engines like a host note-off:
liveEng.noteOff(70);
drainEng.noteOff(70);
CHECK(approx(probeFrame(liveEng), 0.0, 1e-9)); // fresh engine: safe no-op, stays silent
CHECK(liveEng.activeVoiceCount() == 0);
CHECK(approx(probeFrame(drainEng), 0.0, 1e-9)); // flat release: gates off NOW
CHECK(drainEng.activeVoiceCount() == 0); // the old-snapshot voice released
}
// GA2 — bounded-blend overshoot regression: mid-ramp output must stay within full scale.
//
// Construction of the worst case (§1 reviewer finding): retrig a sine at a point where the
// pre-cut level is ~1.0 (old ref ≈ 1). The new voice starts at sin(0) == 0, so the OLD
// frozen-seed declick adds (ref x₀) ≈ 1.0 to the compensation. The new sine has a short
// period (8 frames) so outₙ reaches ~1.0 again within just 2 frames; at that moment the
// frozen seed is still ~0.9 → outₙ + seed ≈ 1.9, roughly +3.8 dB over full scale.
//
// The bounded blend keeps every frame within max(|ref|, |outCurrent|) ≤ 1.0 + tol — this
// test must FAIL against the rev-2 frozen-seed code and PASS with the bounded blend.
static void testDeclickBoundedBlendNoOvershoot() {
// A sine with 8-frame period so it peaks within the declick ramp window (~80 frames).
// 48000 frames, 6000 cycles -> period = 8 frames; quarter period = 2 frames = the peak.
const std::size_t kFrames = 48000;
const double kCycles = 6000.0; // period = 8 frames
SampleData s = sineSample(kFrames, kCycles, 60);
s.play.playMode = PlayMode::Trigger; // no fade-in -> amp 1 on frame 0 (worst case)
SampleData km = (std::move(s));
VoiceEngine eng(1, km, 0, 0, VoiceMode::Mono, MonoTrigger::Retrigger,
/*takeoverDeclick=*/true);
// Start a voice and render to a quarter period so the sine is near its positive peak.
// Period = 48000/6000 = 8 frames. Frame index 2 = sin(2π*6000*2/48000) = sin(π/2) = 1.0.
// We render 3 frames (indices 0,1,2 are visited: readPos 0→1→2→3) so pre[2] reads
// frame index 2 at the sine peak.
eng.noteOn(60, 127);
std::vector<AudioSample> pre;
eng.render(pre, 3); // frame index 2 (read on third render): sin(pi/2) ≈ 1.0
CHECK(pre.back() > 0.99f); // at peak: ref ≈ 1.0 when we cut
// Retrigger: hard restart at sin(0) == 0, ref == ~1.0. The frozen-seed approach would
// add ~0.9 to a new output of ~1.0 two frames later → ~1.9. The bounded blend must not.
eng.noteOn(60, 127);
const double tol = 1e-3;
std::vector<AudioSample> post;
eng.render(post, 200); // 200 frames covers the full ramp (~80 frames at kDeclickDecay=0.95)
for (std::size_t i = 0; i < post.size(); ++i) {
const double v = static_cast<double>(post[i]);
CHECK(v <= 1.0 + tol && v >= -1.0 - tol);
}
// Boundary identity: first frame must reproduce the pre-cut level (±small tol).
CHECK(std::fabs(static_cast<double>(post[0]) - static_cast<double>(pre.back())) < 0.01);
}
// ===========================================================================
// GA3 — Preserve tail wind-down: the final window (and the release riding over it) must be
// a gap-free tone. The GA2 tail clamp HELD THE LAST REAL SAMPLE as the shifter feed once the
// source ran out — a DC plateau with no waveform to correlate on. Splices landing in (or
// referenced against) that region were unalignable, so the tap alternated real-tone / dead-DC
// at the splice cadence: the DAW "periodic troughs, almost like ring modulation, stronger
// toward the end, ~1:20 tone-to-silence at the very end". GA3 freezes the WRITER instead
// (padding never enters the ring) and lets the aligned-splice machinery recycle the frozen
// real tail — these tests render to the natural end and assert the tone survives.
// ===========================================================================
// A sine at explicit per-index frequency f0 (cycles/frame). Period is chosen NON-INTEGER
// (splice alignment must earn the sub-sample fit) but dividing `frames` exactly, so the
// source ENDS at a zero crossing — the held-DC value the GA2 clamp would feed is ~0, making
// the pre-GA3 dead stretches measurable as near-silence.
static SampleData tailSine(std::size_t frames, double f0, int rootNote = 60) {
SampleData s;
s.frames.resize(frames);
for (std::size_t i = 0; i < frames; ++i) {
s.frames[i] = static_cast<float>(std::sin(2.0 * kPi * f0 * static_cast<double>(i)));
}
s.rootNote = rootNote;
return s;
}
// Longest run of consecutive frames with |x| < thresh in [from, to).
static std::size_t worstQuietRun(const std::vector<AudioSample>& out, std::size_t from,
std::size_t to, double thresh) {
std::size_t worst = 0, run = 0;
for (std::size_t i = from; i < to && i < out.size(); ++i) {
if (std::fabs(static_cast<double>(out[i])) < thresh) {
++run;
if (run > worst) worst = run;
} else {
run = 0;
}
}
return worst;
}
// Peak |x| over [from, from+len).
static double blockPeak(const std::vector<AudioSample>& out, std::size_t from, std::size_t len) {
double peak = 0.0;
for (std::size_t i = from; i < from + len && i < out.size(); ++i) {
const double a = std::fabs(static_cast<double>(out[i]));
if (a > peak) peak = a;
}
return peak;
}
// --- Gate no-loop, held to the natural end: the FINAL WINDOW carries the full-amplitude
// tone with no gaps, at up- AND down-shifts. Pre-GA3 this window chopped (RED without
// the writer freeze: quiet runs of hundreds of frames, block peaks collapsing to ~0.04). ---
static void testPreserveTailFinalWindowGapFree() {
const std::size_t frames = 8192;
const std::size_t w = 1024;
const double f0 = 1.0 / 163.84; // 50 exact cycles over 8192: ends at a zero crossing
const int notes[] = {67, 55}; // +7 st (ratio ~1.50) and -5 st (ratio ~0.75)
for (int note : notes) {
SampleData s = tailSine(frames, f0, 60);
s.play.pitchEngine = PitchEngine::Preserve; // Gate, no loop -> runs to the sample end
s.play.adsr = flatAdsr(); // held: amp 1 to the end (isolates the DSP)
SampleData km = (std::move(s));
VoiceEngine eng(1, km, /*preserveCap=*/0, static_cast<std::int64_t>(w));
eng.noteOn(note, 127);
std::vector<AudioSample> out;
eng.render(out, frames); // the voice frees exactly at the natural end
// (a) No dead stretches: a unit sine at period ~164/ratio dwells below 0.05 for only
// a few frames per zero crossing; the pre-GA3 DC stretches ran hundreds.
CHECK(worstQuietRun(out, frames - w, frames, 0.05) < 24);
// (b) Full amplitude to the very end: every 128-frame block in the final window spans
// more than a half period at both ratios, so a clean tone peaks near 1.0 in each.
for (std::size_t b = frames - w; b + 128 <= frames; b += 128) {
CHECK(blockPeak(out, b, 128) > 0.5);
}
}
}
// --- Gate release OVER the final window: the envelope scales amplitude smoothly; the
// underlying tone must stay continuous (no chop) while it fades. Adjacent-block peaks
// may only decay envelope-fast, never gap-fast. ---
static void testPreserveTailReleaseContinuous() {
const std::size_t frames = 8192;
const std::size_t w = 1024;
const double f0 = 1.0 / 163.84;
SampleData s = tailSine(frames, f0, 60);
s.play.pitchEngine = PitchEngine::Preserve;
s.play.adsr = flatAdsr();
s.play.adsr.releaseFrames = static_cast<std::int64_t>(w); // release spans the final window
SampleData km = (std::move(s));
VoiceEngine eng(1, km, /*preserveCap=*/0, static_cast<std::int64_t>(w));
eng.noteOn(67, 127);
std::vector<AudioSample> out;
eng.render(out, frames - w); // sustain up to one window before the end...
eng.noteOff(67); // ...then release exactly over the final window
eng.render(out, w);
// First release block still near full level; thereafter each 128-frame block may lose at
// most envelope-rate level vs its predecessor (linear release loses 12.5% of full scale
// per block). A pre-GA3 chop collapses a mid-release block toward zero and fails the
// ratio bound; assert down to a floor where the fade itself bottoms out.
const std::size_t r0 = frames - w;
CHECK(blockPeak(out, r0, 128) > 0.5);
double prev = blockPeak(out, r0, 128);
for (std::size_t b = r0 + 128; b + 128 <= frames; b += 128) {
const double cur = blockPeak(out, b, 128);
if (prev >= 0.15) CHECK(cur >= 0.3 * prev);
prev = cur;
}
}
// --- Trigger one-shot to its play end (lengthFraction < 1 exercises the playEnd_ feed bound):
// the final window BEFORE the stop point is gap-free at an off-root pitch. ---
static void testPreserveTriggerTailGapFree() {
const std::size_t frames = 8192;
const std::size_t w = 1024;
const double f0 = 1.0 / 163.84;
SampleData s = tailSine(frames, f0, 60);
s.play.pitchEngine = PitchEngine::Preserve;
s.play.playMode = PlayMode::Trigger;
s.play.trigger.lengthFraction = 0.8; // playEnd = 6554 (~40 exact cycles: ends near zero)
SampleData km = (std::move(s));
VoiceEngine eng(1, km, /*preserveCap=*/0, static_cast<std::int64_t>(w));
eng.noteOn(67, 127);
const std::size_t playEnd = 6554; // round(0.8 * 8192)
std::vector<AudioSample> out;
eng.render(out, playEnd);
CHECK(worstQuietRun(out, playEnd - w, playEnd, 0.05) < 24);
for (std::size_t b = playEnd - w; b + 128 <= playEnd; b += 128) {
CHECK(blockPeak(out, b, 128) > 0.5);
}
}
// --- Q-W0 T1-03: the Preserve prime is bounded by the PLAYABLE span. A Trigger zone whose
// play length is shorter than the OLA window must never carry source PAST the user's
// chosen stop into the ring — pre-fix, the prime pulled a full window bounded only by
// frameCount, and an up-shifted tap PLAYED the cut content (transposed) before the voice
// freed. The sample poisons everything past playEnd with amplitude 8: if any of it
// reaches the output, the peak bound fails. ---
static void testPreservePrimeStopsAtTriggerPlayEnd() {
const std::size_t frames = 8000;
const std::size_t w = 2048; // OLA window >> playable span
const std::size_t playLen = 500; // playEnd = round(8000 * 0.0625) = 500
SampleData s;
s.frames.resize(frames);
const double f0 = 1.0 / 50.0; // 10 cycles inside the playable span
for (std::size_t i = 0; i < frames; ++i) {
s.frames[i] = i < playLen
? static_cast<float>(std::sin(2.0 * kPi * f0 * static_cast<double>(i)))
: 8.0f; // POISON: cut content past the play end
}
s.rootNote = 60;
s.play.playMode = PlayMode::Trigger;
s.play.pitchEngine = PitchEngine::Preserve;
s.play.trigger.lengthFraction = 0.0625; // exactly 500 / 8000
SampleData km = (std::move(s));
VoiceEngine eng(1, km, /*preserveCap=*/0, static_cast<std::int64_t>(w));
eng.noteOn(72, 127); // +1 octave: the tap outruns the read head into
// the deepest primed history the ring holds
std::vector<AudioSample> out;
eng.render(out, playLen + 64); // through the voice's own end (readPos >= playEnd)
double peak = 0.0;
for (const AudioSample v : out) {
const double a = std::fabs(static_cast<double>(v));
if (a > peak) peak = a;
}
CHECK(peak < 1.5); // the 8.0 poison never sounds: nothing past playEnd entered the ring
CHECK(peak > 0.4); // ...and the real span genuinely played (the bound is not vacuous)
}
// --- Q-W0 T1-03 (companion): a whole sample SHORTER than the window (Gate, no loop) must not
// get zero padding declared as valid ring history — pre-fix, the prime zero-filled the
// window remainder with filled_ = window, so splices/tap travel landed in silence:
// hundreds-of-frames dead runs inside a sub-window one-shot (the pre-GA2 burst/gap
// artifact re-entering for short material). Post-fix the prime stops at the sample end
// and freezes the tail immediately, so the ring recycles ONLY real content. ---
static void testPreserveSubWindowSampleNoZeroPadInRing() {
const std::size_t frames = 1200; // sample < one window
const std::size_t w = 2048;
const double f0 = 1.0 / 96.0; // period 96: zero crossings dwell ~2 frames
SampleData s = tailSine(frames, f0, 60);
s.play.pitchEngine = PitchEngine::Preserve;
s.play.adsr = flatAdsr();
SampleData km = (std::move(s));
VoiceEngine eng(1, km, /*preserveCap=*/0, static_cast<std::int64_t>(w));
eng.noteOn(72, 127); // +1 octave up-shift (tap sweeps the whole ring)
std::vector<AudioSample> out;
eng.render(out, frames); // voice runs to its natural end (no loop)
// Pre-fix: the tap crossed the declared-valid zero pad repeatedly — quiet runs of 150+
// frames. Post-fix every relocation stays inside the real filled span; only sine zero
// crossings dip below the threshold.
CHECK(worstQuietRun(out, 0, frames, 0.05) < 30);
CHECK(blockPeak(out, 0, frames) > 0.5); // and it genuinely played at full level
}
// ---------------------------------------------------------------------------
// The Preserve read path's stretch generalization: the source is consumed at the playback
// rate while the shifter's read tap runs at the transposition, over ONE delay ring. Only
// their DIFFERENCE reaches the splice machinery.
// ---------------------------------------------------------------------------
// FNV-1a over the raw float bits — an exact-stream witness, not a tolerance.
static std::uint64_t hashStream(const std::vector<AudioSample>& v) {
std::uint64_t h = 1469598103934665603ull;
for (const AudioSample s : v) {
std::uint32_t bits = 0;
std::memcpy(&bits, &s, sizeof(bits));
for (int b = 0; b < 4; ++b) {
h ^= static_cast<std::uint64_t>((bits >> (8 * b)) & 0xffu);
h *= 1099511628211ull;
}
}
return h;
}
// A source with no symmetry a shifter could accidentally satisfy: a sine at a non-integer
// period plus a deterministic pseudo-random dither, so any change in the splice schedule,
// the fed frame sequence or the tap position moves the hash.
static SampleData stretchProbeSample(std::size_t frames, bool stereo) {
SampleData s;
s.frames.resize(frames);
if (stereo) s.framesR.resize(frames);
std::uint32_t lcg = 12345u;
for (std::size_t i = 0; i < frames; ++i) {
lcg = lcg * 1664525u + 1013904223u;
const double n = static_cast<double>(lcg >> 8) / 8388608.0 - 1.0; // [-1,1)
const double t = static_cast<double>(i);
s.frames[i] = static_cast<float>(0.8 * std::sin(2.0 * kPi * t / 196.37) + 0.1 * n);
if (stereo) {
s.framesR[i] =
static_cast<float>(0.8 * std::sin(2.0 * kPi * t / 123.13) - 0.1 * n);
}
}
s.rootNote = 60;
s.sampleRate = 44100;
s.play.adsr = flatAdsr();
s.play.pitchEngine = PitchEngine::Preserve;
return s;
}
// Renders one raw Voice (not through VoiceEngine, which publishes no rate) for `outFrames`.
static void renderVoice(const SampleData& s, int note, double rate, std::int64_t window,
bool stereo, std::vector<AudioSample>& l, std::vector<AudioSample>& r) {
Voice v;
v.presizePreserveShifters(window);
v.start(note, 127, s, /*declickTakeover=*/false, rate);
for (std::size_t i = 0; i < l.size(); ++i) {
if (stereo) {
AudioSample a = 0.0f, b = 0.0f;
v.renderFrameStereo(a, b);
l[i] = a;
r[i] = b;
} else {
l[i] = v.renderFrame();
}
}
}
// --- The null case, asserted against a baseline the SHIPPED engine produced. ---
// The four constants below were captured by running this same function against the
// pre-stretch build (phase-g, before the rate seam existed) and printing the hashes; they are
// therefore a witness that the generalized read path reproduces the shipped Preserve output
// bit for bit at rate 1.0, not a self-consistency check. A change here is a change to what
// every already-saved project sounds like — re-derive the cause before re-baselining.
static void testPreserveUnityRateIsBitIdenticalToTheShippedRead() {
const std::int64_t w = 2205; // the product window at 44.1k
const std::size_t n = 6000;
struct Case {
int note;
bool stereo;
bool loop;
std::uint64_t hashL;
std::uint64_t hashR;
};
const Case cases[] = {
{60, false, false, 16118581538698271917ull, 0ull}, // on root: unity shift
{67, false, false, 17268489061432447375ull, 0ull}, // +7 st: real splices
{55, false, false, 17626155132441637249ull, 0ull}, // -5 st: down-shift
{67, true, true, 116487689553455907ull, 9528575457480122654ull}, // stereo linked + loop
};
for (const Case& c : cases) {
SampleData s = stretchProbeSample(4000, c.stereo);
if (c.loop) {
s.loop.hasLoop = true;
s.loop.start = 1200;
s.loop.end = 3600;
s.loopCrossfadeFrames = 256;
}
std::vector<AudioSample> l(n), r(c.stereo ? n : 0);
renderVoice(s, c.note, /*rate=*/1.0, w, c.stereo, l, r);
const std::uint64_t hl = hashStream(l);
CHECK(hl == c.hashL);
if (hl != c.hashL) std::printf(" note %d L hash %lluull\n", c.note, hl);
if (c.stereo) {
const std::uint64_t hr = hashStream(r);
CHECK(hr == c.hashR);
if (hr != c.hashR) std::printf(" note %d R hash %lluull\n", c.note, hr);
}
}
}
// --- Rate changes DURATION only; the transposition alone sets pitch. ---
static void testPreserveStretchChangesDurationNotPitch() {
// Gate, no loop: the voice's life is exactly how long the source lasts, so the frame at
// which it goes idle IS the note's duration.
const std::int64_t w = 1024;
const std::size_t frames = 24000;
const double srcPeriod = 160.0;
SampleData s;
s.frames.resize(frames);
for (std::size_t i = 0; i < frames; ++i) {
s.frames[i] = static_cast<float>(std::sin(2.0 * kPi * static_cast<double>(i) / srcPeriod));
}
s.rootNote = 60;
s.play.adsr = flatAdsr();
s.play.pitchEngine = PitchEngine::Preserve;
auto run = [&](double rate, PitchEngine engine, std::size_t& lifeFrames) {
SampleData local = s;
local.play.pitchEngine = engine;
Voice v;
v.presizePreserveShifters(w);
v.start(60, 127, local, /*declickTakeover=*/false, rate);
std::vector<AudioSample> out;
out.reserve(frames * 3);
lifeFrames = 0;
for (std::size_t i = 0; i < frames * 3 && v.active(); ++i) {
out.push_back(v.renderFrame());
++lifeFrames;
}
return out;
};
std::size_t lifeUnity = 0, lifeSlow = 0, lifeFast = 0;
const std::vector<AudioSample> unity = run(1.0, PitchEngine::Preserve, lifeUnity);
const std::vector<AudioSample> slow = run(0.5, PitchEngine::Preserve, lifeSlow);
const std::vector<AudioSample> fast = run(2.0, PitchEngine::Preserve, lifeFast);
// Duration scales by 1/rate (the small excess over the source length is the terminal
// declick ring-out Preserve ends on).
CHECK(approx(static_cast<double>(lifeUnity), 24000.0, 200.0));
CHECK(approx(static_cast<double>(lifeSlow), 48000.0, 400.0));
CHECK(approx(static_cast<double>(lifeFast), 12000.0, 200.0));
// ...and the pitch does not move with it. Measured away from the onset and the tail.
auto period = [](const std::vector<AudioSample>& v, std::size_t from, std::size_t to) {
double sum = 0.0;
std::size_t prev = 0, count = 0;
for (std::size_t i = from + 1; i < to && i < v.size(); ++i) {
if (v[i - 1] <= 0.0f && v[i] > 0.0f) {
if (count > 0) sum += static_cast<double>(i - prev);
prev = i;
++count;
}
}
return count > 1 ? sum / static_cast<double>(count - 1) : 0.0;
};
CHECK(approx(period(unity, 2000, 9000), srcPeriod, 8.0));
CHECK(approx(period(slow, 2000, 9000), srcPeriod, 8.0));
CHECK(approx(period(fast, 2000, 9000), srcPeriod, 8.0));
// The non-tautology witness: VARISPEED is the engine that couples them. Reaching the same
// durations there costs exactly the pitch change Preserve refuses to make — so the three
// equal periods above are a property of the stretcher, not of the measurement.
std::size_t lifeVari = 0;
const std::vector<AudioSample> vari = run(0.5, PitchEngine::Varispeed, lifeVari);
CHECK(approx(static_cast<double>(lifeVari), 24000.0, 200.0)); // rate ignored under Varispeed
SampleData down = s;
down.play.pitchEngine = PitchEngine::Varispeed;
Voice vv;
vv.presizePreserveShifters(w);
vv.start(48, 127, down); // -12 st under Varispeed: duration doubles AND pitch halves
std::vector<AudioSample> variDown;
std::size_t variLife = 0;
for (std::size_t i = 0; i < frames * 3 && vv.active(); ++i) {
variDown.push_back(vv.renderFrame());
++variLife;
}
CHECK(approx(static_cast<double>(variLife), 48000.0, 200.0)); // same duration...
CHECK(approx(period(variDown, 2000, 9000), srcPeriod * 2.0, 16.0)); // ...at half pitch
}
// --- The onset is a regression surface: no added latency at ANY rate. ---
static void testPreserveStretchSpeaksOnFrameZeroAtEveryRate() {
const std::int64_t w = 2048;
SampleData s = stretchProbeSample(12000, false);
s.startFrame = 500; // and the first output frame is the START frame, not frame 0
for (double rate : {0.5, 1.0, 2.0}) {
for (int note : {48, 60, 67}) {
Voice v;
v.presizePreserveShifters(w);
v.start(note, 127, s, /*declickTakeover=*/false, rate);
const AudioSample first = v.renderFrame();
// The primed ring parks the tap ON the start frame, so output frame 0 is source
// frame `startFrame` exactly — at every rate and every transposition. A stretcher
// that buffered a window before speaking would fail here, which is the whole point.
CHECK(first == s.frames[500]);
// ...and it keeps speaking: no first-window dip while the schedule settles. The
// 256-frame measuring window spans most of a period even at the lowest note tested
// (-12 st stretches the probe's 196-frame period to 393), so a continuous tone
// peaks well above the floor in every one of them and only a real gap can sink it.
double lo = 1e9;
for (int i = 0; i < 20; ++i) {
double peak = 0.0;
for (int k = 0; k < 256; ++k) {
peak = (std::max)(peak, std::fabs(static_cast<double>(v.renderFrame())));
}
lo = (std::min)(lo, peak);
}
if (!(lo > 0.5)) std::printf(" rate %.2f note %d: lo %.3f\n", rate, note, lo);
CHECK(lo > 0.5);
}
}
}
// --- "Loop the source, shift the output" is unweakened by a stretch. ---
static void testPreserveStretchLoopsTheSourceSpan() {
for (double rate : {0.5, 1.0, 2.0}) {
for (int note : {48, 60, 72}) {
SampleData s;
s.frames.resize(200, 0.0f);
for (int i = 60; i < 120; ++i) s.frames[i] = 0.5f;
s.rootNote = 60;
s.sampleRate = 48000;
s.loop.hasLoop = true;
s.loop.start = 80;
s.loop.end = 120;
s.play.adsr = flatAdsr();
s.play.pitchEngine = PitchEngine::Preserve;
Voice v;
v.presizePreserveShifters(64);
v.start(note, 127, s, /*declickTakeover=*/false, rate);
std::vector<AudioSample> out(4000);
for (std::size_t i = 0; i < out.size(); ++i) out[i] = v.renderFrame();
// The loop is a SOURCE-frame fact, so it keeps the voice alive and at level for as
// long as it is held, whatever the rate consumes it at.
CHECK(v.active());
double sum = 0.0;
for (std::size_t i = out.size() - 200; i < out.size(); ++i) sum += out[i];
CHECK(approx(sum / 200.0, 0.5, 0.05));
}
}
}
// --- The 32-voice measurement gate. Asserts correctness; PRINTS the cost, which is the
// number reported for the algorithm decision (meaningful only in a Release build). ---
static void testPreserveStretchThirtyTwoVoicesHoldUp() {
const std::int64_t w = 2205; // the product window at 44.1k
const std::size_t blockFrames = 44100; // one second of audio
const std::size_t voiceCount = 32;
SampleData s = stretchProbeSample(200000, true);
s.loop.hasLoop = true; // held notes: all 32 sound for the whole run
s.loop.start = 40000;
s.loop.end = 160000;
s.loopCrossfadeFrames = 1024;
// 1.0 is the reference: it is the cost the shipped Preserve read already carries, so the
// two stretched rows are read as a delta against it rather than in isolation.
for (double rate : {1.0, 0.5, 2.0}) {
std::vector<Voice> voices(voiceCount);
for (std::size_t i = 0; i < voiceCount; ++i) {
voices[i].presizePreserveShifters(w);
voices[i].start(48 + static_cast<int>(i), 100, s, /*declickTakeover=*/false, rate);
}
const std::clock_t t0 = std::clock();
double guard = 0.0;
std::size_t sounding = 0;
for (std::size_t f = 0; f < blockFrames; ++f) {
AudioSample l = 0.0f, r = 0.0f;
for (std::size_t i = 0; i < voiceCount; ++i) {
AudioSample a = 0.0f, b = 0.0f;
voices[i].renderFrameStereo(a, b);
l += a;
r += b;
}
guard += static_cast<double>(l) + static_cast<double>(r);
CHECK(std::isfinite(l) && std::isfinite(r));
}
const double secs = static_cast<double>(std::clock() - t0) / CLOCKS_PER_SEC;
for (std::size_t i = 0; i < voiceCount; ++i) {
if (voices[i].active()) ++sounding;
}
CHECK(sounding == voiceCount); // all 32 held the whole second (the loop kept them up)
CHECK(std::fabs(guard) > 0.0); // ...and genuinely produced audio
std::printf(" [measure] 32 stereo Preserve voices @ rate %.2f: %.3f s wall for 1.0 s "
"audio (%.1f%% of one core, %.1f ns/voice/frame)\n",
rate, secs, 100.0 * secs,
secs * 1e9 / (static_cast<double>(blockFrames) *
static_cast<double>(voiceCount)));
}
}
int main() {
testEveryKeyPlaysTheLoadedCapture();
testUnplayableCaptureRefusesEveryNote();
testOutOfRangeNotesAreRefusedInMono();
testPitchRatioMath();
testKeyTrackedRatioMath();
testRepitchObservedPeriod();
testKeyTrackVarispeedObservedPeriod();
testKeyTrackPreserveShiftCollapsesAtZero();
testVelocityPitchIsExactlyOffByDefault();
testDrawnVelocityPitchCurveTransposesBothWays();
testAdsrShape();
testAdsrReleaseBeforeSustain();
testAdsrZeroAttackDecay();
testPolyphonicAllocation();
testNoteOffReleasesNewestSameNote();
testStealsReleasingVoiceFirst();
testStealsOldestWhenNoneReleasing();
testLoopSustainSeamless();
testZeroLengthLoopGoesSilent();
testSingleFrameLoop();
testAbsentLoopGoesSilent();
testStartFrameOffsetsInitialRead();
testStartFrameZeroIsUnchanged();
testStartFrameOutOfRangeClampsToZero();
testStartFrameWithLoop();
testStartAfterLoopEndWrapsIntoLoop();
testLoopReadWrapsSampleExactOverManyCycles();
testZeroCrossfadeLeavesTheSeamHard();
testCrossfadeBlendsMonotonelyAcrossTheRegion();
testSeamStepMatchesTheNaturalStepForAnyCrossfade();
testCrossfadeLengthFollowsItsParameter();
testCrossfadeIsSuppressedForALoopAtFrameZero();
testNoteOffDuringLoopSustainRunsTheReleaseAndFreesTheVoice();
testPreserveLoopsTheSourceSpanAtEveryTransposition();
testVelocityDefaultCurveIsFlatUnity();
testVelocityLinearCurveReproducesRamp();
testVelocityShapedCurveDrivesGain();
testPolyphonyMixesAdditively();
testChannelCount();
testStereoRenderKeepsChannelsDistinct();
testMonoSamplePlaysDualMonoInStereo();
testDualMonoStereoSampleRendersCentered();
testMonoRenderUnchangedByStereoData();
testStereoRenderAdvancesLikeMonoRepitch();
testStereoRenderSumsVoicesPerChannel();
testStereoRenderNullBufferIsNoOp();
testStereoStartFrameLoopShareOneReadHead();
// S15 — sampling modes.
testAhdsrHoldStageShape();
testAhdsrHoldZeroEqualsAdsr();
testTriggerLengthFractionFrames();
testTriggerLengthWithStart();
testTriggerAhdFadeShape();
testTriggerEdgeCases();
testTriggerFracAboveOneClampsToSpan();
testTriggerFracNaNFreesImmediately();
testTriggerIgnoresNoteOff();
// S16 — pitch engine + pitch envelope.
testPreserveDurationInvariance();
testVarispeedStillCouplesDuration();
testPitchEnvOffBitIdentical();
testPitchEnvOnBendsVarispeed();
testPreserveGateStereoLoopComposes();
testPreserveVoiceCap();
// FA1 postscript — the unity demotion is gone with the PreviewCard; the engine keeps a
// uniform Preserve onset at every note. Velocity under Preserve unchanged.
testPreserveUnityEngineVoiceSpeaksImmediately();
testPreserveTransposedVoiceSpeaksImmediately();
testPreserveUnityVoiceCountsTowardCap();
testVelocityCurveAppliesUnderPreserve();
// S12 review fix — per-zone A/D/S/R reaches the voice envelope.
testPerZoneAdsrReachesVoiceEnvelope();
testZeroAdsrIsInstantSustain();
// Phase S — voice count, MONO mode (held stack + Retrigger/Legato); preview as a real
// pool voice (PreviewCard retired — preview routes through the engine).
testMonoLastNotePriorityAndFallback();
testMonoReleaseOfLowerHeldNoteIsInaudible();
testMonoRepressHeldNoteMovesToTop();
testMonoRetriggerFallbackUsesOriginalVelocity();
testMonoOutOfRangeNeverJoinsStack();
testMonoRetriggerRestartsEnvelope();
testMonoLegatoContinuesEnvelope();
testMonoLegatoRetunesWithoutReadRestart();
testMonoLegatoAlwaysGlidesWithinThePhrase();
testMonoLegatoAfterReleaseReattacks();
testMonoIgnoresPreserveCap();
testMonoLegatoTriggerReattacksAfterKeyUp();
testMonoLegatoTriggerHeldKeyStillRetunes();
testAllNotesOffReleasesPolyVoices();
testAllNotesOffClearsMonoHeldStack();
testAllSoundsOffStopsTriggerOneShot();
testAllNotesOffStillReleasesGateVoices();
testMonoLegatoSameNoteRepressReattacks();
testMonoOutOfRangeNotesRejected();
testVoiceCountBoundsPolyphony();
testOverCapChordStealsExactlyOne();
testMonoRetrigTakeoverDeclicksRestart();
testMonoRetrigFallbackDeclicksRestart();
testMonoDeclickOnlyOnTakeover();
testPolyStealDeclicksRestart();
testSameBlockDoubleTakeoverKeepsDeclickSeed();
testZeroAttackTakeoverNeverExceedsFullScale();
testMonoRetrigTriggerZoneDeclicksRestart();
testZeroAttackGateRetrigNoStep();
testPreviewReauditionDeclicksViaEngineSteal();
testDeclickBoundedBlendNoOvershoot();
testPreviewNoteObeysVoicing();
testPreviewNoteJoinsMonoHeldStack();
testPreviewNoteOffRoutesToDrainEngine();
// GA3 — Preserve tail wind-down (writer freeze at source exhaustion).
testPreserveTailFinalWindowGapFree();
testPreserveTailReleaseContinuous();
testPreserveTriggerTailGapFree();
// Q-W0 T1-03 — the prime is bounded by the playable span (Trigger playEnd / sample end),
// with an immediate tail freeze on sub-window spans.
testPreservePrimeStopsAtTriggerPlayEnd();
testPreserveSubWindowSampleNoZeroPadInRing();
// The Preserve read path's stretch generalization.
testPreserveUnityRateIsBitIdenticalToTheShippedRead();
testPreserveStretchChangesDurationNotPitch();
testPreserveStretchSpeaksOnFrameZeroAtEveryRate();
testPreserveStretchLoopsTheSourceSpan();
testPreserveStretchThirtyTwoVoicesHoldUp();
if (g_fail == 0) {
std::printf("all sampler_core tests passed\n");
return 0;
}
std::printf("%d sampler_core check(s) failed\n", g_fail);
return 1;
}