Γ-W1-T5: a real Preserve time-stretcher — write rate is duration, tap rate is pitch

Generalizes the correlation-aligned SOLA delay line so the feed and the shift are
independent rates over one ring. Unity is bit-identical to the shipped read, asserted
against a hash baseline captured pre-change.
This commit is contained in:
2026-08-01 19:06:06 -04:00
parent e589addc54
commit 589a8e078b
10 changed files with 873 additions and 71 deletions
+307
View File
@@ -21,7 +21,10 @@
#include <algorithm>
#include <cmath>
#include <cstdint>
#include <cstdio>
#include <cstring>
#include <ctime>
#include <vector>
using namespace reasampler;
@@ -2880,6 +2883,303 @@ static void testPreserveSubWindowSampleNoZeroPadInRing() {
CHECK(blockPeak(out, 0, frames) > 0.5); // and it genuinely played at full level
}
// ---------------------------------------------------------------------------
// The Preserve read path's stretch generalization: the source is consumed at the playback
// rate while the shifter's read tap runs at the transposition, over ONE delay ring. Only
// their DIFFERENCE reaches the splice machinery.
// ---------------------------------------------------------------------------
// FNV-1a over the raw float bits — an exact-stream witness, not a tolerance.
static std::uint64_t hashStream(const std::vector<AudioSample>& v) {
std::uint64_t h = 1469598103934665603ull;
for (const AudioSample s : v) {
std::uint32_t bits = 0;
std::memcpy(&bits, &s, sizeof(bits));
for (int b = 0; b < 4; ++b) {
h ^= static_cast<std::uint64_t>((bits >> (8 * b)) & 0xffu);
h *= 1099511628211ull;
}
}
return h;
}
// A source with no symmetry a shifter could accidentally satisfy: a sine at a non-integer
// period plus a deterministic pseudo-random dither, so any change in the splice schedule,
// the fed frame sequence or the tap position moves the hash.
static SampleData stretchProbeSample(std::size_t frames, bool stereo) {
SampleData s;
s.frames.resize(frames);
if (stereo) s.framesR.resize(frames);
std::uint32_t lcg = 12345u;
for (std::size_t i = 0; i < frames; ++i) {
lcg = lcg * 1664525u + 1013904223u;
const double n = static_cast<double>(lcg >> 8) / 8388608.0 - 1.0; // [-1,1)
const double t = static_cast<double>(i);
s.frames[i] = static_cast<float>(0.8 * std::sin(2.0 * kPi * t / 196.37) + 0.1 * n);
if (stereo) {
s.framesR[i] =
static_cast<float>(0.8 * std::sin(2.0 * kPi * t / 123.13) - 0.1 * n);
}
}
s.rootNote = 60;
s.sampleRate = 44100;
s.play.adsr = flatAdsr();
s.play.pitchEngine = PitchEngine::Preserve;
return s;
}
// Renders one raw Voice (not through VoiceEngine, which publishes no rate) for `outFrames`.
static void renderVoice(const SampleData& s, int note, double rate, std::int64_t window,
bool stereo, std::vector<AudioSample>& l, std::vector<AudioSample>& r) {
Voice v;
v.presizePreserveShifters(window);
v.start(note, 127, s, /*declickTakeover=*/false, rate);
for (std::size_t i = 0; i < l.size(); ++i) {
if (stereo) {
AudioSample a = 0.0f, b = 0.0f;
v.renderFrameStereo(a, b);
l[i] = a;
r[i] = b;
} else {
l[i] = v.renderFrame();
}
}
}
// --- The null case, asserted against a baseline the SHIPPED engine produced. ---
// The four constants below were captured by running this same function against the
// pre-stretch build (phase-g, before the rate seam existed) and printing the hashes; they are
// therefore a witness that the generalized read path reproduces the shipped Preserve output
// bit for bit at rate 1.0, not a self-consistency check. A change here is a change to what
// every already-saved project sounds like — re-derive the cause before re-baselining.
static void testPreserveUnityRateIsBitIdenticalToTheShippedRead() {
const std::int64_t w = 2205; // the product window at 44.1k
const std::size_t n = 6000;
struct Case {
int note;
bool stereo;
bool loop;
std::uint64_t hashL;
std::uint64_t hashR;
};
const Case cases[] = {
{60, false, false, 16118581538698271917ull, 0ull}, // on root: unity shift
{67, false, false, 17268489061432447375ull, 0ull}, // +7 st: real splices
{55, false, false, 17626155132441637249ull, 0ull}, // -5 st: down-shift
{67, true, true, 116487689553455907ull, 9528575457480122654ull}, // stereo linked + loop
};
for (const Case& c : cases) {
SampleData s = stretchProbeSample(4000, c.stereo);
if (c.loop) {
s.loop.hasLoop = true;
s.loop.start = 1200;
s.loop.end = 3600;
s.loopCrossfadeFrames = 256;
}
std::vector<AudioSample> l(n), r(c.stereo ? n : 0);
renderVoice(s, c.note, /*rate=*/1.0, w, c.stereo, l, r);
const std::uint64_t hl = hashStream(l);
CHECK(hl == c.hashL);
if (hl != c.hashL) std::printf(" note %d L hash %lluull\n", c.note, hl);
if (c.stereo) {
const std::uint64_t hr = hashStream(r);
CHECK(hr == c.hashR);
if (hr != c.hashR) std::printf(" note %d R hash %lluull\n", c.note, hr);
}
}
}
// --- Rate changes DURATION only; the transposition alone sets pitch. ---
static void testPreserveStretchChangesDurationNotPitch() {
// Gate, no loop: the voice's life is exactly how long the source lasts, so the frame at
// which it goes idle IS the note's duration.
const std::int64_t w = 1024;
const std::size_t frames = 24000;
const double srcPeriod = 160.0;
SampleData s;
s.frames.resize(frames);
for (std::size_t i = 0; i < frames; ++i) {
s.frames[i] = static_cast<float>(std::sin(2.0 * kPi * static_cast<double>(i) / srcPeriod));
}
s.rootNote = 60;
s.play.adsr = flatAdsr();
s.play.pitchEngine = PitchEngine::Preserve;
auto run = [&](double rate, PitchEngine engine, std::size_t& lifeFrames) {
SampleData local = s;
local.play.pitchEngine = engine;
Voice v;
v.presizePreserveShifters(w);
v.start(60, 127, local, /*declickTakeover=*/false, rate);
std::vector<AudioSample> out;
out.reserve(frames * 3);
lifeFrames = 0;
for (std::size_t i = 0; i < frames * 3 && v.active(); ++i) {
out.push_back(v.renderFrame());
++lifeFrames;
}
return out;
};
std::size_t lifeUnity = 0, lifeSlow = 0, lifeFast = 0;
const std::vector<AudioSample> unity = run(1.0, PitchEngine::Preserve, lifeUnity);
const std::vector<AudioSample> slow = run(0.5, PitchEngine::Preserve, lifeSlow);
const std::vector<AudioSample> fast = run(2.0, PitchEngine::Preserve, lifeFast);
// Duration scales by 1/rate (the small excess over the source length is the terminal
// declick ring-out Preserve ends on).
CHECK(approx(static_cast<double>(lifeUnity), 24000.0, 200.0));
CHECK(approx(static_cast<double>(lifeSlow), 48000.0, 400.0));
CHECK(approx(static_cast<double>(lifeFast), 12000.0, 200.0));
// ...and the pitch does not move with it. Measured away from the onset and the tail.
auto period = [](const std::vector<AudioSample>& v, std::size_t from, std::size_t to) {
double sum = 0.0;
std::size_t prev = 0, count = 0;
for (std::size_t i = from + 1; i < to && i < v.size(); ++i) {
if (v[i - 1] <= 0.0f && v[i] > 0.0f) {
if (count > 0) sum += static_cast<double>(i - prev);
prev = i;
++count;
}
}
return count > 1 ? sum / static_cast<double>(count - 1) : 0.0;
};
CHECK(approx(period(unity, 2000, 9000), srcPeriod, 8.0));
CHECK(approx(period(slow, 2000, 9000), srcPeriod, 8.0));
CHECK(approx(period(fast, 2000, 9000), srcPeriod, 8.0));
// The non-tautology witness: VARISPEED is the engine that couples them. Reaching the same
// durations there costs exactly the pitch change Preserve refuses to make — so the three
// equal periods above are a property of the stretcher, not of the measurement.
std::size_t lifeVari = 0;
const std::vector<AudioSample> vari = run(0.5, PitchEngine::Varispeed, lifeVari);
CHECK(approx(static_cast<double>(lifeVari), 24000.0, 200.0)); // rate ignored under Varispeed
SampleData down = s;
down.play.pitchEngine = PitchEngine::Varispeed;
Voice vv;
vv.presizePreserveShifters(w);
vv.start(48, 127, down); // -12 st under Varispeed: duration doubles AND pitch halves
std::vector<AudioSample> variDown;
std::size_t variLife = 0;
for (std::size_t i = 0; i < frames * 3 && vv.active(); ++i) {
variDown.push_back(vv.renderFrame());
++variLife;
}
CHECK(approx(static_cast<double>(variLife), 48000.0, 200.0)); // same duration...
CHECK(approx(period(variDown, 2000, 9000), srcPeriod * 2.0, 16.0)); // ...at half pitch
}
// --- The onset is a regression surface: no added latency at ANY rate. ---
static void testPreserveStretchSpeaksOnFrameZeroAtEveryRate() {
const std::int64_t w = 2048;
SampleData s = stretchProbeSample(12000, false);
s.startFrame = 500; // and the first output frame is the START frame, not frame 0
for (double rate : {0.5, 1.0, 2.0}) {
for (int note : {48, 60, 67}) {
Voice v;
v.presizePreserveShifters(w);
v.start(note, 127, s, /*declickTakeover=*/false, rate);
const AudioSample first = v.renderFrame();
// The primed ring parks the tap ON the start frame, so output frame 0 is source
// frame `startFrame` exactly — at every rate and every transposition. A stretcher
// that buffered a window before speaking would fail here, which is the whole point.
CHECK(first == s.frames[500]);
// ...and it keeps speaking: no first-window dip while the schedule settles. The
// 256-frame measuring window spans most of a period even at the lowest note tested
// (-12 st stretches the probe's 196-frame period to 393), so a continuous tone
// peaks well above the floor in every one of them and only a real gap can sink it.
double lo = 1e9;
for (int i = 0; i < 20; ++i) {
double peak = 0.0;
for (int k = 0; k < 256; ++k) {
peak = (std::max)(peak, std::fabs(static_cast<double>(v.renderFrame())));
}
lo = (std::min)(lo, peak);
}
if (!(lo > 0.5)) std::printf(" rate %.2f note %d: lo %.3f\n", rate, note, lo);
CHECK(lo > 0.5);
}
}
}
// --- "Loop the source, shift the output" is unweakened by a stretch. ---
static void testPreserveStretchLoopsTheSourceSpan() {
for (double rate : {0.5, 1.0, 2.0}) {
for (int note : {48, 60, 72}) {
SampleData s;
s.frames.resize(200, 0.0f);
for (int i = 60; i < 120; ++i) s.frames[i] = 0.5f;
s.rootNote = 60;
s.sampleRate = 48000;
s.loop.hasLoop = true;
s.loop.start = 80;
s.loop.end = 120;
s.play.adsr = flatAdsr();
s.play.pitchEngine = PitchEngine::Preserve;
Voice v;
v.presizePreserveShifters(64);
v.start(note, 127, s, /*declickTakeover=*/false, rate);
std::vector<AudioSample> out(4000);
for (std::size_t i = 0; i < out.size(); ++i) out[i] = v.renderFrame();
// The loop is a SOURCE-frame fact, so it keeps the voice alive and at level for as
// long as it is held, whatever the rate consumes it at.
CHECK(v.active());
double sum = 0.0;
for (std::size_t i = out.size() - 200; i < out.size(); ++i) sum += out[i];
CHECK(approx(sum / 200.0, 0.5, 0.05));
}
}
}
// --- The 32-voice measurement gate. Asserts correctness; PRINTS the cost, which is the
// number reported for the algorithm decision (meaningful only in a Release build). ---
static void testPreserveStretchThirtyTwoVoicesHoldUp() {
const std::int64_t w = 2205; // the product window at 44.1k
const std::size_t blockFrames = 44100; // one second of audio
const std::size_t voiceCount = 32;
SampleData s = stretchProbeSample(200000, true);
s.loop.hasLoop = true; // held notes: all 32 sound for the whole run
s.loop.start = 40000;
s.loop.end = 160000;
s.loopCrossfadeFrames = 1024;
// 1.0 is the reference: it is the cost the shipped Preserve read already carries, so the
// two stretched rows are read as a delta against it rather than in isolation.
for (double rate : {1.0, 0.5, 2.0}) {
std::vector<Voice> voices(voiceCount);
for (std::size_t i = 0; i < voiceCount; ++i) {
voices[i].presizePreserveShifters(w);
voices[i].start(48 + static_cast<int>(i), 100, s, /*declickTakeover=*/false, rate);
}
const std::clock_t t0 = std::clock();
double guard = 0.0;
std::size_t sounding = 0;
for (std::size_t f = 0; f < blockFrames; ++f) {
AudioSample l = 0.0f, r = 0.0f;
for (std::size_t i = 0; i < voiceCount; ++i) {
AudioSample a = 0.0f, b = 0.0f;
voices[i].renderFrameStereo(a, b);
l += a;
r += b;
}
guard += static_cast<double>(l) + static_cast<double>(r);
CHECK(std::isfinite(l) && std::isfinite(r));
}
const double secs = static_cast<double>(std::clock() - t0) / CLOCKS_PER_SEC;
for (std::size_t i = 0; i < voiceCount; ++i) {
if (voices[i].active()) ++sounding;
}
CHECK(sounding == voiceCount); // all 32 held the whole second (the loop kept them up)
CHECK(std::fabs(guard) > 0.0); // ...and genuinely produced audio
std::printf(" [measure] 32 stereo Preserve voices @ rate %.2f: %.3f s wall for 1.0 s "
"audio (%.1f%% of one core, %.1f ns/voice/frame)\n",
rate, secs, 100.0 * secs,
secs * 1e9 / (static_cast<double>(blockFrames) *
static_cast<double>(voiceCount)));
}
}
int main() {
testEveryKeyPlaysTheLoadedCapture();
testUnplayableCaptureRefusesEveryNote();
@@ -3006,6 +3306,13 @@ int main() {
testPreservePrimeStopsAtTriggerPlayEnd();
testPreserveSubWindowSampleNoZeroPadInRing();
// The Preserve read path's stretch generalization.
testPreserveUnityRateIsBitIdenticalToTheShippedRead();
testPreserveStretchChangesDurationNotPitch();
testPreserveStretchSpeaksOnFrameZeroAtEveryRate();
testPreserveStretchLoopsTheSourceSpan();
testPreserveStretchThirtyTwoVoicesHoldUp();
if (g_fail == 0) {
std::printf("all sampler_core tests passed\n");
return 0;