Q-W1 pt2: core/shell/app relocation + sub-namespaces; one concrete ui::Rect (LTRB fork retired); slot_map split from bank_book; BankIndex→BankModel; 59/59 green

This commit is contained in:
2026-07-28 20:48:56 -04:00
parent 67a41728f3
commit 847936f813
222 changed files with 2247 additions and 2079 deletions
@@ -0,0 +1,51 @@
// master_gain.cpp — see master_gain.h. Pure math; no LICE/VST3/REAPER includes.
#include "core/instrument/engine/master_gain.h"
#include "core/util/clamp01.h"
#include <algorithm>
#include <cmath>
#include <cstdio>
#include <limits>
namespace reasampler::instrument::engine {
using util::clamp01; // the ONE unit-interval clamp (Q-W1, T4-24)
double masterGainMaxLinear() { return std::pow(10.0, kMasterGainMaxDb / 20.0); }
double masterGainDbFromNorm(double norm) {
norm = clamp01(norm);
if (norm <= 0.0) return -std::numeric_limits<double>::infinity();
return kMasterGainMinDb + norm * (kMasterGainMaxDb - kMasterGainMinDb);
}
double masterGainNormFromDb(double db) {
if (!(db > kMasterGainMinDb)) return 0.0; // -inf, NaN, and the floor all read 0
return clamp01((db - kMasterGainMinDb) / (kMasterGainMaxDb - kMasterGainMinDb));
}
double masterGainLinearFromNorm(double norm) {
norm = clamp01(norm);
if (norm <= 0.0) return 0.0; // TRUE silence at the bottom — not an epsilon
return std::pow(10.0, masterGainDbFromNorm(norm) / 20.0);
}
double masterGainNormFromLinear(double linear) {
if (!std::isfinite(linear) || linear <= 0.0) return 0.0;
return masterGainNormFromDb(20.0 * std::log10(linear));
}
void formatMasterGainLabel(double norm, char* buf, std::size_t len) {
if (!buf || len == 0) return;
norm = clamp01(norm);
if (norm <= 0.0) {
std::snprintf(buf, len, "-inf");
return;
}
const double db = masterGainDbFromNorm(norm);
std::snprintf(buf, len, "%+.1fdB", db);
}
} // namespace reasampler::instrument::engine
+58
View File
@@ -0,0 +1,58 @@
// master_gain.h — PURE dB<->linear<->knob-taper math for the FB1 post-mixer master gain.
// NO VST3, NO REAPER, NO SWELL/LICE types. The mirror of trigger_seam: one tiny module owns
// the ONE formula both sides of a seam share — here the editor's Gain knob (normalized 0..1)
// and the processor's stored/applied linear gain — so the drawn needle, the persisted value,
// and the audio-thread multiply can never drift.
//
// THE CONTROL (Daniel, FB1). A post-mixer master gain, range -inf .. +24 dB, dB-scaled taper
// with -inf at the BOTTOM of the knob: normalized 0 maps to TRUE ZERO linear gain (silence,
// not a tiny epsilon), and the remaining travel maps linearly in dB from kMasterGainMinDb
// (the finite taper floor) up to kMasterGainMaxDb. Unity (0 dB) sits at norm
// kMasterGainMinDb/(kMasterGainMinDb - kMasterGainMaxDb) ~= 0.714 — most of the throw is
// usable trim, the last stretch is boost. The PERSISTED value is the LINEAR gain (a plain
// finite double, 0 = silence — no -inf on the wire); the taper is a UI-side view of it.
//
// RT DISCIPLINE: the processor applies the linear gain as one multiply over the summed
// output — these functions run on the UI/state threads only.
#pragma once
#include <cstddef>
namespace reasampler::instrument::engine {
// The dB taper endpoints. norm 0 is -inf (true zero); norm just above 0 starts at the
// finite floor kMasterGainMinDb and sweeps linearly in dB to kMasterGainMaxDb at norm 1.
inline constexpr double kMasterGainMinDb = -60.0;
inline constexpr double kMasterGainMaxDb = 24.0;
// The largest linear gain the control can produce (kMasterGainMaxDb as a ratio, ~15.849).
double masterGainMaxLinear();
// Knob taper: normalized [0,1] -> dB. norm <= 0 -> -infinity; else the linear-in-dB sweep
// [kMasterGainMinDb, kMasterGainMaxDb]. norm is clamped to [0,1]. Pure.
double masterGainDbFromNorm(double norm);
// Inverse taper: dB -> normalized [0,1]. -infinity (or any dB at or below kMasterGainMinDb,
// including below-floor values like -80 dB) maps to norm 0 (the -inf bottom detent) — the
// finite sweep only covers the range above kMasterGainMinDb; everything at or below it collapses
// to the same true-zero bottom. +24 -> 1. Pure.
double masterGainNormFromDb(double db);
// Knob taper composed with dB->ratio: normalized [0,1] -> LINEAR gain. norm 0 -> exactly
// 0.0 (true silence); norm 1 -> masterGainMaxLinear(). Pure.
double masterGainLinearFromNorm(double norm);
// Inverse: LINEAR gain -> normalized [0,1]. linear <= 0 -> 0 (the -inf bottom); a linear at
// or below the kMasterGainMinDb floor (e.g. 0.001 = -60 dB, or anything below) also maps to 0
// — the floor IS the -inf detent; values between true-zero and the floor cannot be represented
// on the knob and collapse to the bottom. unity -> ~0.714; masterGainMaxLinear() -> 1.
// Out-of-range/non-finite input clamps. Pure.
double masterGainNormFromLinear(double linear);
// The knob's hover/drag value label for a normalized value: "-inf" at the bottom, else a
// signed one-decimal dB string ("-12.0dB", "+0.0dB", "+2.4dB"). Writes at most `len` bytes
// including the terminator. Pure.
void formatMasterGainLabel(double norm, char* buf, std::size_t len);
} // namespace reasampler::instrument::engine
+434
View File
@@ -0,0 +1,434 @@
// pitch_shift — pure implementation. See pitch_shift.h for the contract, the S16-F2
// route-(b) rationale (WDL drags <windows.h>), and the GA-Preserve root cause that replaced
// the naive dual-tap OLA with correlation-aligned splices.
// NO VST3 / REAPER / SWELL / vendor includes; standard library only.
//
// Algorithm: a delay ring of 2*window frames. The write head advances one frame per input
// sample (source rate -> duration preserved). ONE active read tap advances by the shift
// `ratio_` per frame, so its delay behind the writer drifts at (1 - ratio) per frame. When
// that delay leaves the safe band [dLow, dHigh], the tap is RELOCATED by a nominal jump of
// one window (+window toward older content for up-shifts, -window toward the writer for
// down-shifts) — CLAMPED to the filled span so it can never land in unwritten silence (the
// GA2 onset fix) — refined by a cross-correlation search over +/- maxLag PLUS a parabolic
// peak interpolation for a SUB-SAMPLE lag, so the relocated read point is waveform-aligned
// to a fraction of a sample (integer-lag splices left +/-0.5-sample errors: a -59 dB
// sideband comb at the splice cadence on a repitched pure sine — the GA2 "alias lines" on
// the spectrogram). Old and new taps then crossfade over fadeFrames with a raised-cosine,
// amplitude-complementary pair (in-phase content sums to exactly unity gain). For a pure
// sine the correlation snaps the jump to an (integer + fraction) period count, so the output
// stays a single tone at the shifted frequency — the GA-Preserve acceptance bar. At unity
// ratio the delay is frozen mid-band and no splice ever fires: a primed shifter passes the
// stream through with ZERO added latency; a silence-warmed one is a clean window delay.
#include "core/instrument/engine/pitch_shift.h"
#include <algorithm>
#include <cmath>
#include <limits>
namespace reasampler::instrument::engine {
namespace {
constexpr double kPi = 3.14159265358979323846;
} // namespace
void PitchShifter::configure(std::int64_t windowFrames) {
window_ = windowFrames;
if (window_ <= 1) {
// Pass-through: no ring, process() returns input unchanged.
ring_.clear();
ringLen_ = 0;
writePos_ = 0;
posA_ = posB_ = 0.0;
fading_ = false;
fadePos_ = 0;
fadeFrames_ = fadeLen_ = maxLag_ = corrFrames_ = dLow_ = dHigh_ = 0;
filled_ = 0;
ratio_ = 1.0;
tailFrozen_ = false;
lastSplice_ = SpliceEvent{};
return;
}
// 2x-window ring: one window of splice-jump span plus search + fade headroom on each side.
ringLen_ = 2 * window_;
ring_.assign(static_cast<std::size_t>(ringLen_), 0.0f);
// Geometry (all quarters of the window):
// - fadeFrames_: the NOMINAL splice crossfade. This window/4 length is only safe when
// the outgoing tap cannot reach the writer before the fade ends; splice() scales the
// live fade length (fadeLen_) down by the current ratio for up-shifts past ~2x, so
// ordinary sampler transpositions (+24 st = ratio 4) never read stale data mid-fade.
// - maxLag_: the alignment search half-range — one window/4 covers a full period of any
// tone down to 4/window cycles-per-frame (~80 Hz at the product's 50 ms window, 44.1k).
// - dLow_/dHigh_: the safe delay band; unity parks the tap mid-band (window/2 delay).
// - corrFrames_: the correlation segment length. At an up-splice the reference segment
// reads FORWARD from the tap at delay ~dLow_, so dLow_-1 frames is exactly what exists
// between the tap and the writer — the cap expresses that safety rather than leaving
// it coincidental. 512 bounds the splice burst.
fadeFrames_ = std::max<std::int64_t>(window_ / 4, 1);
maxLag_ = window_ / 4;
dLow_ = window_ / 4;
dHigh_ = ringLen_ - window_ / 4;
corrFrames_ = std::max<std::int64_t>(1, std::min<std::int64_t>(dLow_ - 1, 512));
fadeLen_ = 0;
reset();
}
void PitchShifter::reset() {
if (window_ > 1) {
// Zero the ring and seed the active tap one window behind the writer — the exact
// middle of the safe band [dLow, dHigh] = [w/4, 2w - w/4], so unity holds it there
// forever and either shift direction has maximal drift room. No history is declared
// (filled_ = 0): follow with prime() or warm() before streaming.
std::fill(ring_.begin(), ring_.end(), 0.0f);
writePos_ = 0;
posA_ = static_cast<double>(ringLen_ - window_);
posB_ = posA_;
fading_ = false;
fadePos_ = 0;
fadeLen_ = 0;
} else {
writePos_ = 0;
posA_ = posB_ = 0.0;
fading_ = false;
fadePos_ = 0;
fadeLen_ = 0;
}
filled_ = 0;
ratio_ = 1.0;
tailFrozen_ = false;
lastSplice_ = SpliceEvent{};
}
void PitchShifter::freezeTail() {
if (window_ <= 1 || tailFrozen_) return;
tailFrozen_ = true;
// An in-flight crossfade was sized for a RETREATING writer (outgoing tap drains at
// ratio-1 per frame); frozen, the outgoing tap closes at the full ratio. Cap the live
// fade so it completes before tap B reaches the parked writer and reads lapped (oldest-
// window) content mid-fade. fadePos_ is re-anchored to the same fractional t so gNew is
// continuous at the freeze frame (no gain step); see the re-anchor block below.
if (fading_) {
// Preserve t = fadePos_/fadeLen_ across the shortening so gNew is continuous at the
// freeze frame (no gain step). Compute tOld BEFORE overwriting fadeLen_, then
// re-anchor fadePos_ to the same fractional position in the new (shorter) fade.
const double tOld =
static_cast<double>(fadePos_) / static_cast<double>(fadeLen_);
double dB = static_cast<double>(writePos_) - posB_;
const double len = static_cast<double>(ringLen_);
while (dB < 0.0) dB += len;
while (dB >= len) dB -= len;
// Clamp in double before the int64 cast (matches splice() pattern; guards against UB
// when dB/ratio_ is very large, e.g. near-unity ratio at a high sample rate).
double left = (dB - 2.0) / ratio_;
if (left > static_cast<double>(fadeFrames_)) left = static_cast<double>(fadeFrames_);
const std::int64_t leftFrames = left > 1.0 ? static_cast<std::int64_t>(left) : 1;
const std::int64_t newFadeLen = std::min(fadeLen_, fadePos_ + leftFrames);
// Re-anchor: tOld < 1 because we are mid-fade, so newFadePos < newFadeLen (still fading).
fadePos_ = static_cast<std::int64_t>(tOld * static_cast<double>(newFadeLen));
fadeLen_ = newFadeLen;
}
}
void PitchShifter::prime(const AudioSample* src, std::int64_t count) {
if (window_ <= 1) return; // pass-through needs no priming
// Clamp to one window: the intended call primes exactly window() frames, and delay ==
// count must stay inside the safe band so the seed does not itself trigger a splice.
if (count < 0) count = 0;
if (count > window_) count = window_;
std::fill(ring_.begin(), ring_.end(), 0.0f);
for (std::int64_t i = 0; i < count; ++i) ring_[static_cast<std::size_t>(i)] = src[i];
// Writer continues after the primed span; the tap parks ON src[0] (delay == count), so
// the very first process() output is src[0] — zero structural latency at every ratio.
writePos_ = count % ringLen_;
posA_ = posB_ = 0.0;
fading_ = false;
fadePos_ = 0;
fadeLen_ = 0;
filled_ = count;
tailFrozen_ = false; // a fresh note-on always starts with a live writer
lastSplice_ = SpliceEvent{};
// ratio_ deliberately untouched: the voice sets it per frame around the prime.
}
void PitchShifter::warm() {
if (window_ <= 1) return; // pass-through needs no warm-up
// A prime() with one window of silence: same geometry (tap parked mid-band one window
// behind the writer), the zeros declared as valid history. At unity this is a bit-exact
// window() delay; an up-shift plays ~a window of silence before speaking (the pre-GA2
// onset) — stream callers with access to the upcoming source should prime() instead.
std::fill(ring_.begin(), ring_.end(), 0.0f);
writePos_ = window_ % ringLen_;
posA_ = posB_ = 0.0;
fading_ = false;
fadePos_ = 0;
fadeLen_ = 0;
filled_ = window_;
tailFrozen_ = false;
lastSplice_ = SpliceEvent{};
}
void PitchShifter::setShiftRatio(double ratio) {
if (ratio > 0.0) ratio_ = ratio; // ignore non-positive (never run the tap backward/stall)
}
double PitchShifter::readTap(double pos) const {
// Fractional linear interpolation with ring wrap.
double p = pos;
const double len = static_cast<double>(ringLen_);
while (p < 0.0) p += len;
while (p >= len) p -= len;
const std::int64_t i0 = static_cast<std::int64_t>(p);
const double frac = p - static_cast<double>(i0);
std::int64_t i1 = i0 + 1;
if (i1 >= ringLen_) i1 = 0;
const double s0 = static_cast<double>(ring_[static_cast<std::size_t>(i0)]);
const double s1 = static_cast<double>(ring_[static_cast<std::size_t>(i1)]);
return s0 + (s1 - s0) * frac;
}
void PitchShifter::splice(std::int64_t nominalJump, double delay) {
// Relocate the active tap by `nominalJump` frames of ADDED delay (+window_ = jump toward
// older content, -window_ = jump toward the writer), refined by a correlation search so
// the relocated read point is waveform-aligned with the outgoing tap's upcoming content.
// The search is coarse (step 4 over +/- maxLag_) then fine (+/- 3 around the coarse best,
// then a parabolic sub-sample peak): a bounded burst of ~ (maxLag_/2 + 9) * corrFrames_
// multiply-adds, once per splice.
const std::int64_t d = static_cast<std::int64_t>(delay);
// GA2 onset fix: an up-jump may only relocate into VALID history. The deepest slot the
// search (and the +/-1-lag parabolic refinement calls at bestLag ± 1, and the interpolator's
// read-ahead) can touch is delay d + jump + maxLag + 2 (maxLag from the coarse/fine search,
// +1 for the parabola's outer ± 1 probe, +1 for the interpolator's i1 = i0+1 read-ahead),
// so the tight cap is filled_ - d - maxLag_ - 2. The code uses - 1 here — one sample LOOSER
// than that derived cap (not extra margin); ring indexing wraps via modulo everywhere, so
// this never runs off the physical ring_ array. In steady state (filled_ == ringLen_) this is
// > window_ and the nominal jump is untouched; near a primed onset it shrinks the jump to
// what real history exists (still many source periods with a full-window prime). The floor of
// 1 is only reachable on the documented degenerate reset-without-prime path — garbage-tolerant.
std::int64_t jump = nominalJump;
if (jump > 0) {
const std::int64_t maxJump = filled_ - d - maxLag_ - 1;
if (jump > maxJump) jump = maxJump;
if (jump < 1) jump = 1;
}
// The correlation reference reads FORWARD from the tap; keep it strictly behind the
// writer even when the trigger undershot dLow_ by a large per-frame drift (extreme
// up-ratios): d - corr must stay >= 0.
const std::int64_t corr = std::max<std::int64_t>(1, std::min<std::int64_t>(corrFrames_, d - 1));
const std::int64_t iA =
((static_cast<std::int64_t>(posA_) % ringLen_) + ringLen_) % ringLen_;
auto scoreAt = [&](std::int64_t lag) -> double {
std::int64_t ia = iA;
std::int64_t ic = ((iA - jump + lag) % ringLen_ + ringLen_) % ringLen_;
double s = 0.0, ec = 0.0;
for (std::int64_t k = 0; k < corr; ++k) {
const double a = static_cast<double>(ring_[static_cast<std::size_t>(ia)]);
const double c = static_cast<double>(ring_[static_cast<std::size_t>(ic)]);
s += a * c;
ec += c * c;
if (++ia >= ringLen_) ia = 0;
if (++ic >= ringLen_) ic = 0;
}
// NORMALIZED cross-correlation (standard SOLA): a raw dot product is biased toward
// the higher-energy lag, so on a decaying tail every up-splice would prefer the
// loudest candidate over the best-ALIGNED one — a small level step per splice that
// the amplitude-complementary fade cannot hide. The reference segment's energy is
// constant across lags, so dividing by sqrt(Ec) alone ranks identically to the full
// normalized form. A zero-energy candidate scores 0 (splicing into silence is benign).
return ec > 0.0 ? s / std::sqrt(ec) : 0.0;
};
std::int64_t bestLag = 0;
double bestScore = -std::numeric_limits<double>::infinity();
for (std::int64_t lag = -maxLag_; lag <= maxLag_; lag += 4) {
const double s = scoreAt(lag);
if (s > bestScore) {
bestScore = s;
bestLag = lag;
}
}
const std::int64_t coarse = bestLag;
for (std::int64_t lag = coarse - 3; lag <= coarse + 3; ++lag) {
if (lag == coarse || lag < -maxLag_ || lag > maxLag_) continue;
const double s = scoreAt(lag);
if (s > bestScore) {
bestScore = s;
bestLag = lag;
}
}
// SUB-SAMPLE peak (GA2 alias fix): the integer-lag best leaves a residual misalignment of
// up to half a sample; at the splice cadence that residual phase-modulates a pure tone
// into a ~-59 dB sideband comb (the DAW spectrogram "alias lines"). A parabola through
// the scores at bestLag-1/bestLag/bestLag+1 locates the correlation peak to a fraction of
// a sample; readTap()'s linear interpolation realizes the fractional tap position. The
// denominator is negative at a genuine peak — anything else (flat correlation: DC or
// silence) keeps the integer lag, which is already benign there.
double frac = 0.0;
{
const double sM = scoreAt(bestLag - 1);
const double sP = scoreAt(bestLag + 1);
const double den = sM - 2.0 * bestScore + sP;
if (den < 0.0) {
frac = 0.5 * (sM - sP) / den;
if (frac > 0.5) frac = 0.5;
if (frac < -0.5) frac = -0.5;
}
}
// Hand the current position to the outgoing tap and relocate the active one.
posB_ = posA_;
double p = posA_ - static_cast<double>(jump) + static_cast<double>(bestLag) + frac;
const double len = static_cast<double>(ringLen_);
while (p < 0.0) p += len;
while (p >= len) p -= len;
posA_ = p;
// RATIO-SCALED fade length. At an up-splice the OUTGOING tap starts at ~dLow_ delay and
// keeps draining toward the writer at (ratio - 1) per output frame; the nominal window/4
// fade only keeps it behind the writer for ratios up to 2. Beyond that (e.g. +24 st =
// ratio 4, an ordinary sampler transposition) it would cross mid-fade and play stale
// read-ahead data at substantial gain — a periodic seam. So cap the live fade at the
// frames of drain headroom actually available, minus 2 (1 for the trigger's sub-dLow_
// undershoot, 1 for the interpolator's read-ahead). Ratios <= ~2 keep the full nominal
// fade; ratio 4 gets ~window/12 — shorter but still a smooth burst. Down-shifts grow the
// outgoing delay at (1 - ratio) < 1 per frame and cannot reach the ring end within
// window/4 frames, so they always keep the full fade. A pitch-envelope ratio slew
// mid-fade is covered by the same margin for any realistic per-frame bias.
//
// TAIL-FROZEN (GA3): with the writer parked, the outgoing tap closes on it at the FULL
// ratio (there is no retreating write head), in EITHER shift direction — so the drain
// rate is ratio_ instead of (ratio_ - 1), and the cap applies at every ratio (unity
// included: splices fire in the frozen tail because the delay now drains at unity too).
fadeLen_ = fadeFrames_;
const double drainRate = tailFrozen_ ? ratio_ : (ratio_ - 1.0);
if (drainRate > 0.0) {
const double headroom = static_cast<double>(dLow_) - drainRate - 2.0;
// Clamp in double before the int64 cast to avoid UB at pathological near-unity ratios
// at very high sample rates (where headroom/drainRate could overflow int64).
const double safeDbl = headroom > 0.0
? std::min(headroom / drainRate, static_cast<double>(fadeFrames_))
: 1.0;
fadeLen_ = std::max<std::int64_t>(1, static_cast<std::int64_t>(safeDbl));
}
fading_ = true;
fadePos_ = 0;
// Record the decision for a linked follower channel (T1-01): the follower applies this
// verbatim so both channels share one lag and one splice schedule.
lastSplice_ = SpliceEvent{true, jump, bestLag, frac, fadeLen_};
}
void PitchShifter::applySplice(const SpliceEvent& ev) {
// Follower half of the T1-01 linked lag: relocate + fade with the master's decision, no
// correlation search of our own. The master's jump was clamped against ITS filled_/delay,
// which match ours by the lockstep contract (identical configure/prime/ratio history);
// the fade length likewise derives only from shared geometry + ratio.
posB_ = posA_;
double p = posA_ - static_cast<double>(ev.jump) + static_cast<double>(ev.lag) + ev.frac;
const double len = static_cast<double>(ringLen_);
while (p < 0.0) p += len;
while (p >= len) p -= len;
posA_ = p;
fadeLen_ = std::max<std::int64_t>(1, ev.fadeLen);
fading_ = true;
fadePos_ = 0;
lastSplice_ = ev; // observable mirror (tests assert follower == master per frame)
}
AudioSample PitchShifter::process(AudioSample in) { return processImpl(in, nullptr); }
AudioSample PitchShifter::processLinked(AudioSample in, const SpliceEvent& master) {
return processImpl(in, &master);
}
AudioSample PitchShifter::processImpl(AudioSample in, const SpliceEvent* linked) {
if (window_ <= 1) return in; // pass-through (unconfigured / degenerate)
// Copy the linked decision BEFORE clearing lastSplice_ (guards a self-aliased pointer;
// 5 plain fields, negligible on the RT path).
const SpliceEvent linkedEv = linked != nullptr ? *linked : SpliceEvent{};
lastSplice_ = SpliceEvent{}; // cleared every frame; set again if this frame splices
// 1. Write the incoming sample at the write head (source rate). One more slot of the
// ring now holds valid history (capped at the ring length once it has wrapped).
// TAIL-FROZEN (GA3): the source is exhausted — `in` is padding, not stream. Write
// NOTHING (the ring keeps its all-real final two windows) and hold the write head;
// the read/splice/fade machinery below runs unchanged over the frozen content.
if (!tailFrozen_) {
ring_[static_cast<std::size_t>(writePos_)] = in;
if (filled_ < ringLen_) ++filled_;
}
// 2. Read the active tap; while a splice fade is live, crossfade against the outgoing tap.
// Raised-cosine COMPLEMENTARY gains (gNew + gOld == 1): correlation-aligned content is
// in phase, so the sum holds unity amplitude through the fade (equal-power would bulge).
double out = readTap(posA_);
if (fading_) {
const double t = static_cast<double>(fadePos_) / static_cast<double>(fadeLen_);
const double gNew = 0.5 * (1.0 - std::cos(kPi * t));
out = gNew * out + (1.0 - gNew) * readTap(posB_);
if (++fadePos_ >= fadeLen_) fading_ = false;
} else if (linked != nullptr) {
// 3a. FOLLOWER (T1-01): no trigger test, no search — splice exactly when and how the
// master channel did this frame. Lockstep state means our own trigger would have
// fired on the same frame; applying the master's decision keeps the two rings
// sample-aligned (one shared lag, one shared schedule).
if (linkedEv.fired) {
applySplice(linkedEv);
} else {
// Self-healing fallback (review rider): the master not firing normally means this
// channel's own trigger wouldn't fire either (lockstep). But if the processor ever
// renders a mono block mid-note, this follower channel is skipped for that block
// while the master keeps advancing — its writePos_/filled_ falls behind and, with
// only the `if (linkedEv.fired)` path above, could never resync. So check this
// follower's OWN tap distance against the safe band and splice via its own search
// when it has left [dLow_, dHigh_], exactly as the master would. Reuses splice() —
// no allocation, no new RT cost. In the normal (non-mono-block) case this branch
// never triggers: the master's trigger fires first and this whole `if` is false.
double d = static_cast<double>(writePos_) - posA_;
const double len = static_cast<double>(ringLen_);
while (d < 0.0) d += len;
while (d >= len) d -= len;
if (d <= static_cast<double>(dLow_)) {
splice(+window_, d);
} else if (d >= static_cast<double>(dHigh_)) {
splice(-window_, d);
}
}
} else {
// 3. Splice scheduling: relocate when the active tap's delay leaves the safe band.
// Up-shifts (ratio > 1) drain the delay toward 0 -> jump one window OLDER; down-
// shifts grow it toward the ring length -> jump one window TOWARD the writer. At
// unity the delay is frozen at window/2 and neither trigger ever fires.
double d = static_cast<double>(writePos_) - posA_;
const double len = static_cast<double>(ringLen_);
while (d < 0.0) d += len;
while (d >= len) d -= len;
if (d <= static_cast<double>(dLow_)) {
splice(+window_, d);
} else if (d >= static_cast<double>(dHigh_)) {
splice(-window_, d);
}
}
// 4. Advance heads: write head one frame (source rate; parked while tail-frozen),
// tap(s) by the shift ratio.
if (!tailFrozen_) {
++writePos_;
if (writePos_ >= ringLen_) writePos_ = 0;
}
const double len = static_cast<double>(ringLen_);
posA_ += ratio_;
while (posA_ >= len) posA_ -= len;
if (fading_) {
posB_ += ratio_;
while (posB_ >= len) posB_ -= len;
}
return static_cast<AudioSample>(out);
}
} // namespace reasampler::instrument::engine
+230
View File
@@ -0,0 +1,230 @@
#pragma once
// pitch_shift — a PURE, per-voice, duration-preserving pitch shifter: the S16 "Preserve"
// engine's DSP core. Time-domain delay-line shifter with CORRELATION-ALIGNED SPLICES
// (SOLA-style): one active read tap chases the write head at the shift ratio; when it drifts
// out of its safe delay band it is relocated by a nominal window jump REFINED BY A
// CROSS-CORRELATION SEARCH so the new read point is waveform-aligned, then the old and new
// taps are crossfaded (raised-cosine, amplitude-complementary). Source is consumed 1:1 and
// output produced 1:1 (duration held); only the PITCH changes — an octave up plays the same
// wall-clock length as the root note, unlike the Varispeed `readPos_ += ratio_` resample path.
//
// WHY CORRELATED SPLICES (GA-Preserve fix, 2026-07). The first S16 implementation was the
// naive two-tap OLA: taps hard-locked half a window apart, Hann-crossfaded by write-head
// distance. Its taps read the same stream at delays differing by exactly w/2, so their outputs
// carried a FIXED relative phase of 2*pi*f_src*(w/2) — arbitrary and source-frequency-
// dependent. Near anti-phase (roughly half of all frequencies) every crossfade midpoint
// nearly CANCELLED: deep periodic AM + phase slew = strong sidebands. A repitched pure sine
// came out mangled ("multiple partials" on a spectrogram) while the root stayed clean (unity
// freezes the crossfade). The fix is structural: splices must be PHASE-ALIGNED, so each jump
// is snapped to the best waveform match within a bounded lag search — a pure sine's jump
// lands on an integer period count and the output stays a single shifted tone.
//
// WHY A HAND-ROLLED PURE MODULE, NOT WDL (S16-F2, decided at build). The spec's lean was
// route (a) `WDL_SimplePitchShifter`. But its include chain
// (simple_pitchshift.h -> queue.h -> heapbuf.h -> wdltypes.h) does `#ifdef _WIN32 ->
// #include <windows.h>` unconditionally, which CANNOT enter the pure sampler_core module
// (CLAUDE.md load-bearing split: NO vendor/host/SDK types; sampler_core_tests links neither
// SDK and compiles outside the DAW). So the Preserve DSP lands as route (b): a house-native
// pure module alongside peaks / wav_trim, CTest-testable, RT-disciplined. Same
// PitchEngine::Preserve contract behind the seam — if WDL is ever preferred it swaps in at
// the SHELL, never in the pure core.
//
// WHY PRIME WITH REAL CONTENT (GA2-Preserve onset fix, 2026-07). Splices RELOCATE the tap
// into ring HISTORY — at note onset a silence-warmed ring has none, so every early splice
// jumped into zeros: a burst/gap/burst stutter for the first ~2 windows of every off-root
// note (the DAW "zero-sample gaps in the first few ms"; at +48 st the ~300 Hz gap cadence
// reads as a square-ish buzz). But this engine is NOT a streaming context: the caller owns
// the whole decoded sample, so the FUTURE of the stream is known at note-on. `prime()`
// pre-fills the ring with the actual first window of upcoming source and parks the tap on
// its oldest frame — output frame 0 IS source frame 0 (zero structural latency at every
// ratio), and `splice()` clamps its jump to the really-filled span so no splice can ever
// land in unwritten silence.
//
// WHY FREEZE THE TAIL (GA3-Preserve tail fix, 2026-07). GA2's prime fixed the ONSET; the
// mirror problem lived at the note END. When the source ran out, the caller held the LAST
// REAL SAMPLE as the feed — a DC plateau with no waveform for the correlation to align on.
// Splices landing in or referenced against it were unalignable, so the tap alternated
// real-tone / dead-DC at the splice cadence, the dead fraction growing as the plateau
// displaced real ring history (the DAW report: periodic troughs "almost like ring
// modulation", ~1:20 tone-to-silence at the very end). freezeTail() removes the padding at
// the source: the WRITER parks, the ring keeps its all-real final two windows, and the
// aligned-splice machinery recycles that frozen tail — a continuous tone until the caller's
// own note end. See freezeTail() below.
//
// PURE MODULE: NO VST3, NO REAPER, NO SWELL, NO vendor/ includes. Standard library only.
// Shares the `AudioSample` float alias from peaks (the one house precedent — sampler_core /
// wav_trim do the same).
//
// RT DISCIPLINE (S16 hard constraint). `configure()` sizes the ring ONCE (off the audio
// thread, at voice allocation). `prime()` / `warm()` only copy into the pre-sized ring
// (bounded, allocation-free — safe on the audio thread at note-on). `process()` does
// NO allocation and NO locks — it reads/writes the pre-sized ring only. The splice-time
// correlation search is a bounded burst of multiply-adds (coarse+refine over a fixed lag
// range) that fires once per splice cadence (window / |ratio-1| frames), never per frame.
#include <cstddef>
#include <cstdint>
#include <vector>
#include "core/audio/peaks.h" // AudioSample (float)
namespace reasampler::instrument::engine {
using audio::AudioSample;
// The splice decision made by the most recent process()/processLinked() call — the LINKED-LAG
// stereo contract (Q-W0 T1-01). A stereo voice runs channel 0 as the MASTER (full correlation
// search) and channel 1 as the FOLLOWER: after the master's process() for a frame, the caller
// passes master.lastSplice() to the follower's processLinked() for the SAME frame, and the
// follower applies exactly this decision instead of running its own search. Both channels
// therefore share one lag and one splice schedule (standard stereo SOLA) — per-channel
// independent searches re-drew an inter-channel offset of up to +/-maxLag at every splice:
// stereo image wander at the splice cadence plus comb coloration on any mono sum.
struct SpliceEvent {
bool fired = false; // a splice was scheduled on this frame
std::int64_t jump = 0; // the CLAMPED nominal jump actually applied (signed)
std::int64_t lag = 0; // correlation best integer lag
double frac = 0.0; // parabolic sub-sample refinement, [-0.5, 0.5]
std::int64_t fadeLen = 0; // live (ratio-scaled) crossfade length chosen
};
// A per-channel time-domain splice-aligned pitch shifter. One instance transposes ONE channel;
// a stereo voice owns two, LINKED: channel 0 is the master, channel 1 follows its splice
// decisions via processLinked() (see SpliceEvent above) so the two rings stay sample-aligned.
//
// The default-constructed shifter is INERT: with no configure() it passes input through
// unchanged (shift ratio 1.0, empty ring), so a Varispeed voice that never touches it is
// byte-identical to the pre-S16 engine.
class PitchShifter {
public:
// Size the delay ring for `windowFrames` (the nominal splice-jump length; the ring is 2x
// that for splice/search headroom) and derive the fade/search geometry. `windowFrames`
// <= 1 degrades to pass-through (no ring), so a degenerate configure never divides by
// zero or wraps a zero span. Called OFF the audio thread (allocates). Resets all running
// state. A larger window = fewer splices and a deeper alignment search; a PRIMED shifter
// has no added latency regardless (see prime()); the shell picks it from kPreserveWindowMs.
void configure(std::int64_t windowFrames);
// Pre-fill the ring with the first `count` frames of the UPCOMING source stream and park
// the tap on src[0] (delay == count, mid safe band at count == window()). The caller then
// feeds process() the stream CONTINUING at src[count]. Output frame 0 is src[0]: ZERO
// structural latency at every ratio, and splices always have `count` frames of real
// history to land in — the GA2 onset-gap fix. `count` is clamped to [0, window()].
// When the PLAYABLE source is shorter than one window, prime only the real span and call
// freezeTail() immediately after (Q-W0 T1-03): the GA3 machinery then recycles the real
// short tail. Do NOT pad with silence and declare it valid — padded zeros inside the ring
// are splice targets, re-creating the pre-GA2 burst/gap onset on sub-window material.
// RT-safe: bounded copy into the pre-sized ring, no allocation. No-op when unconfigured.
// The current shift ratio is left untouched.
void prime(const AudioSample* src, std::int64_t count);
// prime()-with-silence: zero the ring, park the tap one window behind the writer, and
// declare that window of silence as valid history. Kept for callers with no access to the
// upcoming stream (a silence-primed up-shift plays ~a window of silence before speaking —
// the pre-GA2 onset; the Voice path uses prime() instead). At unity a warmed shifter is a
// bit-exact window() delay. No-op when unconfigured.
void warm();
// The pitch shift ratio: 2^((note - root)/12) plus any per-frame pitch-envelope bias.
// 1.0 = no shift (pass-through-equivalent output, no splices ever fire). Set per frame is
// fine (cheap); the tap advance simply uses the current value. Values <= 0 are ignored
// (kept at the last valid ratio) so a bad input never runs the tap backward or stalls it.
void setShiftRatio(double ratio);
// Transform ONE input frame into ONE output frame (duration-preserving: 1 in, 1 out).
// RT-safe: reads/writes the pre-sized ring only, no allocation, no lock. When unconfigured
// (window <= 1) returns `in` unchanged (pass-through). Otherwise writes `in` at the write
// head, reads the active tap (crossfading against the outgoing tap while a splice fade is
// live), then advances the write head by one and the tap(s) by the shift ratio. When the
// active tap leaves its safe delay band, a correlation-aligned splice is scheduled.
AudioSample process(AudioSample in);
// FOLLOWER-mode process (Q-W0 T1-01, the stereo linked lag): identical to process()
// except the splice decision is NOT computed here — when `master.fired` is true this
// frame splices with exactly the master's jump/lag/frac/fadeLen; otherwise no splice is
// considered. The caller must process the master channel FIRST each frame and pass its
// lastSplice() here, with both shifters configured/primed/ratio'd identically — their
// ring state then advances in lockstep, so the follower's own trigger would have fired
// on the same frame anyway; skipping its search only removes the second correlation
// burst (strictly cheaper, never costlier). RT-safe: same guarantees as process().
AudioSample processLinked(AudioSample in, const SpliceEvent& master);
// The splice decision made by the most recent process()/processLinked() call (fired ==
// false when that frame spliced nothing). Feed to a follower channel's processLinked().
const SpliceEvent& lastSplice() const { return lastSplice_; }
// TAIL WIND-DOWN (GA3, 2026-07). Call when the SOURCE STREAM IS EXHAUSTED — no real frame
// remains to feed process(). Freezes the WRITE head: subsequent process() calls ignore
// their input and write nothing, but read, splice, and crossfade exactly as before over
// the ring's frozen (all-real) final two windows. WHY: the pre-GA3 tail held the last
// real sample as the feed — a DC plateau with no waveform to correlate on. Splices
// landing in or referenced against it were unalignable, so the tap alternated real-tone /
// dead-DC at the splice cadence (the DAW "ring modulation" troughs, growing toward the
// note end as the plateau displaced real history). With the writer frozen the padding
// never enters the ring: every splice stays waveform-aligned against real content and
// the output remains a continuous tone — the final <= one window recycles the frozen
// tail (correlation-aligned, crossfaded) instead of decaying into chopped DC, and the
// caller's own note end (its output-frame anchor) bounds how long that lasts. Idempotent;
// RT-safe (flag + bounded arithmetic, no allocation); cleared by reset()/prime()/warm().
void freezeTail();
bool tailFrozen() const { return tailFrozen_; }
// Reset running state to silence (ring zeroed, heads re-seeded mid-band, fill count zeroed)
// WITHOUT reallocating — for voice reuse without a re-configure. Keeps the current window.
// Follow with prime() (or warm()) before streaming: a bare reset has no declared history,
// so an immediate up-shift would starve its splices.
void reset();
// True once configure() sized a real ring (window > 1). A pass-through shifter is false.
bool configured() const { return window_ > 1; }
std::int64_t window() const { return window_; }
private:
double readTap(double pos) const; // fractional ring read, linear interp
// Relocate the active tap by ~`nominalJump` frames of added delay (clamped to the filled
// span for up-jumps) and start the crossfade. `delay` is the tap's current delay behind
// the writer (the caller just computed it for the trigger test). Records the decision in
// lastSplice_ for a linked follower channel.
void splice(std::int64_t nominalJump, double delay);
// Apply a master channel's already-computed splice decision verbatim (no search) —
// the follower half of the T1-01 linked-lag contract. Mirrors it into lastSplice_.
void applySplice(const SpliceEvent& ev);
// Shared body of process()/processLinked(); `linked` null = master mode (own trigger +
// search), non-null = follower mode (splice iff linked->fired, with linked's decision).
AudioSample processImpl(AudioSample in, const SpliceEvent* linked);
std::vector<AudioSample> ring_; // delay line, length `ringLen_` == 2 * window_
std::int64_t window_ = 0; // nominal splice jump in frames; <= 1 = pass-through
std::int64_t ringLen_ = 0; // ring length (2 * window_): splice + search headroom
std::int64_t writePos_ = 0; // integer write head into the ring (source rate)
double posA_ = 0.0; // active read tap (advances at the shift ratio)
double posB_ = 0.0; // outgoing tap during a splice crossfade
bool fading_ = false; // a splice crossfade is in flight
std::int64_t fadePos_ = 0; // crossfade progress, [0, fadeLen_)
std::int64_t fadeFrames_ = 0; // NOMINAL crossfade length (window_/4)
std::int64_t fadeLen_ = 0; // LIVE crossfade length for the in-flight splice —
// ratio-scaled at splice time so an up-shift's outgoing
// tap can never drain into the writer mid-fade
std::int64_t maxLag_ = 0; // correlation search half-range (window_/4)
std::int64_t corrFrames_ = 0; // correlation segment length (dLow_-1, capped at 512, so
// the reference read forward from the tap stays behind
// the writer BY CONSTRUCTION at an up-splice)
std::int64_t dLow_ = 0; // splice trigger: active-tap delay below this (up-shift)
std::int64_t dHigh_ = 0; // splice trigger: active-tap delay above this (down-shift)
std::int64_t filled_ = 0; // frames of VALID history behind the writer (prime count
// + frames streamed, capped at ringLen_). splice() clamps
// its up-jump to this so no splice lands in unwritten
// silence — the GA2 onset-gap fix.
double ratio_ = 1.0; // current shift ratio (>0)
SpliceEvent lastSplice_{}; // decision of the most recent process*() frame (T1-01):
// cleared at the top of every frame, set on a splice
bool tailFrozen_ = false; // GA3 wind-down: writer frozen (source exhausted); the tap
// recycles the ring's frozen real tail, splices still
// aligned. With the writer parked, a tap drains toward it
// at ratio_ (not ratio_-1) per frame — splice() scales the
// live fade by that rate.
};
} // namespace reasampler::instrument::engine
+994
View File
@@ -0,0 +1,994 @@
// sampler_core — pure sampler engine implementation. See sampler_core.h for the
// contract and the design rationale (keymap resolution, pitch ratio, ADSR shape,
// voice allocation + stealing policy). NO VST3 / REAPER / SWELL / vendor includes.
#include "core/instrument/engine/sampler_core.h"
#include <cmath>
namespace reasampler {
// ---------------------------------------------------------------------------
// pitchRatio
// ---------------------------------------------------------------------------
double pitchRatio(int note, int rootNote) {
// Equal temperament: each semitone is a factor of 2^(1/12). note == root -> 1.0.
return std::pow(2.0, static_cast<double>(note - rootNote) / 12.0);
}
double keyTrackedRatio(int note, int rootNote, double keyTrack) {
// Scale the semitone offset by keyTrack before the ET conversion. keyTrack == 1.0 yields
// (note-root)*1.0, which is EXACT in IEEE-754 for an integer-valued double, so the argument
// to std::pow is bit-identical to pitchRatio(note, rootNote) — the 100% default is byte-for-
// byte unchanged from the pre-S-VIEW-6 engine. keyTrack == 0.0 -> offset 0 -> ratio 1.0 on
// every key (no tracking); keyTrack == 2.0 -> doubled offset. Root note stays unity always.
const double semis = static_cast<double>(note - rootNote) * keyTrack;
return std::pow(2.0, semis / 12.0);
}
// ---------------------------------------------------------------------------
// Keymap
// ---------------------------------------------------------------------------
ZoneResolution Keymap::resolve(int note, int velocity) const {
(void)velocity; // accepted for the Tier-2 seam; does not select at Tier 0-1.
for (std::size_t i = 0; i < zones.size(); ++i) {
const KeyZone& z = zones[i];
if (note >= z.lowNote && note <= z.highNote) {
return ZoneResolution{true, i};
}
}
return ZoneResolution{false, 0};
}
Keymap Keymap::singleSampleChromatic(SampleData sample) {
const int root = sample.rootNote;
Keymap km;
km.samples.push_back(std::move(sample));
KeyZone zone;
zone.lowNote = 0;
zone.highNote = 127;
zone.rootNote = root;
zone.sampleIndex = 0;
km.zones.push_back(zone);
return km;
}
// ---------------------------------------------------------------------------
// AdsrEnvelope
// ---------------------------------------------------------------------------
void AdsrEnvelope::noteOn() {
stage_ = Stage::Attack;
level_ = 0.0;
framesInStage_ = 0;
}
void AdsrEnvelope::noteOff() {
if (stage_ == Stage::Idle || stage_ == Stage::Finished ||
stage_ == Stage::Release) {
return; // already released / not sounding.
}
// Release from the CURRENT level — release-before-sustain releases from the
// partial attack/decay level, not from sustainLevel.
releaseFrom_ = level_;
stage_ = Stage::Release;
framesInStage_ = 0;
}
double AdsrEnvelope::tick() {
switch (stage_) {
case Stage::Idle:
case Stage::Finished:
level_ = 0.0;
return 0.0;
case Stage::Attack: {
if (params_.attackFrames <= 0) {
level_ = 1.0;
} else {
level_ = static_cast<double>(framesInStage_) /
static_cast<double>(params_.attackFrames);
if (level_ > 1.0) level_ = 1.0;
}
const double out = level_;
++framesInStage_;
if (framesInStage_ >= params_.attackFrames) {
// S15: Attack -> Hold (holds 1.0 for holdFrames). holdFrames == 0 falls straight
// through Hold on the next tick to Decay, which is EXACTLY the pre-S15 A->D path.
stage_ = Stage::Hold;
framesInStage_ = 0;
level_ = 1.0;
}
return out;
}
case Stage::Hold: {
// S15 hold stage: level pinned at 1.0 for holdFrames. holdFrames <= 0 leaves the
// stage on this same tick (no frame consumed at 1.0 beyond what Attack already
// emitted), so hold=0 is byte-identical to the pre-S15 envelope.
if (params_.holdFrames <= 0) {
stage_ = Stage::Decay;
framesInStage_ = 0;
// Fall through to Decay this frame so no extra unity sample is emitted for a
// zero-length hold (preserving the exact pre-S15 sample-for-sample shape).
level_ = 1.0;
// Single re-dispatch into Decay (bounded: Hold→Decay only; not a general recursion).
return tick();
}
level_ = 1.0;
const double out = level_;
++framesInStage_;
if (framesInStage_ >= params_.holdFrames) {
stage_ = Stage::Decay;
framesInStage_ = 0;
level_ = 1.0;
}
return out;
}
case Stage::Decay: {
if (params_.decayFrames <= 0) {
level_ = params_.sustainLevel;
} else {
const double t = static_cast<double>(framesInStage_) /
static_cast<double>(params_.decayFrames);
level_ = 1.0 + (params_.sustainLevel - 1.0) * t;
}
const double out = level_;
++framesInStage_;
if (framesInStage_ >= params_.decayFrames) {
stage_ = Stage::Sustain;
framesInStage_ = 0;
level_ = params_.sustainLevel;
}
return out;
}
case Stage::Sustain:
level_ = params_.sustainLevel;
return level_;
case Stage::Release: {
if (params_.releaseFrames <= 0) {
level_ = 0.0;
stage_ = Stage::Finished;
return 0.0;
}
const double t = static_cast<double>(framesInStage_) /
static_cast<double>(params_.releaseFrames);
level_ = releaseFrom_ * (1.0 - t);
if (level_ < 0.0) level_ = 0.0;
const double out = level_;
++framesInStage_;
if (framesInStage_ >= params_.releaseFrames) {
stage_ = Stage::Finished;
level_ = 0.0;
}
return out;
}
}
return 0.0; // unreachable; silences a warning.
}
// ---------------------------------------------------------------------------
// TriggerEnvelope (S15) — a time-boxed fade-in/hold/fade-out amplitude function.
// ---------------------------------------------------------------------------
void TriggerEnvelope::configure(std::int64_t playLengthFrames, std::int64_t fadeInFrames,
std::int64_t fadeOutFrames, FadeCurve curve) {
playLength_ = playLengthFrames > 0 ? playLengthFrames : 0;
curve_ = curve;
finished_ = (playLength_ <= 0);
// Clamp the fades so fadeIn + fadeOut <= playLength (fade-out anchored to the end). A
// negative fade is treated as 0. When both fades together exceed the play length, shrink
// the fade-out first (the head fade-in is the more perceptually load-bearing onset ramp),
// then the fade-in — never letting either go negative or the sum exceed the span.
std::int64_t fi = fadeInFrames > 0 ? fadeInFrames : 0;
std::int64_t fo = fadeOutFrames > 0 ? fadeOutFrames : 0;
if (fi > playLength_) fi = playLength_;
if (fi + fo > playLength_) fo = playLength_ - fi; // fo >= 0 since fi <= playLength_
fadeIn_ = fi;
fadeOut_ = fo;
}
double TriggerEnvelope::amplitudeAt(double sourceOffset) {
if (finished_ || sourceOffset < 0.0 ||
sourceOffset >= static_cast<double>(playLength_)) {
// At/past the play length the one-shot is done; the voice also frees on readPos >= playEnd.
if (sourceOffset >= static_cast<double>(playLength_)) finished_ = true;
return 0.0;
}
// Fade-in: 0->1 over [0, fadeIn_). Fade-out: 1->0 over [playLength_-fadeOut_, playLength_).
// Unity between. The two ramps never overlap (configure clamps fadeIn_ + fadeOut_ <= length).
// The offset is fractional (the read head is fractional under repitch), so the ramps are
// smooth rather than stepped.
double amp = 1.0;
const double foStart = static_cast<double>(playLength_ - fadeOut_);
if (fadeIn_ > 0 && sourceOffset < static_cast<double>(fadeIn_)) {
const double phase = sourceOffset / static_cast<double>(fadeIn_); // 0..1
amp = (curve_ == FadeCurve::EqualPower)
? std::sin(phase * 1.5707963267948966) // sin(phase*pi/2): 0->1 constant power
: phase;
} else if (fadeOut_ > 0 && sourceOffset >= foStart) {
const double phase = (sourceOffset - foStart) / static_cast<double>(fadeOut_); // 0..1
amp = (curve_ == FadeCurve::EqualPower)
? std::cos(phase * 1.5707963267948966) // cos(phase*pi/2): 1->0 constant power
: (1.0 - phase);
}
return amp;
}
// ---------------------------------------------------------------------------
// PitchEnvelope (S16) — AD pitch offset in semitones, off when disabled.
// ---------------------------------------------------------------------------
double PitchEnvelope::tick() {
if (!params_.enabled) return 0.0;
const std::int64_t a = params_.attackFrames > 0 ? params_.attackFrames : 0;
const std::int64_t d = params_.decayFrames > 0 ? params_.decayFrames : 0;
const double peak = params_.peakSemitones;
double offset;
if (pos_ < a) {
// Attack: 0 -> peak over attackFrames (rise into the peak).
offset = peak * (static_cast<double>(pos_) / static_cast<double>(a));
} else if (pos_ < a + d) {
// Decay: peak -> 0 over decayFrames (settle to base pitch).
const double t = static_cast<double>(pos_ - a) / static_cast<double>(d);
offset = peak * (1.0 - t);
} else {
offset = 0.0; // past attack+decay: at base pitch forever.
}
++pos_;
return offset;
}
// ---------------------------------------------------------------------------
// Voice
// ---------------------------------------------------------------------------
void Voice::presizePreserveShifters(std::int64_t windowFrames) {
// OFF the audio thread (allocates). Both channels are sized so a stereo Preserve voice needs
// no allocation at note-on; a mono Preserve voice simply never process()es shiftR_. The
// prime scratch (one window, reused per channel) is sized here for the same reason: start()
// assembles the first window of the upcoming source stream into it with zero allocation.
shiftL_.configure(windowFrames);
shiftR_.configure(windowFrames);
primeBuf_.assign(windowFrames > 1 ? static_cast<std::size_t>(windowFrames) : 0, 0.0f);
}
bool Voice::sustainLoopUsable() const {
if (sample_ == nullptr || playMode_ != PlayMode::Gate) return false;
const SampleLoop& loop = sample_->loop;
return loop.hasLoop && loop.end > loop.start && loop.start >= 0 &&
loop.end <= static_cast<std::int64_t>(sample_->frames.size());
}
void Voice::start(int note, int velocity, const SampleData& sample, int rootNote,
double keyTrack, const VelocityCurve& velocityCurve,
bool declickTakeover) {
// Takeover declick (Phase S GA fix, rev 2): BEFORE any state reset, record the PRE-CUT
// REFERENCE — the last rendered output — and mark the compensation PENDING iff this
// start is a takeover/steal of a SOUNDING voice and the caller opted in. The ramp itself
// is seeded on the FIRST frame rendered after the restart, from the DIFFERENCE between
// this reference and the new voice's raw output that frame (seedDeclick), so the
// boundary frame reproduces the old level EXACTLY — whatever the new envelope does
// (Gate attack, zero attack, Trigger's no-fade-in instant-unity onset) and whatever
// value the new sample starts on. [Rev 1 seeded the OLD value here and gated the add by
// (1 newAmp) in the epilogue: every restart whose new amplitude was instantly ~1 got
// ZERO compensation and kept the full click — exactly the DAW-reported mono-retrig case
// on Trigger / zero-attack zones.] A fresh start (idle voice) clears the declick state —
// no phantom ramp. lastOut{L,R}_ are deliberately NOT zeroed here: a SECOND same-block
// takeover (two steals of this voice with no frame rendered between) must record the
// same pre-cut reference, not a phantom 0. The next rendered frame overwrites lastOut.
if (declickTakeover && active_) {
// Clamp the reference to ±1.0 full scale: a bounded seed whatever the voice was doing.
declickRefL_ = (lastOutL_ > 1.0) ? 1.0 : (lastOutL_ < -1.0) ? -1.0 : lastOutL_;
declickRefR_ = (lastOutR_ > 1.0) ? 1.0 : (lastOutR_ < -1.0) ? -1.0 : lastOutR_;
declickPending_ = true;
} else {
declickPending_ = false;
}
// Any in-flight ramp is superseded: pending re-derives from the reference, which already
// includes the running declick's contribution via lastOut (it tracks post-declick output).
declickActive_ = false;
declickWeight_ = 0.0;
active_ = true;
releasing_ = false;
amplitudeDone_ = false;
note_ = note;
// S-VIEW-9: the velocity->amp transfer curve maps MIDI velocity to gain, ONCE at note-on (the
// per-frame render just multiplies the cached velocityGain_ — no new process-thread work). The
// clamp lives inside eval (velocity box-clamped to [0,127]). Replaces the pre-r10 linear
// velocity/127; the default flat y=1 curve (R10-F1 Option A) plays every velocity at unity.
velocityGain_ = velocityCurve.eval(static_cast<double>(velocity));
// S-VIEW-6: the key-tracked repitch ratio feeds BOTH engines through baseRatio_ (Varispeed
// read-rate bias and Preserve shift amount both derive from it below). keyTrack == 1.0 is
// the pre-S-VIEW-6 pitchRatio bit-for-bit.
baseRatio_ = keyTrackedRatio(note, rootNote, keyTrack);
sample_ = &sample;
const ZonePlayParams& p = sample.play;
playMode_ = p.playMode;
pitchEngine_ = p.pitchEngine;
// Initial read position honors the sample's start-point offset (S11), in BOTH modes. Clamp
// into [0, frames): a start at or past the end degrades to 0 (play from the top) rather than
// starting a voice already off the end. A negative start (shouldn't occur) is pinned to 0.
const std::int64_t frameCount = static_cast<std::int64_t>(sample.frames.size());
std::int64_t start = sample.startFrame;
if (start < 0 || start >= frameCount) start = 0;
readPos_ = static_cast<double>(start);
startFrame_ = start; // Trigger fade offset origin (readPos - startFrame = span offset)
// --- Amplitude envelope: Gate = AHDSR (fully per-zone: A/H/D/S/R all read from the zone's
// play.adsr); Trigger = the time-boxed fade-in/out over the % play length.
//
// All five AHDSR fields come from sample.play.adsr (in FRAMES), resolved by
// buildTier0Keymap / buildZonedKeymap at reload time from the stored SECONDS against
// the live sample rate.
//
// Back-compat invariant: a zone whose stored ADSR seconds carry the tier-0 defaults
// (resolved to frames at the live sample rate) sounds identical to the pre-S12 build at
// every DAW rate — now trivially true, since the times are wall-clock seconds. ---
if (playMode_ == PlayMode::Gate) {
env_.configure(p.adsr);
env_.noteOn();
playEnd_ = 0; // unused in Gate
} else {
// Trigger: play [start, playEnd) where playEnd = start + round(lengthFraction*(frames-start)).
double frac = p.trigger.lengthFraction;
if (frac <= 0.0) frac = 0.0; // %=0 -> zero play length (finishes immediately)
if (frac > 1.0) frac = 1.0;
const std::int64_t span = frameCount - start; // >= 1 (start clamped < frameCount)
std::int64_t playLen = static_cast<std::int64_t>(
static_cast<double>(span) * frac + 0.5); // round
if (playLen < 0) playLen = 0;
if (playLen > span) playLen = span;
playEnd_ = start + playLen;
trigEnv_.configure(playLen, p.trigger.fadeInFrames, p.trigger.fadeOutFrames,
kDefaultFadeCurve);
}
// --- Pitch envelope (S16): per-voice AD, off by default (offset always 0). ---
pitchEnv_.configure(p.pitchEnv);
pitchEnv_.noteOn();
// --- Preserve engine (S16, GA2 onset fix): PRIME the ALREADY-SIZED per-channel shifters
// with the first window of the ACTUAL upcoming source stream (loop-unrolled under the
// sustain-loop wrap rule, silence past the sample end — that silence IS the true
// stream there). The tap parks on source frame `start`, so the voice speaks on output
// frame 0 at EVERY ratio (no ring-fill silence), and every splice has a full window
// of real history to land in — the fix for the DAW onset zero-gaps (a silence-warmed
// ring made every early splice jump into zeros). The rings and the prime scratch were
// allocated off-thread by presizePreserveShifters (the engine calls it at
// construction); this path is a bounded copy — NO allocation here. Varispeed voices
// never touch the shifters (advanceFrame checks configured()), so a Varispeed
// instrument is byte-identical to pre-S16 and pays no per-frame shifter cost. ---
if (pitchEngine_ == PitchEngine::Preserve && shiftL_.configured()) {
const std::int64_t w = shiftL_.window();
const bool loopWrap = sustainLoopUsable();
const SampleLoop& loop = sample.loop;
const std::int64_t loopLen = loopWrap ? (loop.end - loop.start) : 0;
const bool stereoSample = sample.channelCount() == 2 && shiftR_.configured();
// Q-W0 T1-03: the prime may only carry PLAYABLE source. The per-frame feed stops at
// feedBound (playEnd_ for a bounded Trigger span, the sample end for Gate) and
// freezes the writer there (GA3) — but the prime used to pull a FULL window bounded
// only by frameCount: a Trigger ring held real PCM past the user's chosen stop (an
// up-shifted tap could play it, transposed, before the voice freed), and a
// shorter-than-window sample got zero padding declared as valid history (splices
// landing in silence — the pre-GA2 burst/gap onset, re-entered for sub-window
// material). So bound the prime by the same playable span and, when that span is
// shorter than a window, freeze the tail IMMEDIATELY after the prime — the GA3
// machinery then recycles the real short tail, its designed behavior. The sustain-
// loop path is unbounded by construction (the wrap keeps q inside the loop forever).
const std::int64_t primeBound =
(playMode_ == PlayMode::Trigger && playEnd_ > 0 && playEnd_ < frameCount)
? playEnd_ : frameCount;
const std::int64_t primeCount =
loopWrap ? w : std::min<std::int64_t>(w, primeBound - start);
// Both channels walk identical SOURCE positions (the walk depends only on loop geometry,
// not on channel PCM values) — compute `p` once for channel 0, reuse for channel 1.
std::int64_t p = start;
for (int ch = 0; ch < (stereoSample ? 2 : 1); ++ch) {
const std::vector<AudioSample>& pcmCh = ch == 0 ? sample.frames : sample.framesR;
std::int64_t q = start;
for (std::int64_t i = 0; i < primeCount; ++i) {
if (loopWrap) {
while (q >= loop.end) q -= loopLen;
}
// q < frameCount holds by construction on the non-loop path (primeCount is
// bounded); the guard stays as a belt for the loop-wrap walk.
primeBuf_[static_cast<std::size_t>(i)] =
(q < frameCount) ? pcmCh[static_cast<std::size_t>(q)] : 0.0f;
++q;
}
(ch == 0 ? shiftL_ : shiftR_).prime(primeBuf_.data(), primeCount);
if (ch == 0) p = q; // capture the end position once from channel 0's walk
}
// Per-frame feed continues at `p` (== the feed bound when the prime exhausted the
// playable span — advanceFrame's own exhaustion test then holds from frame 0).
feedPos_ = p;
if (!loopWrap && primeCount < w) {
// Sub-window playable span: the source is ALREADY exhausted at prime time.
shiftL_.freezeTail();
if (stereoSample) shiftR_.freezeTail();
}
}
ratio_ = baseRatio_; // seeded; advanceFrame recomputes per frame under the active engine.
}
void Voice::retune(int note, int rootNote, double keyTrack) {
// Mono legato takeover: move the pitch, touch NOTHING else — the amplitude envelope keeps
// running (no re-attack), the read head keeps its position, the shifter keeps its ring
// (Preserve picks the new baseRatio_ up via next frame's setShiftRatio; Varispeed via the
// per-frame ratio_ recompute). Velocity gain deliberately stays the first note's — a legato
// phrase is one gesture, one strike (classic mono-synth behavior).
if (!active_) return;
note_ = note;
baseRatio_ = keyTrackedRatio(note, rootNote, keyTrack);
}
void Voice::release() {
if (!active_) return;
// TRIGGER ignores note-off entirely (S15): the one-shot plays through to its play length.
if (playMode_ == PlayMode::Trigger) return;
releasing_ = true;
env_.noteOff();
}
void Voice::hardStop() {
// CC 120 (All Sounds Off): immediate silence regardless of play mode. Stops Trigger one-shots
// that ignore release(), and short-circuits Gate release tails. RT-safe: no allocation.
active_ = false;
}
double Voice::tickAmplitude() {
double amp;
if (playMode_ == PlayMode::Gate) {
// AHDSR is wall-clock (one tick per output frame), independent of the read rate.
amp = env_.tick();
if (env_.finished()) amplitudeDone_ = true;
} else {
// Trigger fade shape anchored to the SOURCE offset (readPos - startFrame), so the fades
// land on the same source frames under either engine's read rate. The voice ALSO frees on
// readPos_ >= playEnd_ in advanceFrame; finished() here is the belt to that suspenders.
amp = trigEnv_.amplitudeAt(readPos_ - static_cast<double>(startFrame_));
if (trigEnv_.finished()) amplitudeDone_ = true;
}
return amp;
}
void Voice::seedDeclick(double newOutL, double newOutR) {
// First frame after a takeover restart: ARM the bounded blend. The weight starts at 1.0
// so this frame's output is `out*(1-1) + ref*1 == ref` — exact boundary identity whatever
// the new envelope's first value. Each subsequent frame adds `w*(ref outCurrent)` then
// decays w, so output is provably bounded by max(|ref|, |outCurrent|) — mid-ramp overshoot
// is impossible even if outCurrent rises while the weight is still significant.
// [Rev 1 stored the frozen difference (ref x₀); if outₙ rose while that residue was
// still large the sum could exceed full scale. The ±2.0 clamp there was the only guard
// and it silently broke the boundary identity when |x₀| > 1. The bounded blend removes
// both the overshoot hole and the need for a clamp on the stored value.]
// newOutL/R are used only to decide whether an active ramp exists (the seed is purely
// the weight 1.0; ref was clamped to ±1 at start()). The ±2 clamp on the difference is
// gone: the blend formula keeps every output within max(|ref|,|outₙ|) by construction.
(void)newOutL; (void)newOutR; // consumed only for the floor guard below
declickPending_ = false;
declickWeight_ = 1.0; // ONE weight for both channels (T1-09: the per-R copy was dead state)
// The reference is already clamped to ±1.0 at start() (lines in start(): the ±1 clamp
// on lastOutL_/R_ before storing into declickRefL_/R_). No secondary clamp needed here.
// Activate only when the ref itself is above the floor — if ref ≈ 0 there is nothing to blend.
declickActive_ = (declickRefL_ > kDeclickFloor || declickRefL_ < -kDeclickFloor ||
declickRefR_ > kDeclickFloor || declickRefR_ < -kDeclickFloor);
}
AudioSample Voice::advanceFrame(bool stereo, AudioSample& outR) {
// Shared read/advance for the mono and stereo paths. The read-head geometry (loop wrap,
// bracketing indices, interpolation partner) is computed ONCE and applied identically to
// every channel — only the PCM value read differs. The amplitude + pitch envelopes tick ONCE
// per frame and scale all channels equally (a voice is one envelope). The head advances by
// exactly one source-frame step per call, so mono and stereo consume the sample at one rate.
if (!active_ || sample_ == nullptr) {
if (stereo) outR = 0.0f;
return 0.0f;
}
const std::vector<AudioSample>& pcm = sample_->frames;
const std::int64_t frameCount = static_cast<std::int64_t>(pcm.size());
// Read the second channel only for a genuinely stereo sample; a mono sample plays
// dual-mono (channel 0 duplicated), so `pcmR` aliases channel 0 in that case.
const bool haveR = stereo && sample_->channelCount() == 2;
const std::vector<AudioSample>& pcmR = haveR ? sample_->framesR : pcm;
// Loop-aware sustain (GATE only — Trigger is a one-shot with no sustain loop, S15). If a
// valid, non-zero-length loop exists and the read head has advanced past the loop end, wrap
// it back into [start, end). A zero-length loop is treated as "no loop". Under Preserve the
// loop is over the SOURCE read (loop the source, shift the output — S15×S16 contract).
const SampleLoop& loop = sample_->loop;
const bool loopUsable = sustainLoopUsable();
if (loopUsable) {
const double loopLen = static_cast<double>(loop.end - loop.start);
while (readPos_ >= static_cast<double>(loop.end)) {
readPos_ -= loopLen; // wrap by exactly one loop length, preserving phase.
}
}
// TRIGGER end: the voice frees once the read head reaches playEnd (source-frame stop). The
// trigger envelope also finishes at the same frame count; either latches the voice idle.
const bool triggerRanOff =
playMode_ == PlayMode::Trigger && readPos_ >= static_cast<double>(playEnd_);
// Ran off the sample end with no usable loop -> voice is done. Peer path of the
// epilogue: an in-flight takeover declick RINGS OUT here instead of hard-cutting —
// dropping it would re-introduce a step on exactly the path the ramp exists for (a
// restart whose new play span ends within the ~4 ms ramp). The voice stays active only
// until the ramp floors; with no declick (the common case, and the entire opt-out
// baseline) this is byte-identical to the plain idle-out.
if (triggerRanOff || readPos_ >= static_cast<double>(frameCount)) {
if (declickPending_) seedDeclick(0.0, 0.0); // the new output here is silence
if (declickActive_) {
// Bounded blend at silence: outCurrent == 0, so the blend is w*(ref 0) == w*ref.
// The weight decays by kDeclickDecay each frame, floor-checked on the weight itself.
const double l = declickWeight_ * declickRefL_;
const double r = declickWeight_ * declickRefR_; // same weight for both channels
declickWeight_ *= kDeclickDecay;
if (declickWeight_ < kDeclickFloor && declickWeight_ > -kDeclickFloor) {
declickActive_ = false;
active_ = false;
}
lastOutL_ = l;
lastOutR_ = stereo ? r : l;
if (stereo) outR = static_cast<AudioSample>(r);
return static_cast<AudioSample>(l);
}
active_ = false;
if (stereo) outR = 0.0f;
return 0.0f;
}
// Envelopes tick once per output frame. Pitch envelope biases pitch under EITHER engine.
const double amp = tickAmplitude();
const double gain = amp * velocityGain_;
const double pitchEnvSemis = pitchEnv_.tick();
// The pitch-envelope bias factor 2^(semis/12). When the envelope is off (semis exactly 0)
// this is 1.0 and we skip the pow entirely — the Varispeed-off path stays a bare ratio read
// (no per-frame transcendental), byte-identical to pre-S16.
const double envFactor = (pitchEnvSemis == 0.0) ? 1.0 : std::pow(2.0, pitchEnvSemis / 12.0);
double outL, outRlocal = 0.0;
if (pitchEngine_ == PitchEngine::Preserve && shiftL_.configured()) {
// PRESERVE: feed the shifters the SOURCE stream at unity rate (duration held) and
// TRANSPOSE the output by 2^((note-root + pitchEnvSemis)/12). Pitch envelope adds to
// the shift amount, not the read rate — pitch bends, duration unchanged (S16
// contract). The feed runs one window AHEAD of readPos_ (the rings were primed with
// that window at start()), under the SAME sustain-loop wrap rule as the anchor, and
// reads integer source frames (readPos_ advances by exactly 1.0 under Preserve, so
// there is nothing to interpolate). Past the last real frame the shifter's writer is
// FROZEN (GA3 wind-down below) — it recycles the real tail it already holds.
if (loopUsable) {
const std::int64_t loopLen = loop.end - loop.start;
while (feedPos_ >= loop.end) feedPos_ -= loopLen;
}
// GA3 tail wind-down (supersedes the GA2 hold-last-sample clamp). feedPos_ runs one
// window AHEAD of readPos_; the last real source frame is playEnd_-1 for Trigger (the
// user's chosen stop) or frameCount-1 for Gate (the sample's own end). Once feedPos_
// reaches that bound the source is EXHAUSTED — GA2 fed the held last sample from here,
// a DC plateau the splice correlation cannot align on (the DAW tail chop: periodic
// troughs at the splice cadence, growing toward the note end as the plateau displaced
// real ring history). Instead FREEZE the shifter's writer: no padding ever enters the
// ring, and the splice machinery keeps recycling the frozen all-real tail, every jump
// still waveform-aligned — a continuous tone through the final window and the release,
// bounded by the voice's own end (readPos_ >= frameCount / playEnd_ frees it). The
// sustain-loop path never gets here: the wrap above keeps feedPos_ < loop.end forever.
const std::int64_t feedBound =
(playMode_ == PlayMode::Trigger && playEnd_ > 0 && playEnd_ < frameCount)
? playEnd_ : frameCount;
const bool exhausted = feedPos_ >= feedBound;
if (exhausted) shiftL_.freezeTail(); // idempotent; input below is ignored while frozen
const bool feedOk = (!exhausted && feedPos_ >= 0 && feedPos_ < frameCount);
const AudioSample feedL = feedOk ? pcm[static_cast<std::size_t>(feedPos_)] : 0.0f;
const double shift = baseRatio_ * envFactor;
shiftL_.setShiftRatio(shift);
const double shiftedL = static_cast<double>(shiftL_.process(feedL));
outL = shiftedL * gain;
if (stereo) {
if (haveR && shiftR_.configured()) {
// Genuine stereo (Q-W0 T1-01, linked lag): channel 1's shifter FOLLOWS channel
// 0's splice decisions via processLinked — one correlation search, one lag, one
// splice schedule for both channels (standard stereo SOLA). An independent
// per-channel search re-drew an inter-channel offset of up to +/-maxLag at
// every splice: stereo image wander at the splice cadence + mono-sum combing.
// Each shifter is still processed EXACTLY ONCE per output frame (never twice —
// that would advance its heads twice and corrupt the state). Gated on haveR so
// a MONO sample never touches shiftR_ — start() only primes it for genuinely
// stereo samples, and a stale un-primed ring must not leak a previous note.
if (exhausted) shiftR_.freezeTail();
const AudioSample feedR = feedOk ? pcmR[static_cast<std::size_t>(feedPos_)] : 0.0f;
shiftR_.setShiftRatio(shift);
outRlocal =
static_cast<double>(shiftR_.processLinked(feedR, shiftL_.lastSplice())) *
gain;
} else {
// Mono sample in stereo mode (dual-mono): shiftL_ already produced the shifted
// value from the mono feed; mirror it to R. Do NOT call shiftL_.process again
// this frame.
outRlocal = shiftedL * gain;
}
}
++feedPos_;
// Preserve advances the read head at the SOURCE rate (duration preserved).
ratio_ = 1.0;
} else {
// VARISPEED: pitch and duration coupled. The read rate carries the repitch; the pitch
// envelope multiplies the ratio for the read-rate bias (unchanged pre-S16 idiom when the
// envelope is off -> pitchEnvSemis == 0 -> factor 1.0 -> byte-identical).
//
// Linear interpolation between the two bracketing SOURCE frames at the read head. For
// the loop case, the second point wraps to loopStart so the seam is continuous.
const std::int64_t i0 = static_cast<std::int64_t>(readPos_);
const double frac = readPos_ - static_cast<double>(i0);
std::int64_t i1 = i0 + 1;
if (loopUsable && i1 >= loop.end) {
i1 = loop.start; // seamless wrap for the interpolation partner.
}
const bool i0ok = (i0 >= 0 && i0 < frameCount);
const bool i1ok = (i1 >= 0 && i1 < frameCount);
const double srcL = (i0ok ? static_cast<double>(pcm[i0]) : 0.0) +
((i1ok ? static_cast<double>(pcm[i1]) : 0.0) -
(i0ok ? static_cast<double>(pcm[i0]) : 0.0)) * frac;
outL = srcL * gain;
if (stereo) {
const double srcR = (i0ok ? static_cast<double>(pcmR[i0]) : 0.0) +
((i1ok ? static_cast<double>(pcmR[i1]) : 0.0) -
(i0ok ? static_cast<double>(pcmR[i0]) : 0.0)) * frac;
outRlocal = srcR * gain;
}
ratio_ = baseRatio_ * envFactor;
}
// Takeover declick (Phase S GA fix, rev 2, bounded-blend revision): on the FIRST frame
// after a takeover/steal restart, seed the blend weight at 1.0 so this frame's output is
// outₙ*(1w) + ref*w = out*(11) + ref*1 = ref (exact boundary identity).
// Each subsequent frame the blend add is `w*(ref outCurrent)` and then w decays by
// kDeclickDecay. The output is therefore bounded by max(|ref|, |outCurrent|) in every
// frame — mid-ramp overshoot from a rising outCurrent is structurally impossible.
// [Rev 1 added the frozen difference (ref x₀) ungated; if outₙ rose while the residue
// was still large the sum could exceed ±1 by up to ~+3.8 dB on an extreme retrig.]
// Inactive (the common case) costs one branch; the blend itself costs one extra subtract.
if (declickPending_) seedDeclick(outL, stereo ? outRlocal : outL);
if (declickActive_) {
const double addL = declickWeight_ * (declickRefL_ - outL);
const double addR = declickWeight_ * (declickRefR_ - (stereo ? outRlocal : outL));
outL += addL;
if (stereo) outRlocal += addR;
declickWeight_ *= kDeclickDecay; // one shared weight — both channels decay together
if (declickWeight_ < kDeclickFloor && declickWeight_ > -kDeclickFloor) {
declickActive_ = false;
}
}
if (stereo) outR = static_cast<AudioSample>(outRlocal);
// Track the value this voice actually contributed THIS frame (post-gain, incl. any running
// declick) — a future takeover restart seeds its declick from exactly this. In a mono
// render the R track mirrors L (dual-mono semantics, matching the stereo mirror of a mono
// sample), so a later stereo takeover still has a sane R seed.
lastOutL_ = outL;
lastOutR_ = stereo ? outRlocal : outL;
readPos_ += ratio_;
// A finished amplitude envelope frees the voice — unless a takeover declick still rings:
// the envelope contributes 0 from here on, so the remaining frames are the bare ramp
// fading out (bounded: the ramp floors within ~4 ms). Baseline (no declick) unchanged.
if (amplitudeDone_ && !declickActive_) {
active_ = false;
}
return static_cast<AudioSample>(outL);
}
AudioSample Voice::renderFrame() {
AudioSample discard = 0.0f;
return advanceFrame(/*stereo=*/false, discard);
}
void Voice::renderFrameStereo(AudioSample& l, AudioSample& r) {
r = 0.0f;
l = advanceFrame(/*stereo=*/true, r);
}
// ---------------------------------------------------------------------------
// VoiceEngine
// ---------------------------------------------------------------------------
VoiceEngine::VoiceEngine(std::size_t maxVoices, const Keymap& keymap,
std::size_t preserveVoiceCap,
std::int64_t preserveWindowFrames,
VoiceMode voiceMode, MonoTrigger monoTrigger,
bool takeoverDeclick)
// MONO always uses voices_[0] only (last-note priority, single voice); size to 1 so
// the "only voices_[0] is ever driven" invariant is structurally enforced — no latent
// RT-discipline risk if a future mono path touched voices_[1..]. maxVoices == 0 clamps
// to 1 (documented degenerate: at least one voice so a note-on is always serviceable).
: voices_(voiceMode == VoiceMode::Mono ? 1
: (maxVoices == 0 ? 1 : maxVoices)),
keymap_(keymap),
preserveVoiceCap_(preserveVoiceCap),
voiceMode_(voiceMode), monoTrigger_(monoTrigger),
takeoverDeclick_(takeoverDeclick) {
// Pre-size every voice's Preserve shifters HERE (construction is off the audio thread), so
// note-on never allocates. A 0 window leaves them pass-through (no ring). This is the one
// allocation point for the shifter rings across the engine's lifetime.
// MONO: voices_.size() == 1, so the loop below sizes exactly one voice regardless of
// maxVoices — the Poly path sizes the whole pool as before.
if (preserveWindowFrames > 1) {
for (std::size_t i = 0; i < voices_.size(); ++i) {
voices_[i].presizePreserveShifters(preserveWindowFrames);
}
}
}
std::size_t VoiceEngine::activePreserveVoices() const {
// Count only voices that are SOUNDING A NOTE (playable span still running), not voices
// that have finished their note but are still ringing out a declick tail. A ramp-only
// past-end voice must not consume a cap slot — that would cause a new Preserve note-on to
// be dropped (kNoVoice return at :797-800) during the narrow ~4 ms window the ramp lives.
std::size_t n = 0;
for (const Voice& v : voices_) {
if (v.soundingNote() && v.pitchEngine() == PitchEngine::Preserve) ++n;
}
return n;
}
std::size_t VoiceEngine::allocateVoice() {
// 1. A free (idle) voice, lowest index for determinism.
for (std::size_t i = 0; i < voices_.size(); ++i) {
if (!voices_[i].active()) return i;
}
// 2. All busy -> steal. Prefer the oldest voice already in release (a dying tail),
// else the oldest voice overall. "Oldest" = smallest startOrder.
std::size_t bestReleasing = kNoVoice;
std::uint64_t bestReleasingOrder = 0;
std::size_t bestOverall = kNoVoice;
std::uint64_t bestOverallOrder = 0;
for (std::size_t i = 0; i < voices_.size(); ++i) {
const std::uint64_t order = voices_[i].startOrder();
if (voices_[i].releasing()) {
if (bestReleasing == kNoVoice || order < bestReleasingOrder) {
bestReleasing = i;
bestReleasingOrder = order;
}
}
if (bestOverall == kNoVoice || order < bestOverallOrder) {
bestOverall = i;
bestOverallOrder = order;
}
}
return bestReleasing != kNoVoice ? bestReleasing : bestOverall;
}
void VoiceEngine::removeHeld(int note) {
for (std::size_t i = 0; i < heldCount_; ++i) {
if (heldStack_[i].note == static_cast<std::uint8_t>(note)) {
// Shift the notes above it down one slot (press order preserved).
for (std::size_t j = i + 1; j < heldCount_; ++j) heldStack_[j - 1] = heldStack_[j];
--heldCount_;
return;
}
}
}
std::size_t VoiceEngine::monoNoteOn(int note, int velocity) {
// Reject out-of-range notes BEFORE touching the held stack: HeldNote stores the note as a
// uint8, so an unguarded value (e.g. 256, or a negative) would alias mod 256 onto a real
// held note and corrupt the stack. Mirrored in monoNoteOff.
if (note < 0 || note > 127) return kNoVoice;
const ZoneResolution res = keymap_.resolve(note, velocity);
if (!res.matched) return kNoVoice; // out-of-zone: defined no-play, never joins the stack.
const KeyZone& zone = keymap_.zones[res.zoneIndex];
if (zone.sampleIndex >= keymap_.samples.size()) return kNoVoice;
const SampleData& sample = keymap_.samples[zone.sampleIndex];
// The note joins (or moves to) the top of the held stack. Velocity is clamped into the
// byte for storage only; the voice start below receives the caller's value untouched.
removeHeld(note);
if (heldCount_ < heldStack_.size()) {
const int vclamped = velocity < 0 ? 0 : (velocity > 127 ? 127 : velocity);
heldStack_[heldCount_++] = HeldNote{static_cast<std::uint8_t>(note),
static_cast<std::uint8_t>(vclamped)};
}
Voice& v = voices_[0];
// LEGATO takeover, keyed on the HELD-STACK DEPTH: after the push above, heldCount_ >= 2
// means another note was already physically held — the exact "takeover within a phrase"
// predicate. (The previous guard, `active && !releasing`, broke for TRIGGER zones:
// Voice::release() is a no-op in Trigger, so releasing_ never latches, and a one-shot
// still ringing after the last key-up was silently RETUNED in place instead of
// re-attacked. NOTE: a one-held-note same-note re-press (heldCount_ becomes 1 after the
// removeHeld/re-push above — so heldCount_ < 2) re-attacks rather than retuning, which is
// the correct fresh-phrase behavior for that edge case.) Same-sample requirement unchanged.
//
// soundingNote() (not just active()): a voice whose note has run to its play-end but is
// still ringing a declick tail must NOT be retuned — that would move the pitch of a dying
// ramp rather than restarting the new note, producing a silent note on the common
// "hammer same key while a past-end ring-out is active" path. The tail should keep fading;
// the new note-on restarts the voice normally (monoNoteOn falls through to start() below).
if (v.soundingNote() && heldCount_ >= 2 && monoTrigger_ == MonoTrigger::Legato &&
v.playingSample() == &sample) {
v.retune(note, zone.rootNote, zone.keyTrack);
return 0;
}
// RETRIGGER takeover / first note of a phrase / cross-sample legato: (re)start the voice.
// The declick opt-in rides every mono restart: start() self-gates it on the voice being
// ACTIVE, so a first-note fresh start never ramps — only a hard cut of a sounding tone.
v.start(note, velocity, sample, zone.rootNote, zone.keyTrack, zone.velocityCurve,
/*declickTakeover=*/takeoverDeclick_);
v.setStartOrder(nextStartOrder_++);
return 0;
}
void VoiceEngine::monoNoteOff(int note) {
// Same range guard as monoNoteOn: removeHeld compares against the uint8-cast note, so an
// unguarded out-of-range off (e.g. 256 -> 0 mod 256) would evict a legitimately held note.
if (note < 0 || note > 127) return;
removeHeld(note);
Voice& v = voices_[0];
// Releasing a note that is not the sounding one (a lower held note or an already-released
// note) changes nothing audible.
if (!v.active() || v.releasing() || v.note() != note) return;
if (heldCount_ == 0) {
v.release(); // last finger up: gate off (Trigger zones ignore this and play through).
return;
}
// FALLBACK: the most-recent still-held note takes the voice back (last-note priority).
const HeldNote fb = heldStack_[heldCount_ - 1];
const ZoneResolution res = keymap_.resolve(fb.note, fb.velocity);
if (!res.matched || keymap_.zones[res.zoneIndex].sampleIndex >= keymap_.samples.size()) {
v.release(); // defensive: only resolving notes are pushed, so this shouldn't happen.
return;
}
const KeyZone& zone = keymap_.zones[res.zoneIndex];
const SampleData& sample = keymap_.samples[zone.sampleIndex];
if (monoTrigger_ == MonoTrigger::Legato && v.playingSample() == &sample) {
v.retune(fb.note, zone.rootNote, zone.keyTrack); // glide back, no re-attack
return;
}
// Retrigger (or cross-sample) fallback: re-strike the fallen-back-to note at its own
// original velocity. Peer restart site of monoNoteOn's takeover — same declick opt-in
// (the fallback also hard-cuts the sounding tone).
v.start(fb.note, fb.velocity, sample, zone.rootNote, zone.keyTrack, zone.velocityCurve,
/*declickTakeover=*/takeoverDeclick_);
v.setStartOrder(nextStartOrder_++);
}
std::size_t VoiceEngine::noteOn(int note, int velocity) {
if (voiceMode_ == VoiceMode::Mono) return monoNoteOn(note, velocity);
const ZoneResolution res = keymap_.resolve(note, velocity);
if (!res.matched) return kNoVoice; // out-of-zone: defined no-play.
const KeyZone& zone = keymap_.zones[res.zoneIndex];
if (zone.sampleIndex >= keymap_.samples.size()) {
return kNoVoice; // zone points at a missing sample — refuse rather than UB.
}
const SampleData& sample = keymap_.samples[zone.sampleIndex];
// S16 Preserve voice cap: a Preserve note is materially heavier than Varispeed (a per-voice
// OLA shifter). When a cap is set and it is already reached, DROP a new Preserve note-on
// rather than glitch (a defined no-play, mirroring out-of-zone — no shifter is allocated).
// Varispeed notes are unaffected. A voice already sounding is never cut by this cap; only
// NEW Preserve onsets past the cap are refused (the spec's "cap kicks in rather than glitch").
if (preserveVoiceCap_ > 0 && sample.play.pitchEngine == PitchEngine::Preserve &&
activePreserveVoices() >= preserveVoiceCap_) {
return kNoVoice;
}
// The voice's Preserve shifters were pre-sized at engine construction (off-thread), so
// start() only reset()s + warm()s them — no allocation on this audio-thread path.
// The takeover declick rides the STEAL restart too (GA fix): start() self-gates on the
// voice being active, so a free-voice start never ramps — only an at-cap steal, which is
// the same hard cut of a sounding tone as the mono retrig takeover.
const std::size_t v = allocateVoice();
voices_[v].start(note, velocity, sample, zone.rootNote, zone.keyTrack, zone.velocityCurve,
/*declickTakeover=*/takeoverDeclick_);
voices_[v].setStartOrder(nextStartOrder_++);
return v;
}
void VoiceEngine::noteOff(int note) {
if (voiceMode_ == VoiceMode::Mono) { monoNoteOff(note); return; }
// Release the NEWEST active, non-releasing voice on this note (largest startOrder),
// so a re-triggered note releases its newest instance first and older tails ring.
std::size_t target = kNoVoice;
std::uint64_t bestOrder = 0;
for (std::size_t i = 0; i < voices_.size(); ++i) {
if (voices_[i].active() && !voices_[i].releasing() &&
voices_[i].note() == note) {
const std::uint64_t order = voices_[i].startOrder();
if (target == kNoVoice || order > bestOrder) {
target = i;
bestOrder = order;
}
}
}
if (target != kNoVoice) voices_[target].release();
}
void VoiceEngine::allNotesOff() {
// CC 123. Clear the mono held stack so no fallback can resurrect a phantom note (the
// stuck-note scenario: a lost note-off leaves an entry that monoNoteOff's fallback
// restarts and sustains forever with no key held), then gate off every active voice.
// Gate voices enter their release tail; Trigger one-shots ignore release by design and
// play through their bounded play length. RT-safe: no allocation, bounded by the pool size.
heldCount_ = 0;
for (Voice& v : voices_) {
if (v.active()) v.release();
}
}
void VoiceEngine::allSoundsOff() {
// CC 120. Hard-stop EVERY voice immediately (no release ramp — silences Trigger one-shots
// that allNotesOff() cannot stop) and clear the mono held stack. RT-safe: no allocation,
// bounded by the pool size.
heldCount_ = 0;
for (Voice& v : voices_) {
v.hardStop();
}
}
void VoiceEngine::render(AudioSample* out, std::size_t frameCount) {
// Real-time safe: no allocation, no resize — mix straight into the caller's buffer.
// The VST3 process callback hands us the host's output channel buffer here, so the
// audio thread never touches the heap (S4 real-time discipline).
if (out == nullptr || frameCount == 0) return;
for (Voice& voice : voices_) {
if (!voice.active()) continue;
for (std::size_t f = 0; f < frameCount; ++f) {
if (!voice.active()) break;
out[f] += voice.renderFrame();
}
}
}
void VoiceEngine::render(AudioSample* left, AudioSample* right, std::size_t frameCount) {
// Real-time safe stereo mix: no allocation, no resize. Sum each active voice's per-channel
// contribution into the caller's two buffers. Mirrors the mono loop exactly (same voice
// iteration, same mid-block idle short-circuit) so stereo and mono share one stealing/idle
// discipline; only the per-frame call differs (renderFrameStereo vs renderFrame).
if (left == nullptr || right == nullptr || frameCount == 0) return;
for (Voice& voice : voices_) {
if (!voice.active()) continue;
for (std::size_t f = 0; f < frameCount; ++f) {
if (!voice.active()) break;
AudioSample l = 0.0f, r = 0.0f;
voice.renderFrameStereo(l, r);
left[f] += l;
right[f] += r;
}
}
}
void VoiceEngine::render(std::vector<AudioSample>& out, std::size_t frameCount) {
// Off-thread / test path: grow the buffer (this allocates — never call under
// process), zero-fill the appended span, then delegate to the RT mix loop so both
// overloads share exactly one summation path.
const std::size_t base = out.size();
out.resize(base + frameCount, 0.0f);
render(out.data() + base, frameCount);
}
std::size_t VoiceEngine::activeVoiceCount() const {
std::size_t n = 0;
for (const Voice& v : voices_) {
if (v.active()) ++n;
}
return n;
}
} // namespace reasampler
+770
View File
@@ -0,0 +1,770 @@
#pragma once
// sampler_core — the HEART of the Phase S MIDI-playback instrument (D3), deliberately
// free of any VST3 *and* any REAPER type so it compiles and unit-tests OUTSIDE the DAW
// and outside any plugin host. It owns the pure sampler engine: polyphonic voice
// allocation with bounded stealing, an ADSR amplitude envelope, a key/velocity keymap
// with (note, velocity) -> zone resolution, and repitch/interpolation from a root note
// with loop-point-aware sustain.
//
// PURE MODULE (CLAUDE.md §load-bearing split): NO VST3 types, NO REAPER types, NO SWELL,
// NO vendor/ includes, no include from either SDK. Standard library only. The VST3 shell
// (src/vst/reasampler_processor.cpp) marshals MIDI events + audio buffers to and from
// this core; the core never sees a VST3 ProcessData or a REAPER MediaTrack. Enforced
// structurally: sampler_core_tests links neither SDK (see CMakeLists §2i).
//
// It shares the `AudioSample` float alias from peaks — the one house precedent for a
// pure module leaning on peaks for the audio-domain type (wav_trim does the same). The
// S2 seam fields (root note, loop points) enter as plain int / frame-index inputs; the
// core does no file I/O — it is handed decoded sample frames and produces audio frames.
#include <array>
#include <cstddef>
#include <cstdint>
#include <vector>
#include "core/audio/peaks.h" // AudioSample (float)
#include "core/instrument/engine/pitch_shift.h" // PitchShifter (S16 Preserve engine DSP core)
#include "core/instrument/engine/velocity_curve.h" // VelocityCurve (S-VIEW-9 velocity->amp transfer curve; eval at start)
namespace reasampler {
// Q-W1 interim: the engine deps live in their sub-namespace homes now; sampler_core
// re-namespaces in its own split wave (Q-W2v).
using audio::AudioSample;
using instrument::engine::PitchShifter;
using instrument::engine::VelocityCurve;
using instrument::engine::VelocityPoint;
// The instrument's per-instance output channel mode (S7, D-E). MONO keeps the pre-S7
// downmix path (one channel out); STEREO negotiates a 2-channel output bus and renders
// per-channel. A PERFORMANCE choice the instrument owns (component state), never written
// to the bank. Default Mono preserves current behavior. Lives in the pure core as a plain
// value so the shell (bus negotiation, state) and the engine share one spelling; the core
// itself never branches on it — the mode only picks which render overload the shell drives.
enum class ChannelMode { Mono, Stereo };
// The instrument's per-instance VOICE MODE (Phase S voice redesign). POLY is today's
// polyphonic engine (fixed pool + bounded stealing); MONO is a single voice with LAST-NOTE
// priority over a held-note stack (classic mono synth: a new note takes the voice over; the
// release of the top note falls back to the most-recent still-held note). A PERFORMANCE
// choice the instrument owns (component state), never a bank fact. Default Poly preserves
// current behavior.
enum class VoiceMode { Poly, Mono };
// How a MONO takeover treats the envelopes (Phase S — Daniel: explicitly toggleable).
// RETRIGGER restarts the amplitude (and pitch) envelope on every new mono note. LEGATO keeps
// the envelope running when a note is taken over while another is held — pitch moves without
// a re-attack (and the fallback on top-note release glides back the same way). Legato applies
// only to a SAME-SAMPLE takeover: crossing into a zone playing a different sample restarts
// the voice (one read head cannot glide between two PCM streams; a re-attack on a sample
// change is the deterministic, documented fallback). Meaningless in Poly. Default Retrigger.
enum class MonoTrigger { Retrigger, Legato };
// The user-parameterized polyphony bound (Phase S): a per-instance persisted voice count.
// One spelling shared by the engine, the component-state (de)serializer, and the editor's
// control so the range can never drift apart. Default 16 == the pre-Phase-S fixed pool.
inline constexpr int kMinVoiceCount = 1;
inline constexpr int kMaxVoiceCount = 32;
inline constexpr int kDefaultVoiceCount = 16;
// ---------------------------------------------------------------------------
// S15/S16 per-zone play PARAMETERS (plain data). Defined up here (before SampleData) because
// SampleData carries a ZonePlayParams by value — a voice reads it at start(). The matching
// per-frame EVALUATOR classes (AHDSR AdsrEnvelope, TriggerEnvelope, PitchEnvelope) live lower
// with the rest of the engine machinery; only the value structs need to precede SampleData.
// ---------------------------------------------------------------------------
// AHDSR amplitude envelope parameters (S15 grows the S3 ADSR with a HOLD stage between Attack
// and Decay). holdFrames == 0 is EXACTLY the pre-S15 ADSR (back-compat). See AdsrEnvelope below.
struct AdsrParams {
std::int64_t attackFrames = 0;
std::int64_t holdFrames = 0; // S15: hold at 1.0 between Attack and Decay; 0 = pre-S15 ADSR
std::int64_t decayFrames = 0;
double sustainLevel = 1.0; // 0..1
std::int64_t releaseFrames = 0;
};
// S15 play mode. GATE = classic held note (AHDSR + sustain loop + note-off release, today's
// behavior grown by the hold stage). TRIGGER = one-shot: note-off-immune, no sustain loop,
// plays a % of the sample length shaped by fade-in/out. Both honor the start point. Per-zone
// (D-B); DEFAULT Gate so an instrument with no S15 params plays exactly as before.
enum class PlayMode { Gate, Trigger };
// Trigger amplitude envelope parameters (S15). Playback covers the source-frame span
// [startFrame, playEnd), playEnd = startFrame + round(lengthFraction*(frames - startFrame)),
// lengthFraction in (0,1]. Amplitude ramps 0->1 over fadeInFrames at the head and 1->0 over
// fadeOutFrames anchored to playEnd; unity between. Fades clamp so fadeIn + fadeOut <= play
// length. The voice frees when the head reaches playEnd. Note-off is a no-op in Trigger.
struct TriggerParams {
double lengthFraction = 1.0; // (0,1] of the post-start span to play
std::int64_t fadeInFrames = 0; // 0->1 ramp at the head
std::int64_t fadeOutFrames = 0; // 1->0 ramp anchored to playEnd
};
// The fade curve for Trigger's ramps. EQUAL_POWER (constant-power sin/cos) is the default
// (click-free on one-shots, per spec); LINEAR is the build-time residual. An enum (not a bool)
// so a third curve can join without a signature change.
enum class FadeCurve { EqualPower, Linear };
// The DEFAULT fade curve (S15 spec: equal-power). One constant to flip if linear is wanted.
inline constexpr FadeCurve kDefaultFadeCurve = FadeCurve::EqualPower;
// The per-zone pitch engine. VARISPEED = today's path (readPos_ += ratio_): pitch and duration
// coupled (an octave up plays half as long). PRESERVE = duration-preserving: the read advances
// at the SOURCE rate while a PitchShifter transposes the output (an octave up keeps its length).
enum class PitchEngine { Varispeed, Preserve };
// The PRODUCT DEFAULT pitch engine (S16-F1 — Daniel's "I want duration-preserving repitching"
// directive). ONE constant to flip if Varispeed should be the default instead. This is the
// default a NEW or absent-in-the-blob zone gets — APPLIED AT THE STATE BOUNDARY (sample_map's
// deserialize / editor zone-creation), NOT the pure-core struct default. The pure-core
// ZonePlayParams.pitchEngine member defaults to VARISPEED so that "no params == the pre-S16
// engine" holds for the core's own regression tests (an octave up still halves duration in the
// bare engine); the Preserve product default is layered on above at (de)serialization.
inline constexpr PitchEngine kDefaultPitchEngine = PitchEngine::Preserve;
// The OLA window (frames) the Preserve PitchShifter uses, derived from a window in milliseconds
// at the voice's sample rate. ~50 ms is the WDL quality-0 window the spec cites; larger =
// smoother on big transpositions. Onset latency is ZERO: start() primes the ring with the first
// window of real source, so output frame 0 IS source frame 0 regardless of window size (GA2 fix).
// One knob, resolved at voice allocation.
inline constexpr double kPreserveWindowMs = 50.0;
// A per-voice AD pitch-modulation envelope (S16), OFF by default (enabled=false -> offset always
// 0 -> playback bit-identical to the un-modulated engine). At note-on the pitch offset rises to
// peakSemitones over attackFrames, then falls to 0 (base pitch) over decayFrames. A zero attack
// gives the pure "start high, drop to base" percussive drop. peakSemitones is signed (+/-).
struct PitchEnvParams {
bool enabled = false;
std::int64_t attackFrames = 0;
std::int64_t decayFrames = 0;
double peakSemitones = 0.0; // signed depth at the peak
};
// The bundle of S15/S16 per-zone play parameters a voice reads at start(). Lives on SampleData
// (each zone owns one SampleData in the zoned keymap). DEFAULTS are EXACTLY the pre-S15/S16
// engine: Gate mode, AHDSR with hold 0 (= the S3 ADSR), VARISPEED pitch engine, pitch envelope
// disabled — so a bare-core voice with default play is byte-identical to the pre-S15 build (the
// core regression tests rely on this). The PRODUCT default of Preserve (S16-F1) is applied one
// layer up at (de)serialization for new/absent zones — see kDefaultPitchEngine.
struct ZonePlayParams {
PlayMode playMode = PlayMode::Gate;
AdsrParams adsr; // Gate: the AHDSR envelope
TriggerParams trigger; // Trigger: %-length + fades
PitchEngine pitchEngine = PitchEngine::Varispeed;
PitchEnvParams pitchEnv; // AD pitch modulation, off by default
};
// ---------------------------------------------------------------------------
// Sample data the core plays. Plain, decoded PCM + the S2 bank intrinsics that
// govern playback. The shell decodes the on-disk WAV and fills this; the core
// never touches a file.
// ---------------------------------------------------------------------------
// A loop over [start, end) frames, half-open. A zero-length loop (start == end)
// is the "no sustain loop" marker — a held note past the sample end goes silent
// rather than looping a zero span. absent-loop is modeled by leaving hasLoop false.
struct SampleLoop {
bool hasLoop = false;
std::int64_t start = 0; // first looped frame (inclusive)
std::int64_t end = 0; // one-past-last looped frame (exclusive); start <= end
};
// One decoded audio sample the engine can voice. DEINTERLEAVED, per-channel: `frames` is
// channel 0 (always present) and `framesR` is channel 1 (present only for a STEREO sample).
// A sample is stereo iff `framesR` is non-empty AND the same length as `frames`; otherwise
// it is mono (the degenerate, byte-identical Tier 0-1 case — `framesR` stays empty). Both
// channels share `readPos_`, `rootNote`, and `loop`, so repitch/loop are per-frame identical
// across channels; only the sampled value differs. `rootNote` is the MIDI note the file was
// recorded at (S2 intrinsic) — the pitch that plays back at unity ratio.
struct SampleData {
std::vector<AudioSample> frames; // channel 0 PCM (mono, or L of a stereo sample)
std::vector<AudioSample> framesR; // channel 1 PCM (R); EMPTY for a mono sample
int sampleRate = 0; // frames per second (for reference; ratio is
// note-relative, so rate cancels for repitch).
// 0 is explicitly invalid — every consumer must
// receive a real rate before use.
int rootNote = 60; // MIDI note recorded at (plays at unity here)
SampleLoop loop; // sustain loop, if any
// Initial read position (frame offset) a voice starts playback at — frame 0 by
// default, so an unset start point is exactly the pre-S11 behavior. S11 makes this
// an instrument-side per-zone override (the "start point" marker); S15 builds on it
// (both play modes carry a modifiable start). Clamped into [0, frames) at note-on:
// a start >= the sample length is a no-op (voice starts at 0), never out of bounds.
std::int64_t startFrame = 0;
// S15/S16 per-zone play parameters (play mode, AHDSR/Trigger envelope, pitch engine, pitch
// envelope). Defaults reproduce the pre-S15 engine EXCEPT the pitch engine default is
// Preserve (S16-F1). A voice reads this at start(). Struct defined above SampleData.
ZonePlayParams play;
// 2 iff a matching-length second channel exists; else 1. A framesR of a different
// length than frames is treated as absent (mono) — a malformed pair never half-plays.
int channelCount() const {
return (!framesR.empty() && framesR.size() == frames.size()) ? 2 : 1;
}
};
// ---------------------------------------------------------------------------
// Keymap — the performance map (instrument-owned, D-B). A note+velocity resolves
// to at most one zone; a zone names which SampleData to play and the root note to
// repitch from. Tier-0 degenerate case: a single zone spanning [0,127] with the
// sample's own root. Tier-1: several zones, each a key range with its own root.
//
// TIER-2 EXTENSION (velocity layers / round-robin) — designed for, not built:
// resolution returns a zone; a zone today owns one sampleIndex. Tier 2 makes a zone
// own a *list* of (velocity-range, sampleIndex) layers (and round-robin sets), and
// resolve() gains the velocity dimension it already receives but currently ignores
// for selection. The (note, velocity) signature and the "resolve to a zone, then a
// sample within it" shape are already in place — Tier 2 fills in the second step
// without changing callers or the voice engine. See the report note.
// ---------------------------------------------------------------------------
// A key range [lowNote, highNote] (inclusive both ends) mapping to one sample, with
// the root note to repitch from (defaults to the sample's own root, overridable in
// the performance map per S5). velocityLow/High reserved for Tier-2 layers; today a
// zone accepts the full 1..127 velocity range (0 is note-off by MIDI convention).
struct KeyZone {
int lowNote = 0;
int highNote = 127;
int rootNote = 60; // repitch reference for this zone
// S-VIEW-6 key-tracking scalar: how far keyboard pitch tracks the root. 1.0 (100%) is
// standard 12-tone-ET (default; bit-identical to pre-S-VIEW-6); 0.0 = no tracking (every
// key plays root pitch); 2.0 = double-rate tracking. Scales the (note-root) semitone offset
// in the repitch math (keyTrackedRatio); rides BOTH engines via the voice's baseRatio_.
double keyTrack = 1.0;
// S-VIEW-9 velocity->amp transfer curve: maps the note-on velocity (0..127) to the voice's amp
// gain, replacing the fixed linear velocity/127. A per-zone performance characteristic (mirror
// of keyTrack), carried from PerformanceZone by resolvePerformance and eval'd ONCE in
// Voice::start (never per frame). DEFAULT flat y=1 (R10-F1 Option A) — every velocity plays at
// unity, a deliberate behavior change from the pre-r10 linear map.
VelocityCurve velocityCurve = VelocityCurve::flat();
std::size_t sampleIndex = 0; // index into Keymap::samples
};
// Result of resolving a (note, velocity). `matched == false` means the note falls in
// no zone (out-of-zone) — a defined no-play result, NOT an error and NOT voice 0.
struct ZoneResolution {
bool matched = false;
std::size_t zoneIndex = 0; // valid only when matched
};
// The keymap: the decoded samples plus the zones that map keys onto them. Owns
// resolution. Pure: no host types. Zones are tested first-match in order, so an
// earlier zone wins an overlap (deterministic, documented).
struct Keymap {
std::vector<SampleData> samples;
std::vector<KeyZone> zones;
// Resolves (note, velocity) to a zone. First zone (in order) whose [low,high]
// contains `note` wins. velocity is accepted now (Tier-2 seam) but does not
// affect zone choice at Tier 0-1. Returns {matched=false} when no zone contains
// the note.
ZoneResolution resolve(int note, int velocity) const;
// Convenience: build the Tier-0 degenerate keymap — one sample mapped
// chromatically across the whole keyboard from its own root note.
static Keymap singleSampleChromatic(SampleData sample);
};
// The chromatic pitch ratio to play `note` given a sample recorded at `rootNote`:
// 2^((note - rootNote) / 12). note == rootNote -> 1.0 (unity). One octave up -> 2.0,
// one octave down -> 0.5. Pure equal-temperament; no reference-frequency needed.
double pitchRatio(int note, int rootNote);
// The key-tracked pitch ratio (S-VIEW-6): 2^(((note - rootNote) * keyTrack) / 12). The
// keyTrack scalar scales the semitone offset before the ET conversion, so it governs how
// far playback pitch tracks the keyboard around the root:
// keyTrack == 1.0 -> standard 12-tone-ET (BIT-IDENTICAL to pitchRatio(note, rootNote) —
// (note-root)*1.0 is exact in IEEE-754, feeding the same std::pow call).
// keyTrack == 0.0 -> no tracking: every key plays the root pitch (ratio 1.0 for all notes).
// keyTrack == 2.0 -> double-rate tracking: each key is twice as far from the root in pitch.
// At the root note the offset is 0 regardless of keyTrack, so the root always plays at unity.
// Pure; both repitch engines (Varispeed read-rate, Preserve shift-amount) derive from it via
// the voice's baseRatio_.
double keyTrackedRatio(int note, int rootNote, double keyTrack);
// ---------------------------------------------------------------------------
// AHDSR amplitude envelope (S15 grows the S3 ADSR with a HOLD stage). Sample-based
// (times in frames), linear segments. A gate: noteOn() enters Attack; noteOff() enters
// Release from wherever it is. Asserted against a known signal in the tests (mirror of peaks).
//
// Segment math (all linear ramps):
// Attack: 0 -> 1 over attackFrames
// Hold: hold 1 over holdFrames (S15: NEW stage between A and D)
// Decay: 1 -> sustainLevel over decayFrames
// Sustain: hold sustainLevel until noteOff
// Release: currentLevel -> 0 over releaseFrames
// A zero-length attack jumps straight to 1 on the first frame; HOLDFRAMES == 0 skips Hold
// entirely, which is EXACTLY the pre-S15 ADSR (back-compat — existing Gate play is unchanged);
// zero decay jumps to sustain; a noteOff during attack/hold/decay (release-before-sustain)
// releases from the current partial level, not from sustainLevel. AdsrParams is defined above
// (with the other per-zone value structs); this section holds only the per-frame evaluator.
// ---------------------------------------------------------------------------
class AdsrEnvelope {
public:
enum class Stage { Idle, Attack, Hold, Decay, Sustain, Release, Finished };
void configure(const AdsrParams& params) { params_ = params; }
// Gate on: (re)start from Attack.
void noteOn();
// Gate off: enter Release from the current level.
void noteOff();
// Advances one frame and returns the amplitude for THIS frame (before advancing).
// Once Release completes the envelope latches Finished and returns 0.0 forever
// (until the next noteOn). A single, monotonic per-frame step — the caller pulls
// one value per output frame.
double tick();
Stage stage() const { return stage_; }
bool finished() const { return stage_ == Stage::Finished; }
double level() const { return level_; }
private:
AdsrParams params_;
Stage stage_ = Stage::Idle;
double level_ = 0.0;
std::int64_t framesInStage_ = 0;
double releaseFrom_ = 0.0; // level at the moment noteOff() was called
};
// ---------------------------------------------------------------------------
// S15 Trigger amplitude envelope (per-frame evaluator). The PlayMode / TriggerParams /
// FadeCurve value structs are defined above with the other per-zone params.
// ---------------------------------------------------------------------------
// Trigger amplitude envelope: a stateless-shape amplitude function over the play span, evaluated
// at a SOURCE-frame offset into the span. Anchoring the fades to SOURCE frames (not output
// frames) is what makes S15 compose with S16: under Preserve the read advances at source rate so
// output and source frames coincide, but under Varispeed a transposed voice consumes source
// faster — driving the fades off the read position keeps the fade-in/out anchored to the SAME
// source frames regardless of engine (the play-length end is a source-frame fact, S15×S16). The
// voice reports the read offset; this maps it to amplitude. Distinct from AHDSR — time-boxed by
// the play length and note-off-immune. Reports finished() once the offset reaches the play length.
class TriggerEnvelope {
public:
// Configure from the play span + fades. `playLengthFrames` is (playEnd - startFrame): the
// SOURCE-frame length of the play span. Fades are clamped so fadeIn + fadeOut <= playLength
// (fadeOut anchored to the end). A zero/negative play length finishes immediately.
void configure(std::int64_t playLengthFrames, std::int64_t fadeInFrames,
std::int64_t fadeOutFrames, FadeCurve curve = kDefaultFadeCurve);
// Amplitude in [0,1] at `sourceOffset` = (readPos - startFrame) source frames into the play
// span. Latches finished() once the offset reaches the play length (>= playLength). Pure over
// the offset (no internal advance) so it composes with either pitch engine's read rate.
double amplitudeAt(double sourceOffset);
bool finished() const { return finished_; }
private:
std::int64_t playLength_ = 0;
std::int64_t fadeIn_ = 0;
std::int64_t fadeOut_ = 0;
FadeCurve curve_ = kDefaultFadeCurve;
bool finished_ = false;
};
// ---------------------------------------------------------------------------
// S16 pitch envelope (per-frame evaluator). The PitchEngine / PitchEnvParams value structs
// and the kDefaultPitchEngine / kPreserveWindowMs constants are defined above.
// ---------------------------------------------------------------------------
// Per-frame AD pitch-envelope evaluator. tick() returns the CURRENT pitch offset in semitones
// (0 when disabled or past attack+decay), advancing one frame. The voice converts the semitone
// offset to a ratio multiply (Varispeed) or a shift-amount add (Preserve). Pure, unit-tested
// for offset at t=0, peak at t=attack, and 0 at t=attack+decay.
class PitchEnvelope {
public:
void configure(const PitchEnvParams& params) { params_ = params; pos_ = 0; }
void noteOn() { pos_ = 0; }
// Advance one frame, return this frame's pitch offset in semitones.
double tick();
private:
PitchEnvParams params_;
std::int64_t pos_ = 0;
};
// Takeover declick (Phase S GA fix, rev 2 — audible click when a sounding voice is
// restarted). A takeover restart HARD-CUTS the sounding tone: the read head and envelope
// restart in one frame, a step discontinuity that clicks. This is the same physics on EVERY
// restart-of-a-sounding-voice path — the MONO Retrigger takeover/fallback, the mono
// cross-sample legato restart, and the POLY at-cap voice steal (the editor's preview is a
// plain engine noteOn since the PreviewCard retirement, so a preview re-fire at cap is just
// an at-cap steal). When the caller opts in (start()'s declickTakeover; the engine passes
// it on all of those restart paths when constructed with takeoverDeclick),
// the restart smooths the ACTUAL output discontinuity: start() records the last rendered
// output as the pre-cut reference, and the FIRST frame rendered after the restart seeds a
// compensation equal to (reference that frame's raw new output). The compensation is
// summed into the output UNGATED and decays by kDeclickDecay per frame, so the boundary
// frame reproduces the old level EXACTLY — zero step whatever the new envelope's first
// value (Gate attack, zero attack, or Trigger's no-fade-in instant-unity onset) and
// whatever value the new sample starts on — and the residue fades in ~2-4 ms to the -80 dB
// floor across 44.1-96 kHz (a per-FRAME DSP micro-ramp, not a stored wall-clock quantity).
// [Rev 1 decayed the OLD output gated by (1 newAmp): any restart whose new amplitude was
// instantly ~1 — a Trigger zone with no fade-in, a zero-attack Gate — got ZERO compensation
// and kept the full click. The difference seed has no such hole and needs no gate: when old
// and new levels already match, the seed is ~0 and nothing is added, so the +6 dB sum the
// gate defended against is structurally impossible.] OFF by default so the bare core stays
// byte-identical to the pre-fix engine (the regression baseline); the processor shell opts
// in for the engine, mirroring the kDefaultPitchEngine layering.
inline constexpr double kDeclickDecay = 0.95; // per-frame decay of the compensation
inline constexpr double kDeclickFloor = 1e-4; // below this the ramp is done (~ -80 dB)
// ---------------------------------------------------------------------------
// A single voice: one active note playing one repitched, enveloped sample. Reads
// the sample by fractional frame position with linear interpolation, advancing by
// the pitch ratio; loops the sustain region for held notes past the loop end.
// ---------------------------------------------------------------------------
class Voice {
public:
// Starts this voice on `note` at `velocity`, playing `sample` (a stable reference
// the caller must keep alive for the voice's lifetime — the Keymap owns it), repitched
// from `rootNote`. All five AHDSR fields (A/H/D/S/R) are read directly from
// sample.play.adsr — the per-zone values (in FRAMES) resolved from the stored seconds by
// buildTier0Keymap / buildZonedKeymap against the live sample rate. The S15 play MODE +
// Trigger params and the S16 pitch ENGINE + pitch envelope are read from `sample.play`.
// The Preserve shifters MUST already be pre-sized (presizePreserveShifters, off-thread) —
// start() only reset()s + warm()s them (RT-safe, no allocation) since it runs on the audio
// thread inside process(). The warm silence pass settles the OLA taps before the first
// output frame (no cold-start click). Byte-identical to the pre-S15 engine when sample.play
// is default (Gate + Varispeed + no pitch env).
// `keyTrack` (S-VIEW-6) scales the (note-root) semitone offset feeding the repitch ratio;
// 1.0 (the default) is standard 12-tone-ET, bit-identical to the pre-S-VIEW-6 baseRatio_.
// `velocityCurve` (S-VIEW-9) maps the note-on velocity to the voice's amp gain, evaluated ONCE
// here (off the per-frame path); defaults to flat y=1 (R10-F1) — every velocity plays at unity.
// `declickTakeover` (Phase S GA fix): when TRUE and this voice is currently ACTIVE (a
// takeover/steal restart, not a fresh start), smooth the restart's output discontinuity —
// the pre-cut output is recorded here and the difference-seeded compensation is armed on
// the first frame rendered after the restart (see the takeover-declick block above
// kDeclickDecay). A fresh start never declicks.
void start(int note, int velocity, const SampleData& sample, int rootNote,
double keyTrack = 1.0,
const VelocityCurve& velocityCurve = VelocityCurve::flat(),
bool declickTakeover = false);
// MONO LEGATO takeover (Phase S): re-pitch this ACTIVE voice to `note` without touching the
// amplitude envelope, the read position, or the shifter state — pitch moves, no re-attack.
// Both engines pick the new baseRatio_ up on the next frame (Varispeed via the read rate,
// Preserve via the per-frame setShiftRatio). No-op on an idle voice. The caller guarantees
// the voice is playing the SAME SampleData the (note-resolved) zone names — a cross-sample
// takeover must restart the voice instead (see MonoTrigger).
void retune(int note, int rootNote, double keyTrack = 1.0);
// Gate off — begins the amplitude release. In GATE mode this enters the AHDSR release; in
// TRIGGER mode it is a NO-OP (Trigger ignores note-off and plays through to its play length).
void release();
// HARD STOP — CC 120 (All Sounds Off) semantics. Immediately silences this voice regardless
// of play mode: sets active_ = false with no release ramp. Stops a ringing Trigger one-shot
// instantly (which release() cannot do). RT-safe: no allocation, no lock.
void hardStop();
// True while this voice is producing (or about to produce) sound (including any
// declick ring-out tail past the note's playable span).
bool active() const { return active_; }
// True while this voice is sounding a PLAYABLE NOTE — active AND the amplitude
// envelope has not yet finished. A voice whose note has run to its end but is still
// ringing out a declick tail is active() but NOT soundingNote(). Use this to
// distinguish "note is alive" (active) from "note occupies a voice slot" (soundingNote)
// for the Preserve-cap count and the mono-Legato takeover predicate — both must ignore
// a ramp-only past-end voice or a new note-on can be dropped / silently muted.
bool soundingNote() const { return active_ && !amplitudeDone_; }
// The note this voice was started on (for note-off routing). Meaningless if idle.
int note() const { return note_; }
// Monotonic age counter — higher = started earlier relative to others. The voice
// engine uses this for its stealing policy (oldest first). Set by the engine.
std::uint64_t startOrder() const { return startOrder_; }
void setStartOrder(std::uint64_t order) { startOrder_ = order; }
bool releasing() const { return releasing_; }
// The S16 pitch engine this voice is running (for the engine's Preserve-voice tally). Only
// meaningful while active(). NOTE: the FA1 unity-shift demotion to Varispeed is GONE —
// it was scoped to the retired PreviewCard, and since GA2 the primed shifter speaks on
// frame 0 at every ratio, so a Preserve voice keeps its shifter at every note (one code
// path, uniform onset across the keyboard).
PitchEngine pitchEngine() const { return pitchEngine_; }
// The SampleData this voice is playing (nullptr when never started). The engine's mono
// legato path compares it against the new note's resolved sample — a same-sample takeover
// retunes; a cross-sample one restarts. Identity only; callers never mutate through it.
const SampleData* playingSample() const { return sample_; }
// Pre-SIZE this voice's Preserve pitch shifters (both channels) to `windowFrames`, OFF the
// audio thread (this allocates; also sizes the prime scratch buffer). The engine calls it
// once at construction so start() — which runs on the audio thread inside process() — never
// allocates: start() only prime()s the already-sized rings with the first window of source
// (a bounded copy). `windowFrames` <= 1 leaves the shifters as pass-through (Varispeed
// instruments pay no ring cost). Idempotent: a re-presize to the same window is a cheap
// no-op in the underlying vector.
void presizePreserveShifters(std::int64_t windowFrames);
// Renders one frame's contribution, advancing the read head and envelope by one
// output frame. Returns 0.0 (and goes idle) once the envelope finishes or the
// sample runs out with no loop. The value is already velocity- and
// envelope-scaled — the engine sums voices directly. This is the MONO path (channel
// 0 only) — byte-identical to the pre-S7 engine, so mono play is unchanged.
AudioSample renderFrame();
// STEREO render: writes THIS frame's per-channel contribution into `l`/`r` and advances
// the read head + envelope by exactly one frame (the same single advance the mono path
// performs — the envelope ticks ONCE per frame, shared across both channels). For a mono
// sample (channelCount()==1) both `l` and `r` receive the same value (dual-mono / centered).
// Both outputs are already velocity- and envelope-scaled. Goes idle on the same conditions
// as the mono path (envelope finished / sample exhausted with no loop) writing 0 to both.
void renderFrameStereo(AudioSample& l, AudioSample& r);
private:
// Shared read/advance for both render paths: computes the interpolated per-channel
// value(s) at the current read head, ticks the amplitude + pitch envelopes once, applies
// the pitch engine (Varispeed read-rate bias OR Preserve shift), advances the head, and
// latches idle on exhaustion. `stereo` selects whether the second channel is read (and
// returned in `outR`); when false `outR` is left untouched. Returns the channel-0 value.
AudioSample advanceFrame(bool stereo, AudioSample& outR);
// This frame's amplitude in [0,1] from the active envelope. GATE: the AHDSR ticks once per
// output frame (independent of the read rate — envelope time is wall-clock). TRIGGER: the
// fade shape is evaluated at the SOURCE offset (readPos - startFrame) so the fades anchor to
// source frames and compose with either pitch engine. Sets amplitudeDone_ when the envelope
// finishes (Gate: release complete; Trigger: play length reached) so advanceFrame frees the voice.
double tickAmplitude();
// True when the sustain loop applies to this voice: GATE mode with a valid, non-empty loop
// inside the sample (S15 — Trigger one-shots never loop). The single source of truth for
// the wrap rule shared by the output anchor (readPos_), the Preserve feed (feedPos_), and
// the start()-time ring prime.
bool sustainLoopUsable() const;
bool active_ = false;
bool releasing_ = false;
int note_ = 0;
double velocityGain_ = 1.0;
double baseRatio_ = 1.0; // 2^((note-root)/12): the un-modulated repitch ratio
double ratio_ = 1.0; // fractional SOURCE frames advanced per output frame (this frame)
double readPos_ = 0.0; // fractional frame index into the sample
const SampleData* sample_ = nullptr;
// S15 play mode + amplitude envelopes. Gate uses env_ (AHDSR); Trigger uses trigEnv_. Only
// one is active per voice (selected by playMode_ at start). playEnd_ is Trigger's source-frame
// stop (the voice frees when readPos_ >= playEnd_, mirroring the run-off-end idle).
PlayMode playMode_ = PlayMode::Gate;
AdsrEnvelope env_;
TriggerEnvelope trigEnv_;
std::int64_t startFrame_ = 0; // clamped initial read frame; Trigger fade offset origin
std::int64_t playEnd_ = 0; // Trigger: source-frame end; Gate: unused
bool amplitudeDone_ = false; // set when the active amplitude envelope finished
// S16 pitch engine + pitch envelope. pitchEngine_ selects Varispeed (ratio bias) vs Preserve
// (source-rate read + shifter). shiftL_/shiftR_ transpose the Preserve output per channel
// (one read head, per-channel shift — S7 compose). pitchEnv_ rides EITHER engine.
//
// GA2 onset fix: the shifter rings are PRIMED at start() with the first window of the
// actual upcoming source (loop-unrolled, silence past the end) — output frame 0 is source
// frame `start`, no ring-fill silence, and splices always land in real history. feedPos_
// is the integer SOURCE frame the shifters are fed next; it runs exactly one window AHEAD
// of readPos_ (the wall-clock output anchor) under the same sustain-loop wrap rule.
// GA3 tail wind-down: once feedPos_ passes the last real frame (Gate: sample end;
// Trigger: playEnd_) the shifters' writers are FROZEN — no padding enters the rings and
// the splice machinery recycles the frozen real tail through the note end (see
// advanceFrame). primeBuf_ is the presized scratch the prime stream is assembled into
// (never touched outside start()).
PitchEngine pitchEngine_ = PitchEngine::Varispeed;
PitchEnvelope pitchEnv_;
PitchShifter shiftL_;
PitchShifter shiftR_;
std::int64_t feedPos_ = 0;
std::vector<AudioSample> primeBuf_;
// Seeds the takeover compensation on the FIRST frame after a restart: the ramp is the
// ACTUAL discontinuity — (pre-cut reference the new voice's raw output this frame) —
// applied ungated so the boundary frame reproduces the old level exactly. See the
// takeover-declick block above kDeclickDecay.
void seedDeclick(double newOutL, double newOutR);
// Takeover declick state (see kDeclickDecay above). lastOut{L,R}_ track the voice's most
// recent rendered output (post-gain, incl. any running declick). A takeover/steal start()
// records them as declickRef{L,R}_ (the clamped pre-cut reference) and sets declickPending_;
// the first frame rendered after the restart calls seedDeclick to arm the BOUNDED BLEND:
// outₙ = outₙ*(1w) + ref*w where w = declickWeight_ (ONE weight, deliberately shared
// by both channels so L/R can never diverge — Q-W0 T1-09 removed the dead per-R copy)
// starts at 1.0 and decays by
// kDeclickDecay each frame. This is algebraically `outₙ + w*(ref outₙ)`, so the
// boundary frame (w=1) is exactly `ref` and every subsequent output is bounded by
// max(|ref|, |outₙ|) — mid-ramp overshoot is impossible regardless of outₙ rising.
// [Rev 1 stored the frozen difference (ref x₀); when outₙ rose while that residue
// was still large the sum could exceed full scale by up to ~+3.8 dB.]
// lastOut is NOT zeroed by start() — a second same-block takeover (no frame rendered
// between) must record the same pre-cut reference, not a phantom 0.
// The whole declick state is cleared on a fresh (non-takeover) start.
bool declickPending_ = false;
bool declickActive_ = false;
double declickRefL_ = 0.0; // clamped pre-cut reference (bounded blend target)
double declickRefR_ = 0.0;
double declickWeight_ = 0.0; // blend weight w; 1.0 on seed, decays by kDeclickDecay/frame
double lastOutL_ = 0.0;
double lastOutR_ = 0.0;
std::uint64_t startOrder_ = 0;
};
// ---------------------------------------------------------------------------
// The polyphonic voice engine: a fixed pool of voices, note-on allocation with
// bounded voice stealing, note-off routing, and block rendering (sum of voices).
//
// VOICE-STEALING POLICY (deterministic, documented): when all voices are busy and a
// new note-on arrives, steal in this priority order:
// 1. the oldest voice already in RELEASE (finishing anyway — cheapest to cut),
// 2. else the oldest voice overall (longest-held note gives way to the new one).
// "Oldest" = smallest startOrder (assigned monotonically at note-on). This is the
// standard hardware-sampler policy: prefer to sacrifice a dying tail, and failing
// that, the note that has already had the most time.
// ---------------------------------------------------------------------------
class VoiceEngine {
public:
// Builds an engine with `maxVoices` voices (the polyphony bound) playing from
// `keymap`. The keymap must outlive the engine (the engine holds a reference — it
// reads zones and sample data through it, never copies PCM). Every AHDSR field (A/H/D/S/R)
// + play mode + pitch engine rides on each zone's SampleData::play (in FRAMES, resolved
// from the stored seconds at keymap build); the engine holds no instrument-wide ADSR.
// `preserveVoiceCap` (S16) bounds how many Preserve-engine voices may sound at once (the
// shifter is materially heavier than Varispeed) — a Preserve note-on beyond the cap is
// dropped rather than glitching; 0 means "no separate Preserve cap" (bounded only by
// maxVoices). `preserveWindowFrames` is the OLA window (in OUTPUT frames) every voice's
// Preserve pitch shifters are PRE-SIZED to at construction (OFF the audio thread), so
// note-on (which runs in process()) never allocates; 0 leaves them pass-through (a
// Varispeed-only instrument pays no ring cost). The processor derives it from the host
// sample rate (kPreserveWindowMs). Defaulted so existing callers (and the pure-core tests)
// are unaffected.
//
// `voiceMode` (Phase S): POLY is the pool-with-stealing engine above; MONO drives a single
// voice (voices_[0]) with last-note priority over the held-note stack, per `monoTrigger`
// (Retrigger restarts the envelopes on every takeover/fallback; Legato retunes a same-sample
// takeover without a re-attack). Both default to today's behavior (Poly / Retrigger). The
// engine's config is immutable — a mode/count change rebuilds the engine off-thread through
// the processor's drain-slot reload, so ringing tails survive the swap.
//
// `takeoverDeclick` (Phase S GA fix): when TRUE, every RESTART of a SOUNDING voice —
// the MONO Retrigger takeover, the retrigger fallback on note-off, the cross-sample
// legato restart, and the POLY at-cap voice STEAL — seeds the per-voice declick ramp
// (see kDeclickDecay) so the hard cut of the old tone does not click. start() self-gates
// on the voice being active, so a fresh start (free voice) never ramps. Default FALSE
// keeps the bare core byte-identical to the pre-fix engine (regression baseline); the
// processor shell opts in — the same layering as the kDefaultPitchEngine product default.
VoiceEngine(std::size_t maxVoices, const Keymap& keymap,
std::size_t preserveVoiceCap = 0, std::int64_t preserveWindowFrames = 0,
VoiceMode voiceMode = VoiceMode::Poly,
MonoTrigger monoTrigger = MonoTrigger::Retrigger,
bool takeoverDeclick = false);
// MIDI note-on. Resolves the note+velocity to a zone; if none matches (out of
// zone) it is a defined no-op (no voice consumed). Otherwise allocates a free
// voice, or steals one per the policy above. Returns the index of the voice used,
// or kNoVoice for an out-of-zone (unplayed) note.
std::size_t noteOn(int note, int velocity);
// MIDI note-off. Releases the most-recently-started active, non-releasing voice
// playing `note` (so a re-triggered same note releases the newest first, leaving
// the older tail to ring — matches hardware behavior). No-op if none match.
void noteOff(int note);
// CC 123 — MIDI All-Notes-Off: clears the MONO held stack and RELEASES every active voice
// (Gate voices enter their AHDSR release tail; Trigger one-shots ignore release and play
// through their bounded play length). This is the mono stack's ONLY reset path — a phantom
// entry left by a lost note-off would otherwise be resurrected by the fallback and sustain
// forever with no key held. RT-safe (no allocation, bounded by maxVoices).
void allNotesOff();
// CC 120 — MIDI All-Sounds-Off: hard-stops EVERY voice immediately (active_ = false, no
// release ramp), clears the MONO held stack, and silences even Trigger one-shots that would
// ignore a release. Use for panic; CC 123 for the softer "let gates release" behavior.
// RT-safe (no allocation, bounded by maxVoices); callable from the audio thread.
void allSoundsOff();
// REAL-TIME render (S4): sums all active voices into the caller-provided buffer
// `out[0..frameCount)`, ADDING to whatever is there (the caller clears or mixes —
// this never touches memory it does not own and NEVER allocates). This is the
// audio-thread entry point: the VST3 process callback passes the host's own output
// channel buffer, so no allocation, resize, or heap traffic happens under process.
// Voices that finish mid-block go idle and stop contributing. `out` must point at
// at least `frameCount` writable samples; a null `out` or zero count is a no-op.
void render(AudioSample* out, std::size_t frameCount);
// REAL-TIME stereo render (S7): sums all active voices per-channel into the caller's two
// buffers `left`/`right` (each `frameCount` writable samples), ADDING to whatever is there
// (the caller clears/mixes). Same RT discipline as the mono overload — no allocation, no
// resize, no lock. A mono sample plays dual-mono (same value to both channels, centered);
// a stereo sample plays its two channels. A null buffer or zero count is a no-op. The mono
// and stereo render paths are independent output shapes over the SAME voice pool; the active
// channel mode (mono vs stereo bus) picks which one the process callback drives per block.
void render(AudioSample* left, AudioSample* right, std::size_t frameCount);
// TEST / off-thread convenience: appends `frameCount` summed frames to `out`
// (grows it — DO NOT call on the audio thread; it allocates). Delegates to the
// real-time overload after sizing the buffer, so both paths share one mix loop.
// Does not clear existing contents — appends, matching the pre-S4 contract the
// unit tests rely on.
void render(std::vector<AudioSample>& out, std::size_t frameCount);
// Count of currently active voices (for tests / diagnostics).
std::size_t activeVoiceCount() const;
std::size_t maxVoices() const { return voices_.size(); }
static constexpr std::size_t kNoVoice = static_cast<std::size_t>(-1);
private:
// Picks a voice to (re)use for a new note-on: a free voice if any, else a stolen
// one per the documented policy. Always returns a valid index (maxVoices >= 1).
std::size_t allocateVoice();
// Count of active Preserve-engine voices (for the S16 Preserve cap). Rescanned per note-on
// (cheap: bounded by maxVoices) rather than maintained as a running tally.
std::size_t activePreserveVoices() const;
// --- MONO mode (Phase S): last-note priority over a held-note stack ------------
// The stack holds every currently-held, ZONE-RESOLVING note in press order (top = most
// recent = the sounding note while the voice is gated). An out-of-zone note never joins
// (it cannot sound, so it must not later take the voice back on a fallback). Re-pressing
// a held note moves it to the top. Fixed-capacity (128 distinct MIDI notes) — no
// allocation on the audio thread. Velocity is kept per held note so a RETRIGGER fallback
// re-strikes the fallen-back-to note at ITS original velocity.
struct HeldNote { std::uint8_t note; std::uint8_t velocity; };
// Mono note-on: push to the stack and take the voice over (legato retune on a same-sample
// takeover, else a fresh start). Returns 0 (the mono voice) or kNoVoice for out-of-zone
// or out-of-range (note outside [0,127] — rejected BEFORE the stack, which stores uint8).
// The S16 Preserve cap is NOT applied in mono — a single voice runs at most one shifter,
// inherently within any cap; applying it would wrongly drop a Preserve->Preserve takeover.
std::size_t monoNoteOn(int note, int velocity);
// Mono note-off: pop from the stack; if the released note was sounding, fall back to the
// most-recent still-held note (retrigger or legato per monoTrigger_), else release.
void monoNoteOff(int note);
// Drops `note` from the held stack (order of the remaining notes preserved). No-op if absent.
void removeHeld(int note);
std::vector<Voice> voices_;
const Keymap& keymap_;
std::size_t preserveVoiceCap_ = 0; // S16: max simultaneous Preserve voices (0 = no separate cap)
std::uint64_t nextStartOrder_ = 1; // monotonic; 0 reserved for "never started"
VoiceMode voiceMode_ = VoiceMode::Poly;
MonoTrigger monoTrigger_ = MonoTrigger::Retrigger;
bool takeoverDeclick_ = false; // GA fix: declick every restart/steal of a sounding voice
std::array<HeldNote, 128> heldStack_{}; // mono held notes, press order; top = heldCount_-1
std::size_t heldCount_ = 0;
};
// NOTE (preview redesign): the Phase S PreviewCard — a dedicated preview voice isolated
// from the MIDI pool — is RETIRED. The editor's preview trigger is now a synthetic note-on
// at the loaded capture's root note through the SAME VoiceEngine host MIDI drives, so a
// preview is a real voice: it counts against the voice count, can steal / be stolen, and
// respects Poly/Mono + Retrigger/Legato (a deliberate reversal of the earlier isolation
// decision). The FA1 unity-Varispeed demotion in Voice::start went with it — since the GA2
// prime fix the shifter speaks on frame 0 at every ratio, so the demotion bought nothing
// but a second code path.
} // namespace reasampler
@@ -0,0 +1,272 @@
// velocity_curve.cpp — see velocity_curve.h. Pure eval + editing/clamp/inverse map; no host types.
#include "core/instrument/engine/velocity_curve.h"
#include <algorithm> // std::max, std::min, std::abs, std::stable_sort
#include <cmath> // std::fabs
#include <utility> // std::move
namespace reasampler::instrument::engine {
namespace {
double clampVelocity(double v) { return std::clamp(v, kVelMin, kVelMax); }
double clampAmp(double a) { return std::clamp(a, kAmpMin, kAmpMax); }
// Pixel<->box maps (mirror of envelope_edit's timeToX/levelToY). X spans the width for [0,127]; Y
// spans (height-1) rows for amp [0,1] with amp 1 at the TOP (y increases downward).
double velPerPixel(const VelocityCurve::Box& box) {
const int w = std::max(0, box.width);
if (w <= 0) return 0.0;
return (kVelMax - kVelMin) / static_cast<double>(w);
}
double ampPerPixel(const VelocityCurve::Box& box) {
const int h = std::max(0, box.height);
if (h <= 1) return 0.0;
return (kAmpMax - kAmpMin) / static_cast<double>(h - 1);
}
int velToX(const VelocityCurve::Box& box, double velocity) {
const int w = std::max(0, box.width);
if (w <= 0) return box.left;
const double frac = (clampVelocity(velocity) - kVelMin) / (kVelMax - kVelMin);
return box.left + static_cast<int>(frac * static_cast<double>(w) + 0.5);
}
int ampToY(const VelocityCurve::Box& box, double amp) {
const int h = std::max(0, box.height);
if (h <= 1) return box.top;
// amp 1 at top (box.top), amp 0 at bottom (box.top + h - 1).
const double frac = (clampAmp(amp) - kAmpMin) / (kAmpMax - kAmpMin);
return box.top + static_cast<int>((1.0 - frac) * static_cast<double>(h - 1) + 0.5);
}
} // namespace
VelocityCurve VelocityCurve::flat() {
VelocityCurve c;
c.points_ = {{kVelMin, kAmpMax}, {kVelMax, kAmpMax}}; // y = 1 everywhere (R10-F1 Option A)
return c;
}
VelocityCurve VelocityCurve::linear() {
VelocityCurve c;
c.points_ = {{kVelMin, kAmpMin}, {kVelMax, kAmpMax}}; // y = velocity/127
return c;
}
VelocityCurve VelocityCurve::fromPoints(std::vector<VelocityPoint> pts) {
// Box-clamp every point, then stable-sort by velocity (X-order; stable so coincident-X points
// keep their wire order). A stable sort keeps the eval well-defined for duplicate-X knots.
for (VelocityPoint& p : pts) {
p.velocity = clampVelocity(p.velocity);
p.amp = clampAmp(p.amp);
}
std::stable_sort(pts.begin(), pts.end(),
[](const VelocityPoint& a, const VelocityPoint& b) {
return a.velocity < b.velocity;
});
// Fewer than 2 usable points -> can't span [0,127] as a function; fall back to the flat default.
if (pts.size() < 2) return flat();
// Force endpoints present at velocity 0 and 127 (they must exist for eval to be total).
if (pts.front().velocity > kVelMin) {
pts.insert(pts.begin(), VelocityPoint{kVelMin, pts.front().amp});
} else {
pts.front().velocity = kVelMin; // snap a near-0 first point exactly onto the endpoint
}
if (pts.back().velocity < kVelMax) {
pts.push_back(VelocityPoint{kVelMax, pts.back().amp});
} else {
pts.back().velocity = kVelMax; // snap a near-127 last point exactly onto the endpoint
}
VelocityCurve c;
c.points_ = std::move(pts);
return c;
}
namespace {
// FritschCarlson monotone-cubic tangent for one interior knot i, given the secant slopes of the
// two adjacent segments (dPrev = secant into knot i, dNext = secant out of knot i). Returns the
// limited tangent that keeps the cubic Hermite piece monotone and inside the data range.
//
// The rule: a tangent whose adjacent secants have opposite signs (or either is flat) is a local
// extremum — pin the tangent to 0 so the curve does not overshoot past the knot. Otherwise use the
// weighted-harmonic-mean tangent (FritschCarlson eq. 4), which for COLLINEAR knots (dPrev==dNext)
// reduces to that common secant — so collinear control points reproduce the straight line to within
// floating-point rounding (~1e-15), preserving the Option-B / null-response contract for linear().
double fritschCarlsonTangent(double dPrev, double dNext, double spanPrev, double spanNext) {
if (dPrev * dNext <= 0.0) return 0.0; // sign change or a flat neighbour -> local extremum
// Weighted harmonic mean of the two secants (weights = the two segment widths). Collinear case:
// dPrev==dNext==d makes this (w1+w2)*d / ((w1+w2)/... ) collapse to d exactly.
const double w1 = 2.0 * spanNext + spanPrev;
const double w2 = spanNext + 2.0 * spanPrev;
return (w1 + w2) / (w1 / dPrev + w2 / dNext);
}
} // namespace
double VelocityCurve::eval(double velocity) const {
if (points_.empty()) return kAmpMax; // degenerate (shouldn't occur) -> flat unity
if (points_.size() == 1) return clampAmp(points_[0].amp); // 1-point -> that point's amp
const double v = clampVelocity(velocity);
// At or before the first point / at or after the last, read the endpoint amp (the endpoints are
// at 0 and 127, so this only fires exactly at the ends for an in-range velocity).
if (v <= points_.front().velocity) return clampAmp(points_.front().amp);
if (v >= points_.back().velocity) return clampAmp(points_.back().amp);
// Find the segment [points_[i], points_[i+1]] containing v (X-ordered, so a linear scan).
for (std::size_t i = 0; i + 1 < points_.size(); ++i) {
const VelocityPoint& a = points_[i];
const VelocityPoint& b = points_[i + 1];
if (v >= a.velocity && v <= b.velocity) {
const double span = b.velocity - a.velocity;
// Coincident-X neighbours (a step): jump straight to the later point's amp — the segment
// has zero width so there is no interior to blend.
if (span <= 0.0) return clampAmp(b.amp);
// --- Monotone cubic Hermite (FritschCarlson) interpolation on segment [a,b] ---------
// Curved (spline) response, not straight lines. The interpolant provably stays within
// [a.amp, b.amp] between the two knots (no bulge below 0 / above 1), and for collinear
// control points its tangents reduce to the secant slope — so it reproduces the straight
// line to within floating-point rounding (~1e-15), preserving linear()'s null-response
// contract (y = velocity/127 to ~1e-15; the test tolerance of 1e-12 is appropriate).
const double d = (b.amp - a.amp) / span; // secant of THIS segment
// Tangent at a: 0 if a is the first knot (endpoint), else the FC-limited tangent using
// the previous segment's secant. Same for the tangent at b (0 at the last knot).
double mA = d;
if (i > 0) {
const VelocityPoint& prev = points_[i - 1];
const double spanPrev = a.velocity - prev.velocity;
if (spanPrev > 0.0) {
const double dPrev = (a.amp - prev.amp) / spanPrev;
mA = fritschCarlsonTangent(dPrev, d, spanPrev, span);
} else {
mA = 0.0; // coincident-X predecessor (a step at a) -> flat tangent
}
}
double mB = d;
if (i + 2 < points_.size()) {
const VelocityPoint& next = points_[i + 2];
const double spanNext = next.velocity - b.velocity;
if (spanNext > 0.0) {
const double dNext = (next.amp - b.amp) / spanNext;
mB = fritschCarlsonTangent(d, dNext, span, spanNext);
} else {
mB = 0.0; // coincident-X successor (a step at b) -> flat tangent
}
}
// Cubic Hermite basis on the normalized position t across [a,b]. For collinear knots
// mA==mB==d, so h00*a + (h10*span)*d + h01*b + (h11*span)*d collapses to the straight
// line to within floating-point rounding (~1e-15).
const double t = (v - a.velocity) / span;
const double t2 = t * t;
const double t3 = t2 * t;
const double h00 = 2.0 * t3 - 3.0 * t2 + 1.0;
const double h10 = t3 - 2.0 * t2 + t;
const double h01 = -2.0 * t3 + 3.0 * t2;
const double h11 = t3 - t2;
const double y = h00 * a.amp + h10 * span * mA + h01 * b.amp + h11 * span * mB;
return clampAmp(y);
}
}
return clampAmp(points_.back().amp); // unreachable (v is between the endpoints)
}
std::size_t VelocityCurve::addPoint(double velocity, double amp) {
const VelocityPoint p{clampVelocity(velocity), clampAmp(amp)};
// Insert keeping X-order: first index whose velocity is STRICTLY greater than the new one, so a
// duplicate-X point lands immediately after the existing one (a later move can separate them).
std::size_t i = 0;
while (i < points_.size() && points_[i].velocity <= p.velocity) ++i;
points_.insert(points_.begin() + static_cast<std::ptrdiff_t>(i), p);
return i;
}
VelocityPoint VelocityCurve::movePoint(std::size_t index, double velocity, double amp) {
if (index >= points_.size()) return VelocityPoint{}; // no-op (out of range)
const bool isFirst = (index == 0);
const bool isLast = (index + 1 == points_.size());
double newAmp = clampAmp(amp);
double newVel;
if (isFirst) {
newVel = kVelMin; // endpoint pinned in X at 0 — only amp moves
} else if (isLast) {
newVel = kVelMax; // endpoint pinned in X at 127 — only amp moves
} else {
// Interior point: clamp X strictly within its immediate neighbours so it can't cross them.
const double lo = points_[index - 1].velocity;
const double hi = points_[index + 1].velocity;
newVel = std::clamp(clampVelocity(velocity), lo, hi);
}
points_[index] = VelocityPoint{newVel, newAmp};
return points_[index];
}
bool VelocityCurve::deletePoint(std::size_t index) {
if (index >= points_.size()) return false;
if (index == 0 || index + 1 == points_.size()) return false; // endpoints are not deletable
points_.erase(points_.begin() + static_cast<std::ptrdiff_t>(index));
return true;
}
VelocityCurve::CurvePixel VelocityCurve::pixelFromPoint(const Box& box, const VelocityPoint& p) {
return CurvePixel{velToX(box, p.velocity), ampToY(box, p.amp)};
}
VelocityPoint VelocityCurve::pointFromPixel(const Box& box, int x, int y) {
// The exact inverse of velToX/ampToY (within the one-pixel rounding quantum). Degenerate
// dimensions collapse the same way the forward map does: velToX pins to box.left (velocity 0),
// ampToY pins to box.top (amp 1).
VelocityPoint p;
const int w = std::max(0, box.width);
const int h = std::max(0, box.height);
p.velocity = (w <= 0)
? kVelMin
: clampVelocity(kVelMin + static_cast<double>(x - box.left) / static_cast<double>(w) *
(kVelMax - kVelMin));
p.amp = (h <= 1)
? kAmpMax
: clampAmp(kAmpMax - static_cast<double>(y - box.top) / static_cast<double>(h - 1) *
(kAmpMax - kAmpMin));
return p;
}
int VelocityCurve::pointAtPixel(const Box& box, int x, int y) const {
for (std::size_t i = 0; i < points_.size(); ++i) {
const int px = velToX(box, points_[i].velocity);
const int py = ampToY(box, points_[i].amp);
if (std::abs(x - px) <= kCurveNodeGrabRadius && std::abs(y - py) <= kCurveNodeGrabRadius) {
return static_cast<int>(i);
}
}
return -1;
}
VelocityCurve VelocityCurve::resolvePointDrag(const VelocityCurve& grabCurve, std::size_t index,
const Box& box, int dxPixels, int dyPixels) {
VelocityCurve out = grabCurve;
if (index >= out.points_.size()) return out; // out of range -> no motion
const double velPerPx = velPerPixel(box);
const double ampPerPx = ampPerPixel(box);
if (velPerPx <= 0.0 || ampPerPx <= 0.0) return out; // degenerate box -> no motion
const VelocityPoint& grab = grabCurve.points_[index];
const double newVel = grab.velocity + static_cast<double>(dxPixels) * velPerPx;
// Y increases downward but amp increases upward, so a downward drag (positive dy) LOWERS amp.
const double newAmp = grab.amp - static_cast<double>(dyPixels) * ampPerPx;
out.movePoint(index, newVel, newAmp); // applies box + neighbour-X + endpoint-pin clamps
return out;
}
bool VelocityCurve::equals(const VelocityCurve& other, double eps) const {
if (points_.size() != other.points_.size()) return false;
for (std::size_t i = 0; i < points_.size(); ++i) {
if (std::fabs(points_[i].velocity - other.points_[i].velocity) > eps) return false;
if (std::fabs(points_[i].amp - other.points_[i].amp) > eps) return false;
}
return true;
}
} // namespace reasampler::instrument::engine
+171
View File
@@ -0,0 +1,171 @@
// velocity_curve.h — PURE velocity->amp transfer curve (S-VIEW-9, r10). NO VST3, NO REAPER, NO
// SWELL/LICE, NO vendor/ includes at the boundary. The mirror of envelope_edit / card_drag: the
// eval + the clamp/order/inverse-map arithmetic live here, unit-tested outside the DAW; the future
// editor shell (reasampler_editor.cpp, S-VIEW-10) draws the box + node handles and feeds each move's
// pixel delta back through here, committing the result to the zone through the same off-audio-thread
// path a slider edit uses.
//
// WHAT IT IS. A monotonic-in-x transfer function mapping MIDI velocity (X: 0..127) to an amp scalar
// (Y: 0..1), authored as an ordered list of control points. eval(velocity) is called ONCE per
// note-on in Voice::start() (never per frame) to set the voice's velocityGain_, replacing the fixed
// linear velocity/127 map. The curve is a per-PerformanceZone performance characteristic (D-B) — a
// sibling of the AHDSR envelope, pitch engine, and keyTrack scalar — so it varies per sound, stored
// on PerformanceZone and resolved onto the KeyZone at keymap build (mirror of keyTrack).
//
// DEFAULT — flat y=1 (fork R10-F1 Option A, Daniel 2026-07-27). VelocityCurve::flat() is the seeded
// default: EVERY velocity plays at unity amp. This is a DELIBERATE, Daniel-approved behavior change
// vs. the shipped linear velocity/127 map — soft hits are now full level until a curve is drawn.
// NOT bit-identical to the pre-r10 engine, by design; do not "preserve" the linear response.
//
// THE INVARIANT (mirror of envelope_edit's S-VIEW-F2). A drag/edit can NEVER produce a curve eval
// couldn't handle:
// * X-ORDERED — a point clamps between its predecessor's and successor's velocity, so control
// points never cross in X. This is what makes eval a well-defined FUNCTION (one amp per
// velocity): each X falls in exactly one [p_i, p_{i+1}] segment.
// * BOX-CLAMPED — velocity clamps to [0,127], amp clamps to [0,1] (the drawn box).
// Both endpoints (velocity 0 and 127) are always present so eval is total over [0,127]; delete
// refuses to remove them, and the constructors seed them.
#pragma once
#include <cstdint>
#include <vector>
// DELIBERATELY dependency-free at the boundary (no editor_geometry / Rect). This module sits BELOW
// sampler_core in the link graph (KeyZone carries a VelocityCurve; Voice::start calls eval), and the
// engine must not gain a transitive dependency on the editor's layout types. The editor hit-test /
// inverse-map therefore takes an explicit pixel box (boxLeft/boxTop/boxWidth/boxHeight) rather than a
// Rect — the future editor shell (S-VIEW-10) passes its box coords directly. Mirror of envelope_edit's
// role, but one layer lower, so the coupling stays out of the engine core.
namespace reasampler::instrument::engine {
// The MIDI velocity domain [0,127] and the amp range [0,1] — the box every point clamps into.
inline constexpr double kVelMin = 0.0;
inline constexpr double kVelMax = 127.0;
inline constexpr double kAmpMin = 0.0;
inline constexpr double kAmpMax = 1.0;
// One control point: a (velocity, amp) knot the curve passes through. Both fields are box-clamped
// by the mutators; a raw-constructed point is NOT auto-clamped (the mutators own the invariant), so
// build curves through the named constructors / addPoint rather than pushing raw points.
struct VelocityPoint {
double velocity = 0.0; // X, [0,127]
double amp = 0.0; // Y, [0,1]
};
// The pick radius (px) around a node's drawn point for the editor hit-test. Mirrors
// envelope_edit::kNodeGrabRadius / waveform_view::kMarkerGrabWidth.
inline constexpr int kCurveNodeGrabRadius = 6;
// A velocity->amp transfer curve: an X-ORDERED list of control points spanning [0,127], evaluated by
// a MONOTONE cubic Hermite spline (FritschCarlson slope limiting) through the knots — a genuine
// curved response (Daniel 2026-07-27: "straight lines sound like shit"), not a polyline. Each
// velocity still maps to exactly one amp: the interpolant is single-valued and provably stays within
// each segment's amp range, so the curve never overshoots below 0 or above 1. For COLLINEAR knots the
// FritschCarlson tangents reduce to the secant slope, so the spline reproduces the straight line to
// within floating-point rounding (~1e-15) — that preserves linear()'s null-response contract
// (y = velocity/127 to ~1e-15; the 1e-12 test tolerance is deliberately conservative). The two endpoints
// (velocity 0 and 127) are load-bearing: they keep eval total and are never deletable.
class VelocityCurve {
public:
// R10-F1 default (Option A): flat y=1 — endpoints (0,1) and (127,1); every velocity -> unity.
static VelocityCurve flat();
// The classic linear ramp y = velocity/127 — endpoints (0,0) and (127,1). Retained for tests
// and as the Option-B seed; NOT the default (see R10-F1).
static VelocityCurve linear();
// Rebuild a curve from a deserialized point list, REPAIRING the invariant defensively (the
// deserialization seam, sample_map's zones-payload v7). Each point is box-clamped; the list is
// stable-sorted by velocity (X-ordered); endpoints at velocity 0 and 127 are forced present
// (an absent endpoint is synthesized at the nearest interior amp, or unity for an empty list).
// A list with fewer than 2 usable points falls back to flat(). Never trusts the wire blindly —
// a corrupt/truncated blob yields a well-formed curve, never an invariant-violating one.
static VelocityCurve fromPoints(std::vector<VelocityPoint> pts);
// The control points, X-ordered, first at velocity 0 and last at velocity 127 (invariant).
const std::vector<VelocityPoint>& points() const { return points_; }
std::size_t size() const { return points_.size(); }
// Evaluate the curve at `velocity` -> amp in [0,1]. Velocity is box-clamped to [0,127] first,
// so an out-of-range note (shouldn't occur) reads the nearest endpoint. Between two adjacent
// points the amp follows a MONOTONE cubic Hermite spline (FritschCarlson slope limiting) — a
// true curve that provably stays within the two knots' amp range (no overshoot below 0 / above
// 1) and reproduces the straight line to within floating-point rounding (~1e-15) for collinear
// knots. Single-valued / monotonic in X.
// Degenerate cases (shouldn't occur post-construction): an EMPTY curve returns kAmpMax (flat
// unity); a ONE-point curve returns that point's amp.
double eval(double velocity) const;
// --- Editing (for the S-VIEW-10 editor UI) --------------------------------------------------
// Insert a new control point, box-clamped, keeping the list X-ordered by velocity. Returns the
// index of the inserted point. A new point at a velocity that duplicates an existing one is
// inserted immediately AFTER it (so a subsequent move can separate them); the endpoints are not
// special-cased on insert (a point at exactly 0 or 127 inserts adjacent to that endpoint).
std::size_t addPoint(double velocity, double amp);
// Move point `index` to (velocity, amp), box-clamped AND X-clamped between its immediate
// neighbours so it cannot cross them (monotonic-X grammar). The two ENDPOINTS are pinned in X
// (index 0 stays at velocity 0, the last stays at 127) — only their AMP moves; their velocity
// argument is ignored. An out-of-range index is a no-op. Returns the (possibly clamped)
// resulting point.
VelocityPoint movePoint(std::size_t index, double velocity, double amp);
// Delete point `index`. The two endpoints (index 0 and the last) are NOT deletable — a request
// to remove either, or an out-of-range index, is a no-op returning false. Returns true iff a
// point was removed.
bool deletePoint(std::size_t index);
// --- Editor hit-test + inverse map (mirror of envelope_edit) --------------------------------
// The drawn box, in pixels: origin (boxLeft, boxTop), `boxWidth` px wide, `boxHeight` px tall.
// X = velocity across the width (0 at boxLeft, 127 at boxLeft+boxWidth); Y = amp UP the height
// (amp 1 at boxTop, amp 0 at boxTop+boxHeight-1). Passed explicitly (not a Rect) so this module
// stays free of editor-layout types — see the header preamble.
struct Box {
int left = 0;
int top = 0;
int width = 0;
int height = 0;
};
// Which control point a grab at (x,y) lands on, given the drawn `box`. Returns the index of the
// first point within the pick radius in BOTH axes, or -1 for a miss. First-match in point order
// for determinism (mirror of nodeAtPoint).
int pointAtPixel(const Box& box, int x, int y) const;
// A node's drawn pixel position (S-VIEW-10). The ONE point->pixel mapping — the same mapping
// pointAtPixel hit-tests against — exposed so the editor shell draws the trace + node handles
// at exactly the coordinates the hit-test expects (draw and grab can never drift).
struct CurvePixel {
int x = 0;
int y = 0;
};
static CurvePixel pixelFromPoint(const Box& box, const VelocityPoint& p);
// The absolute pixel -> (velocity, amp) inverse (S-VIEW-10): where an empty-space click lands
// as a NEW control point, box-clamped. The exact inverse of pixelFromPoint's mapping (within
// the one-pixel quantum), so an added point appears under the cursor. Degenerate box: a
// zero-width box reads velocity 0; a height <= 1 box reads amp 1 (the top row), mirroring
// pixelFromPoint's degenerate collapse.
static VelocityPoint pointFromPixel(const Box& box, int x, int y);
// Resolve a drag of point `index` by a pixel delta since grab, given the curve AS OF GRAB TIME
// (`grabCurve` — the shell snapshots it on mouse-down so the delta is absolute) and the box.
// Maps the pixel delta to a (velocity, amp) delta over the box, then applies movePoint's clamp
// (box + neighbour X + endpoint X-pin). A zero-width/height box or out-of-range index returns
// `grabCurve` unchanged. Pure — mirror of resolveNodeDrag.
static VelocityCurve resolvePointDrag(const VelocityCurve& grabCurve, std::size_t index,
const Box& box, int dxPixels, int dyPixels);
// Equality (for tests + round-trip assertions): same point count + each point equal within a
// tight epsilon.
bool equals(const VelocityCurve& other, double eps = 1e-9) const;
private:
// Points are always X-ordered with an endpoint at 0 and 127. Constructed only through the named
// constructors + deserialize (see sample_map), which establish that invariant; the mutators
// preserve it.
std::vector<VelocityPoint> points_;
};
} // namespace reasampler::instrument::engine