DSPark 1.8.0
Header-only C++20 DSP for real-time and offline audio
Loading...
Searching...
No Matches
PitchShifter.h
1// DSPark - Professional Audio DSP Framework
2// Copyright (c) 2026 Cristian Moresi - MIT License
3
4#pragma once
5
77#include "../Core/AudioBuffer.h"
78#include "../Core/AudioSpec.h"
79#include "../Core/DenormalGuard.h"
80#include "../Core/DspMath.h"
81#include "../Core/Interpolation.h"
82#include "../Core/StateBlob.h"
83#include "detail/PhaseVocoderEngine.h"
84#include "detail/StudioVocoder.h"
85
86#include <algorithm>
87#include <atomic>
88#include <cmath>
89#include <cstddef>
90#include <cstdint>
91#include <vector>
92
93namespace dspark {
94
101template <FloatType T>
103{
104public:
105 // -- Lifecycle -------------------------------------------------------------
106
125 void prepare(const AudioSpec& spec, int fftSize = 0)
126 {
127 if (!spec.isValid()) return;
128 if (fftSize != 0 && ((fftSize & (fftSize - 1)) != 0
129 || fftSize < 256 || fftSize > (1 << 20)))
130 return;
131
132 prepared_.store(false, std::memory_order_relaxed);
133
134 numChannels_ = std::max(1, spec.numChannels);
135 // 0 selects each engine's default frame: 2048 for Standard/High (the
136 // published rendering), about 43 ms for Studio (2048 at 44.1 and
137 // 48 kHz, 4096 at 88.2 and 96 kHz).
138 const int fftStudio = fftSize != 0 ? fftSize : studioDefaultFrame(spec.sampleRate);
139 if (fftSize == 0) fftSize = 2048;
140
141 // The reader trails the write head by readOffset_; the window/OLA
142 // chain adds another fftSize - synthHop, so the measured wet latency
143 // is readOffset_ + fftSize - synthHop = 2 * fftSize (exact at unity
144 // ratio). The dry path must delay by the SAME value or partial mixes
145 // comb-filter.
146 const int synthHop = fftSize / 4;
147 readOffset_ = fftSize + synthHop;
148 latency_ = readOffset_ + fftSize - synthHop;
149 // Studio: the resample reader runs at the pitch ratio behind the
150 // completed stream. At ratio a an input sample reaches the completed
151 // stream after half a frame, a frame of analysis, the onset
152 // lookahead and one analysis hop, and the reader needs its sinc
153 // reach beyond that; the worst case is the lowest ratio, 0.5:
154 // N/(2a) + N/2 + lookahead + taps/a = 1.5 N + lookahead + 32. A
155 // strike anchor may move an analysis frame ahead of the nominal
156 // timeline by up to a quarter frame at that ratio, which the last
157 // term covers. One fixed latency for every ratio, so the dry path
158 // and host compensation hold at any pitch.
159 studio_.prepare(spec.sampleRate, numChannels_, fftStudio, true);
160 studio_.setAnchorLeadLimit(static_cast<double>(fftStudio / 4));
161 latencyStudio_ = fftStudio + fftStudio / 2 + studio_.lookahead() + fftStudio / 4 + 64;
162 latencyLegacy_ = latency_;
163 drySize_ = 1;
164 while (drySize_ < std::max(latencyLegacy_, latencyStudio_) + 1) drySize_ <<= 1;
165 dryMask_ = drySize_ - 1;
166
167 // The engine owns the analysis rings, spectral state and OLA ring;
168 // resample-back compensation stages (anti-alias taper, formant
169 // pre-warp target) are enabled because this owner resamples.
170 //
171 // The three remaining capabilities are refused, and the last two are
172 // written out rather than defaulted because a future default that
173 // flips would change what this effect sounds like. The locked
174 // analysis hop breaks the resample-back reader's Ra = Rs/ratio
175 // assumption and buys nothing here. The spectral-flux onset detector
176 // fires earlier and more often than the frame-energy test, and every
177 // firing resets phase, so adopting it smears energy ahead of a strike
178 // - audible pre-echo on percussive material - and re-renders every
179 // release already in a user's hands. A mix finished last month is not
180 // restored by a release note. How large the smear is depends on the
181 // material and on the estimator it is measured with, so no figure is
182 // quoted here; that it is audible on percussive material is the
183 // durable part, and it is why this effect keeps the detector it
184 // shipped with.
185 engine_.prepare(spec.sampleRate, numChannels_, fftSize, true, false, false, false);
186 mixMaxStep_ = static_cast<T>(1.0 / std::max(1.0, spec.sampleRate * 0.02));
187 accumMask_ = engine_.olaMask();
188
189 dryRing_.assign(static_cast<size_t>(numChannels_), {});
190 for (int ch = 0; ch < numChannels_; ++ch)
191 dryRing_[static_cast<size_t>(ch)].assign(static_cast<size_t>(drySize_), T(0));
192
193 publishEngineParams();
194 prepared_.store(true, std::memory_order_relaxed);
195 reset();
196 }
197
199 void reset() noexcept
200 {
201 if (!prepared_.load(std::memory_order_relaxed)) return;
202 studioActive_ = quality_.load(std::memory_order_relaxed) == Quality::Studio;
203 latency_ = studioActive_ ? latencyStudio_ : latencyLegacy_;
204 engine_.reset();
205 studio_.reset();
206 for (auto& r : dryRing_) std::fill(r.begin(), r.end(), T(0));
207 studioOut_ = 0;
208 {
209 // The reader starts where output 0 shows input -latency, and
210 // advances at the pitch ratio from there.
211 const double start = studio_.streamPositionOf(-static_cast<double>(latencyStudio_));
212 studioReadInt_ = static_cast<int64_t>(std::floor(start));
213 studioReadFrac_ = start - static_cast<double>(studioReadInt_);
214 }
215
216 dryPos_ = 0;
217 readPosInt_ = engine_.writeHead() - readOffset_;
218 readPosFrac_ = 0.0;
219 currentMix_ = mix_.load(std::memory_order_relaxed);
220 highReader_ = quality_.load(std::memory_order_relaxed) != Quality::Standard;
221 readerFadeLeft_ = 0; // start settled on the selected reader
222 }
223
224 // -- Parameters (thread-safe) -----------------------------------------------
225
227 enum class Quality
228 {
229 Standard,
230 High,
231 Studio
232 };
233
241 void setQuality(Quality quality) noexcept
242 {
243 quality = static_cast<Quality>(std::clamp(static_cast<int>(quality), 0,
244 static_cast<int>(Quality::Studio)));
245 quality_.store(quality, std::memory_order_relaxed);
246 }
247
249 [[nodiscard]] Quality getQuality() const noexcept
250 {
251 return quality_.load(std::memory_order_relaxed);
252 }
253
261 void setSemitones(T st) noexcept
262 {
263 if (!std::isfinite(st)) return;
264 semitones_.store(std::clamp(st, T(-12), T(12)), std::memory_order_relaxed);
265 publishEngineParams();
266 }
267
270 void setPitchRatio(T ratio) noexcept
271 {
272 if (!std::isfinite(ratio)) return;
273 ratio = std::clamp(ratio, T(0.5), T(2));
274 semitones_.store(static_cast<T>(12.0 * std::log2(static_cast<double>(ratio))),
275 std::memory_order_relaxed);
276 publishEngineParams();
277 }
278
283 void setMix(T mix) noexcept
284 {
285 if (!std::isfinite(mix)) return;
286 mix_.store(std::clamp(mix, T(0), T(1)), std::memory_order_relaxed);
287 }
288
290 void setTransientPreserve(bool enabled) noexcept
291 {
292 transientPreserve_.store(enabled, std::memory_order_relaxed);
293 publishEngineParams();
294 }
295
305 void setFormantPreserve(bool enabled) noexcept
306 {
307 formantPreserve_.store(enabled, std::memory_order_relaxed);
308 publishEngineParams();
309 }
310
312 [[nodiscard]] T getSemitones() const noexcept
313 {
314 return semitones_.load(std::memory_order_relaxed);
315 }
316
318 [[nodiscard]] T getMix() const noexcept { return mix_.load(std::memory_order_relaxed); }
319
321 [[nodiscard]] bool getTransientPreserve() const noexcept
322 {
323 return transientPreserve_.load(std::memory_order_relaxed);
324 }
325
327 [[nodiscard]] bool getFormantPreserve() const noexcept
328 {
329 return formantPreserve_.load(std::memory_order_relaxed);
330 }
331
336 [[nodiscard]] int getLatency() const noexcept { return latency_; }
337
339 [[nodiscard]] std::vector<uint8_t> getState() const
340 {
341 StateWriter w(stateId("PSHF"), 1);
342 w.write("semitones", static_cast<float>(semitones_.load(std::memory_order_relaxed)));
343 w.write("mix", static_cast<float>(mix_.load(std::memory_order_relaxed)));
344 w.write("transient", transientPreserve_.load(std::memory_order_relaxed));
345 w.write("formant", formantPreserve_.load(std::memory_order_relaxed));
346 w.write("quality", static_cast<int32_t>(quality_.load(std::memory_order_relaxed)));
347 return w.blob();
348 }
349
351 bool setState(const uint8_t* data, size_t size)
352 {
353 StateReader r(data, size);
354 if (!r.isValid() || r.processorId() != stateId("PSHF")) return false;
355 setSemitones(static_cast<T>(r.read("semitones", 0.0f)));
356 setMix(static_cast<T>(r.read("mix", 1.0f)));
357 setTransientPreserve(r.read("transient", true));
358 setFormantPreserve(r.read("formant", false));
359 setQuality(static_cast<Quality>(r.read("quality", 0))); // clamped inside
360 return true;
361 }
362
363 // -- Processing --------------------------------------------------------------
364
373 void processBlock(AudioBufferView<T> buffer) noexcept
374 {
375 if (!prepared_.load(std::memory_order_relaxed)) return;
376 DenormalGuard guard;
377
378 const int nCh = std::min(buffer.getNumChannels(), numChannels_);
379 const int nS = buffer.getNumSamples();
380 // Crossing between Studio and the Standard/High engine restarts the stream:
381 // the two run different latencies, so no crossfade can join them.
382 if ((quality_.load(std::memory_order_relaxed) == Quality::Studio) != studioActive_)
383 reset();
384 if (studioActive_)
385 {
386 processStudio(buffer, nCh, nS);
387 return;
388 }
389
390 // Rate-limited mix ramp (moveTowards, exact landing; settled it
391 // reduces to the constant, bit-identically). A per-block ramp landed
392 // in 0.7 ms with 32-sample blocks.
393 const T mixTarget = mix_.load(std::memory_order_relaxed);
394 const T mixStart = currentMix_;
395
396 // Reader selection: a change crossfades from the old reader to the
397 // new one over kReaderFade samples (the readers differ at HF).
398 const bool highQuality = quality_.load(std::memory_order_relaxed) == Quality::High;
399 if (highQuality != highReader_)
400 {
401 highReader_ = highQuality;
402 readerFadeLeft_ = kReaderFade;
403 }
404
405 int i = 0;
406 while (i < nS)
407 {
408 const int chunk = std::min(nS - i, engine_.samplesToNextHop());
409
410 // 1. Push input into the analysis ring and the dry-compensation ring.
411 for (int ch = 0; ch < nCh; ++ch)
412 {
413 const T* in = buffer.getChannel(ch) + i;
414 engine_.pushInput(ch, in, chunk);
415 auto& dry = dryRing_[static_cast<size_t>(ch)];
416 int dp = dryPos_;
417 for (int k = 0; k < chunk; ++k)
418 {
419 dry[static_cast<size_t>(dp)] = in[k];
420 dp = (dp + 1) & dryMask_;
421 }
422 }
423
424 // 2. Produce output: fractional read of the synthesis stream + mix.
425 const double ratio = engine_.activeRatio();
426 int64_t rpEnd = readPosInt_;
427 double rfEnd = readPosFrac_;
428 for (int ch = 0; ch < nCh; ++ch)
429 {
430 T* out = buffer.getChannel(ch) + i;
431 const T* acc = engine_.olaData(ch);
432 const auto& dry = dryRing_[static_cast<size_t>(ch)];
433
434 int64_t rp = readPosInt_;
435 double rf = readPosFrac_;
436 int dp = dryPos_;
437 int fadeLeft = readerFadeLeft_;
438
439 for (int k = 0; k < chunk; ++k)
440 {
441 T wet = readWet(highReader_, acc, rp, rf);
442 if (fadeLeft > 0)
443 {
444 const T w = static_cast<T>(fadeLeft) * (T(1) / T(kReaderFade));
445 const T old = readWet(!highReader_, acc, rp, rf);
446 wet += (old - wet) * w;
447 --fadeLeft;
448 }
449 const int dryIdx = (dp - latency_) & dryMask_;
450 const T drySample = dry[static_cast<size_t>(dryIdx)];
451 const T mixVal = moveTowards(mixStart, mixTarget, mixMaxStep_ * static_cast<T>(i + k + 1));
452 // Two-product blend: exact at both ends (mix 1 emits the
453 // wet stream bit-exactly, mix 0 the delayed dry).
454 out[k] = drySample * (T(1) - mixVal) + wet * mixVal;
455
456 rf += ratio;
457 const auto adv = static_cast<int64_t>(rf);
458 rp += adv;
459 rf -= static_cast<double>(adv);
460 dp = (dp + 1) & dryMask_;
461 }
462 if (ch == 0) { rpEnd = rp; rfEnd = rf; } // recurrence result
463 }
464
465 // Commit shared positions once per chunk, using the SAME
466 // per-sample recurrence result the output loops computed (taken
467 // from channel 0, whose loop ran it already): a single
468 // frac + ratio * chunk product rounds differently for different
469 // chunk sizes, and chunk boundaries follow the host block size,
470 // so committing the product form made the output depend on how
471 // the host chopped the stream (~1 ulp per flip, but a bit-exact
472 // contract is a bit-exact contract).
473 {
474 if (nCh == 0) // channel-less call: advance the stream anyway
475 {
476 for (int k = 0; k < chunk; ++k)
477 {
478 rfEnd += ratio;
479 const auto adv = static_cast<int64_t>(rfEnd);
480 rpEnd += adv;
481 rfEnd -= static_cast<double>(adv);
482 }
483 }
484 readPosInt_ = rpEnd;
485 readPosFrac_ = rfEnd;
486 dryPos_ = (dryPos_ + chunk) & dryMask_;
487 readerFadeLeft_ = std::max(0, readerFadeLeft_ - chunk);
488 }
489
490 // 3. Advance the engine (runs an STFT hop at the analysis boundary).
491 engine_.commitInput(chunk, nCh);
492
493 i += chunk;
494 }
495
496 currentMix_ = moveTowards(mixStart, mixTarget, mixMaxStep_ * static_cast<T>(nS));
497 }
498
499private:
510 void processStudio(AudioBufferView<T> buffer, int nCh, int nS) noexcept
511 {
512 const T mixTarget = mix_.load(std::memory_order_relaxed);
513 const T mixStart = currentMix_;
514 const int64_t mask = studio_.olaMask();
515 int i = 0;
516 while (i < nS)
517 {
518 const int need = studio_.samplesToNextHop();
519 if (need == 0)
520 {
521 studio_.commitInput(0, nCh);
522 continue;
523 }
524 const int chunk = std::min(nS - i, need);
525 for (int ch = 0; ch < nCh; ++ch)
526 {
527 const T* in = buffer.getChannel(ch) + i;
528 studio_.pushInput(ch, in, chunk);
529 auto& dry = dryRing_[static_cast<size_t>(ch)];
530 int dp = dryPos_;
531 for (int k = 0; k < chunk; ++k)
532 {
533 dry[static_cast<size_t>(dp)] = in[k];
534 dp = (dp + 1) & dryMask_;
535 }
536 }
537
538 const double ratio = studio_.activeRatio();
539 int64_t rpEnd = studioReadInt_;
540 double rfEnd = studioReadFrac_;
541 for (int ch = 0; ch < nCh; ++ch)
542 {
543 T* out = buffer.getChannel(ch) + i;
544 const T* acc = studio_.olaData(ch);
545 const auto& dry = dryRing_[static_cast<size_t>(ch)];
546 int64_t rp = studioReadInt_;
547 double rf = studioReadFrac_;
548 int dp = dryPos_;
549 for (int k = 0; k < chunk; ++k)
550 {
551 const T wet = reader_.readRing(acc, mask, rp, rf);
552 const T drySample = dry[static_cast<size_t>((dp - latency_) & dryMask_)];
553 const T mixVal = moveTowards(mixStart, mixTarget, mixMaxStep_ * static_cast<T>(i + k + 1));
554 out[k] = drySample * (T(1) - mixVal) + wet * mixVal;
555 rf += ratio;
556 const auto adv = static_cast<int64_t>(rf);
557 rp += adv;
558 rf -= static_cast<double>(adv);
559 dp = (dp + 1) & dryMask_;
560 }
561 if (ch == 0) { rpEnd = rp; rfEnd = rf; }
562 }
563 if (nCh == 0)
564 {
565 for (int k = 0; k < chunk; ++k)
566 {
567 rfEnd += ratio;
568 const auto adv = static_cast<int64_t>(rfEnd);
569 rpEnd += adv;
570 rfEnd -= static_cast<double>(adv);
571 }
572 }
573 studioReadInt_ = rpEnd;
574 studioReadFrac_ = rfEnd;
575 dryPos_ = (dryPos_ + chunk) & dryMask_;
576 studioOut_ += chunk;
577
578 // Steer the frame after the next one: where the reader will stand
579 // on the input timeline when it reaches that frame's centre.
580 const double c = studio_.nextStreamCentre() + static_cast<double>(studio_.synthHop());
581 const double here = static_cast<double>(studioReadInt_) + studioReadFrac_;
582 studio_.steerTimeline(static_cast<double>(studioOut_) + (c - here) / ratio
583 - static_cast<double>(latencyStudio_));
584 studio_.commitInput(chunk, nCh);
585 i += chunk;
586 }
587 currentMix_ = moveTowards(mixStart, mixTarget, mixMaxStep_ * static_cast<T>(nS));
588 }
589
591 [[nodiscard]] static int studioDefaultFrame(double sampleRate) noexcept
592 {
593 int n = 256;
594 while (n < (1 << 16) && static_cast<double>(n) * 1.5 < 0.0427 * sampleRate) n <<= 1;
595 return n;
596 }
597
599 [[nodiscard]] T readWet(bool high, const T* acc, int64_t ip, double frac) const noexcept
600 {
601 return high ? reader_.readRing(acc, accumMask_, ip, frac)
602 : readCatmullRom(acc, ip, frac);
603 }
604
606 [[nodiscard]] T readCatmullRom(const T* acc, int64_t ip, double frac) const noexcept
607 {
608 const int64_t m = accumMask_;
609 const T x0 = acc[static_cast<size_t>((ip - 1) & m)];
610 const T x1 = acc[static_cast<size_t>(ip & m)];
611 const T x2 = acc[static_cast<size_t>((ip + 1) & m)];
612 const T x3 = acc[static_cast<size_t>((ip + 2) & m)];
613 const T f = static_cast<T>(frac);
614 return x1 + T(0.5) * f * (x2 - x0
615 + f * (T(2) * x0 - T(5) * x1 + T(4) * x2 - x3
616 + f * (T(3) * (x1 - x2) + x3 - x0)));
617 }
618
621 void publishEngineParams() noexcept
622 {
623 typename detail::PhaseVocoderEngine<T>::Params p;
624 p.targetSemitones = static_cast<double>(semitones_.load(std::memory_order_relaxed));
625 p.transientPreserve = transientPreserve_.load(std::memory_order_relaxed);
626 p.formantPreserve = formantPreserve_.load(std::memory_order_relaxed);
627 engine_.publishParams(p);
628 typename detail::StudioVocoder<T>::Params q;
629 q.targetSemitones = p.targetSemitones;
630 q.transientPreserve = p.transientPreserve;
631 q.formantPreserve = p.formantPreserve;
632 studio_.publishParams(q);
633 }
634
635 // -- Members -----------------------------------------------------------------
636 int numChannels_ = 0;
637 std::atomic<bool> prepared_ { false };
638
639 int readOffset_ = 2560;
640 int latency_ = 4096;
641 int drySize_ = 8192;
642 int dryMask_ = 8191;
643 int64_t accumMask_ = 8191;
644
645 detail::PhaseVocoderEngine<T> engine_;
646 detail::StudioVocoder<T> studio_;
647 bool studioActive_ = true;
648 int latencyStudio_ = 5184;
649 int latencyLegacy_ = 4096;
650 int64_t studioReadInt_ = 0;
651 double studioReadFrac_ = 0.0;
652 int64_t studioOut_ = 0;
653 SincInterpolator<T> reader_;
656 static constexpr int kReaderFade = 64;
657 bool highReader_ = false;
658 int readerFadeLeft_ = 0;
659
660 std::vector<std::vector<T>> dryRing_;
661
662 int dryPos_ = 0;
663 int64_t readPosInt_ = 0;
664 double readPosFrac_ = 0.0;
665 T currentMix_ = T(1);
666 T mixMaxStep_ = T(1.0 / 960.0);
667
668 std::atomic<T> semitones_ { T(0) };
669 std::atomic<T> mix_ { T(1) };
670 std::atomic<bool> transientPreserve_ { true };
671 std::atomic<bool> formantPreserve_ { false };
672 std::atomic<Quality> quality_ { Quality::Studio };
673};
674
675} // namespace dspark
Non-owning view over audio channel data.
Definition AudioBuffer.h:50
RAII scope guard to disable denormalised (subnormal) floating-point numbers.
Real-time phase-vocoder pitch shifter (+-12 semitones, stereo-linked).
void setQuality(Quality quality) noexcept
Selects the engine and reader. Thread-safe. Standard and High crossfade into each other over 64 sampl...
bool getFormantPreserve() const noexcept
Quality getQuality() const noexcept
void setTransientPreserve(bool enabled) noexcept
Enables phase reset on detected transients (default on).
T getSemitones() const noexcept
bool setState(const uint8_t *data, size_t size)
Restores parameters from a blob (tolerant; rejects foreign ids).
std::vector< uint8_t > getState() const
Serializes the parameter state (setup/UI threads; allocates).
void reset() noexcept
Clears all signal state (keeps parameters). Safe on the audio thread.
void setMix(T mix) noexcept
Dry/wet mix, [0, 1]. The dry path is latency-compensated and the mix is ramped over at least 20 ms (t...
void setSemitones(T st) noexcept
Sets the pitch shift in semitones, clamped to +-12.
void setFormantPreserve(bool enabled) noexcept
Keeps formants (vocal timbre) in place while pitch moves.
void prepare(const AudioSpec &spec, int fftSize=0)
Allocates all rings and spectral state.
int getLatency() const noexcept
Reports total latency in samples. Studio: 1.5 * fftSize + lookahead + fftSize / 4 + 64 (5184,...
void setPitchRatio(T ratio) noexcept
Sets the pitch shift as a frequency ratio, clamped to [0.5, 2]. Non-finite values are ignored.
void processBlock(AudioBufferView< T > buffer) noexcept
Processes audio in-place.
bool getTransientPreserve() const noexcept
Quality
Engine and resample-back reader (see the file overview).
@ High
Standard/High engine, 32-tap windowed-sinc reader: transparent HF.
@ Studio
Studio engine (default): see the file overview.
@ Standard
Standard/High engine, 4-point Catmull-Rom reader: the earlier default rendering.
T getMix() const noexcept
Tolerant reader: missing keys yield defaults, unknown keys are skipped.
Definition StateBlob.h:161
float read(const char *key, float defaultValue) const
Reads a float, or defaultValue when the key is absent.
Definition StateBlob.h:204
bool isValid() const noexcept
Definition StateBlob.h:199
uint32_t processorId() const noexcept
Definition StateBlob.h:200
Serializes key/value parameters into a versioned blob.
Definition StateBlob.h:53
std::vector< uint8_t > blob() const
Finalizes and returns the blob.
Definition StateBlob.h:105
void write(const char *key, float value)
Writes a float parameter.
Definition StateBlob.h:71
Main namespace for the DSPark framework.
T moveTowards(T from, T to, T maxDelta) noexcept
Moves a value toward a target by at most a given distance.
Definition DspMath.h:134
constexpr uint32_t stateId(const char(&tag)[5]) noexcept
Builds a FOURCC processor id, e.g. dspark::stateId("COMP").
Definition StateBlob.h:651
Describes the audio environment for a DSP processor.
Definition AudioSpec.h:37
constexpr bool isValid() const noexcept
Checks if the specification contains valid, processable parameters.
Definition AudioSpec.h:71
int numChannels
Number of audio channels (e.g., 1 = mono, 2 = stereo).
Definition AudioSpec.h:58
double sampleRate
Sample rate in Hz.
Definition AudioSpec.h:45