226#include "../Analysis/PitchDetector.h"
227#include "../Core/AudioBuffer.h"
228#include "../Core/AudioSpec.h"
229#include "../Core/DenormalGuard.h"
230#include "../Core/DspMath.h"
231#include "../Core/StateBlob.h"
232#include "../Music/HarmonyConstants.h"
233#include "PitchShifter.h"
251template <FloatType T>
257 static_assert(std::atomic<T>::is_always_lock_free,
258 "audio-thread stores must not lock");
259 static_assert(std::atomic<std::uint32_t>::is_always_lock_free,
260 "audio-thread stores must not lock");
290 const int frame = fftSize > 0 ? sanitizeFrame(fftSize)
298 shifter_.prepare(spec, frame);
302 latency_ = shifter_.getLatency();
316 if (!prepared_)
return;
321 publishedCorrection_ = 0.0;
322 shifter_.setSemitones(T(0));
325 smoothingForMs_ = -1.0;
326 warmupRemaining_ = detector_.getWindowSize();
341 void setScale(std::uint16_t scaleBitmask,
int rootPitchClass)
noexcept
343 const auto mask =
static_cast<std::uint32_t
>(scaleBitmask & 0x0FFFu);
344 const auto root =
static_cast<std::uint32_t
>((rootPitchClass % 12 + 12) % 12);
350 scaleWord_.store(absolute | (root << 12) | (mask << 16),
351 std::memory_order_relaxed);
364 if (!std::isfinite(ms))
return;
365 retuneSpeedMs_.store(std::clamp(ms, T(0), T(kMaxRetuneMs)),
366 std::memory_order_relaxed);
373 formantPreserve_.store(on, std::memory_order_relaxed);
379 return static_cast<std::uint16_t
>(
380 (scaleWord_.load(std::memory_order_relaxed) >> 16) & 0x0FFFu);
386 return static_cast<int>(
387 (scaleWord_.load(std::memory_order_relaxed) >> 12) & 0x0Fu);
393 return retuneSpeedMs_.load(std::memory_order_relaxed);
399 return formantPreserve_.load(std::memory_order_relaxed);
403 [[nodiscard]] std::vector<uint8_t>
getState()
const
420 setScale(
static_cast<std::uint16_t
>(r.
read(
"scaleMask", int32_t(0x0FFF)) & 0x0FFF),
421 r.
read(
"root", int32_t(0)));
432 return prepared_ ? latency_ : 0;
439 return prepared_ ? frameSize_ : 0;
455 if (!prepared_)
return;
456 const int numSamples = buffer.getNumSamples();
457 if (numSamples <= 0)
return;
464 const std::uint32_t absoluteMask =
465 scaleWord_.load(std::memory_order_relaxed) & 0x0FFFu;
467 const double retuneMs =
static_cast<double>(
468 retuneSpeedMs_.load(std::memory_order_relaxed));
469 if (retuneMs != smoothingForMs_) rebuildSmoothing(retuneMs);
471 const bool formant = formantPreserve_.load(std::memory_order_relaxed);
472 if (formant != publishedFormant_)
476 shifter_.setFormantPreserve(formant);
477 publishedFormant_ = formant;
480 const T*
const detectionInput =
481 buffer.getNumChannels() > 0 ? buffer.getChannel(0) :
nullptr;
484 while (offset < numSamples)
495 if (controlPhase_ == 0) advanceCorrection(absoluteMask);
498 std::min(numSamples - offset, kControlInterval - controlPhase_);
500 if (detectionInput !=
nullptr)
501 detector_.pushSamples(std::span<const T>(
502 detectionInput + offset,
static_cast<std::size_t
>(chunk)));
504 if (warmupRemaining_ > 0)
505 warmupRemaining_ = std::max(0, warmupRemaining_ - chunk);
507 shifter_.processBlock(buffer.getSubView(offset, chunk));
509 controlPhase_ += chunk;
510 if (controlPhase_ >= kControlInterval) controlPhase_ = 0;
520 static constexpr double kAutoSpanRef = 2048.0;
521 static constexpr double kAutoSpanRate = 48000.0;
522 static constexpr int kAutoMinFrame = 512;
523 static constexpr int kAutoMaxFrame = 32768;
526 [[nodiscard]]
static int automaticFrame(
double sampleRate)
noexcept
528 const double target = sampleRate * (kAutoSpanRef / kAutoSpanRate);
529 int frame = kAutoMinFrame;
530 while (frame < kAutoMaxFrame &&
static_cast<double>(frame) < target)
539 [[nodiscard]]
static int sanitizeFrame(
int requested)
noexcept
541 constexpr int kMax = 1 << 20;
543 while (frame < requested && frame < kMax) frame <<= 1;
550 static constexpr int kControlInterval = 64;
555 static constexpr double kLandingSemitones = 0.001;
558 static constexpr double kMaxRetuneMs = 10000.0;
565 void rebuildSmoothing(
double retuneMs)
noexcept
567 smoothingForMs_ = retuneMs;
568 if (retuneMs <= 0.0 || sampleRate_ <= 0.0)
573 const double tauSamples = retuneMs * 0.001 * sampleRate_;
574 smoothing_ = 1.0 - std::exp(-
static_cast<double>(kControlInterval) / tauSamples);
581 void advanceCorrection(std::uint32_t absoluteMask)
noexcept
583 if (absoluteMask == 0u)
587 else if (warmupRemaining_ == 0)
595 const double frequency =
static_cast<double>(detector_.getFrequencyHz());
596 if (frequency > 0.0 && frequency <= 0.25 * sampleRate_)
598 const double midi = 69.0 + 12.0 * std::log2(frequency / 440.0);
599 if (std::isfinite(midi))
600 target_ = std::clamp(nearestInScale(midi, absoluteMask) - midi,
605 if (smoothing_ >= 1.0)
607 correction_ = target_;
611 correction_ += (target_ - correction_) * smoothing_;
612 if (std::abs(target_ - correction_) < kLandingSemitones)
613 correction_ = target_;
616 if (correction_ != publishedCorrection_)
618 shifter_.setSemitones(
static_cast<T
>(correction_));
619 publishedCorrection_ = correction_;
631 [[nodiscard]]
static double nearestInScale(
double midi,
632 std::uint32_t absoluteMask)
noexcept
634 const int base =
static_cast<int>(std::floor(midi));
636 double bestDistance = 1.0e9;
637 for (
int note = base - 12; note <= base + 13; ++note)
639 const int pitchClass = ((note % 12) + 12) % 12;
640 if ((absoluteMask & (1u << pitchClass)) == 0u)
continue;
641 const double distance = std::abs(
static_cast<double>(note) - midi);
642 if (distance < bestDistance)
644 bestDistance = distance;
648 return static_cast<double>(bestNote);
653 PitchDetector<T> detector_;
654 PitchShifter<T> shifter_;
656 double sampleRate_ = 48000.0;
659 bool prepared_ =
false;
662 double target_ = 0.0;
663 double correction_ = 0.0;
664 double publishedCorrection_ = 0.0;
665 double smoothing_ = 1.0;
666 double smoothingForMs_ = -1.0;
667 bool publishedFormant_ =
false;
668 int controlPhase_ = 0;
669 int warmupRemaining_ = 0;
673 std::atomic<std::uint32_t> scaleWord_ { 0x0FFF0FFFu };
674 std::atomic<T> retuneSpeedMs_ { T(0) };
675 std::atomic<bool> formantPreserve_ {
false };
Non-owning view over audio channel data.
RAII scope guard to disable denormalised (subnormal) floating-point numbers.
Scale-aware monophonic retune over the framework's YIN detector and phase-vocoder shifter.
T getRetuneSpeedMs() const noexcept
std::vector< uint8_t > getState() const
Serializes the parameter state (setup/UI threads; allocates).
void setFormantPreserve(bool on) noexcept
Enables the shifter's cepstral formant preservation (the anti-chipmunk envelope pre-warp)....
int getFrameSize() const noexcept
The analysis frame in effect, in samples: the automatic choice for this rate, or the rounded explicit...
void processBlock(AudioBufferView< T > buffer) noexcept
Processes audio in-place. Pass-through until prepare() succeeds.
bool getFormantPreserve() const noexcept
int getLatency() const noexcept
Reports the signal latency in samples: twice the analysis frame, which is 4096 samples at 44....
void setScale(std::uint16_t scaleBitmask, int rootPitchClass) noexcept
Selects the scale the output snaps to.
bool setState(const uint8_t *data, size_t size)
Restores parameters from a blob (tolerant; rejects foreign ids). A missing field restores its default...
int getRootPitchClass() const noexcept
std::uint16_t getScaleMask() const noexcept
void prepare(const AudioSpec &spec, int fftSize=0)
Allocates the detector and shifter state (setup thread).
void setRetuneSpeedMs(T ms) noexcept
Sets the retune speed: the time constant, in milliseconds, of the glide from the sung pitch to the ta...
void reset() noexcept
Clears all signal state and keeps the parameters (stream owner).
Real-time phase-vocoder pitch shifter (+-12 semitones, stereo-linked).
Tolerant reader: missing keys yield defaults, unknown keys are skipped.
float read(const char *key, float defaultValue) const
Reads a float, or defaultValue when the key is absent.
bool isValid() const noexcept
uint32_t processorId() const noexcept
Serializes key/value parameters into a versioned blob.
std::vector< uint8_t > blob() const
Finalizes and returns the blob.
void write(const char *key, float value)
Writes a float parameter.
std::uint16_t NoteSet
A 12-bit bitmask representing the 12 pitch-classes of the chromatic scale.
constexpr NoteSet scaleAtRoot(NoteSet base, int root) noexcept
Circularly rotate a NoteSet so it becomes rooted at a specific key.
Main namespace for the DSPark framework.
constexpr uint32_t stateId(const char(&tag)[5]) noexcept
Builds a FOURCC processor id, e.g. dspark::stateId("COMP").
Describes the audio environment for a DSP processor.
constexpr bool isValid() const noexcept
Checks if the specification contains valid, processable parameters.
double sampleRate
Sample rate in Hz.