188 static_assert(std::atomic<double>::is_always_lock_free,
189 "PhaseVocoderEngine requires lock-free std::atomic<double>");
190 static_assert(std::atomic<unsigned>::is_always_lock_free,
191 "PhaseVocoderEngine requires lock-free std::atomic<unsigned>");
192 static_assert(std::atomic<bool>::is_always_lock_free,
193 "PhaseVocoderEngine requires lock-free std::atomic<bool>");
281 bool resampleCompensation,
bool separationSupport =
false,
282 bool transientLockedHop =
false,
283 bool spectralFluxDetector =
false)
285 if (!(sampleRate > 0.0) || !std::isfinite(sampleRate) || numChannels < 1
286 || (
fftSize & (
fftSize - 1)) != 0 || fftSize < 256 || fftSize > (1 << 20))
291 sampleRate_ = sampleRate;
292 numChannels_ = numChannels;
294 numBins_ = fftSize_ / 2 + 1;
295 synthHop_ = fftSize_ / 4;
296 ringMask_ = fftSize_ - 1;
297 resampleCompensation_ = resampleCompensation;
299 accumSize_ = fftSize_ * 4;
300 accumMask_ = accumSize_ - 1;
302 fft_ = std::make_unique<FFTReal<T>>(
static_cast<size_t>(fftSize_));
304 window_.resize(
static_cast<size_t>(fftSize_));
306 for (
auto& w : window_) w = std::sqrt(w);
308 const int nCh = numChannels_;
309 inputRing_.assign(
static_cast<size_t>(nCh), {});
310 accum_.assign(
static_cast<size_t>(nCh), {});
311 for (
int ch = 0; ch < nCh; ++ch)
313 inputRing_[
static_cast<size_t>(ch)].assign(
static_cast<size_t>(fftSize_), T(0));
314 accum_[
static_cast<size_t>(ch)].assign(
static_cast<size_t>(accumSize_), T(0));
317 fftIn_.resize(
static_cast<size_t>(fftSize_));
318 spec_.resize(
static_cast<size_t>(fftSize_ + 2));
319 fftResult_.resize(
static_cast<size_t>(fftSize_));
320 prevAnalysis_.resize(
static_cast<size_t>(fftSize_ + 2));
321 prevSynth_.resize(
static_cast<size_t>(fftSize_ + 2));
324 cepsTime_.resize(
static_cast<size_t>(fftSize_));
325 cepsSpec_.resize(
static_cast<size_t>(fftSize_ + 2));
326 envLog_.resize(
static_cast<size_t>(numBins_));
327 formantGain_.resize(
static_cast<size_t>(numBins_));
328 mag_.resize(
static_cast<size_t>(numBins_));
329 prevMag_.resize(
static_cast<size_t>(numBins_));
330 rotRe_.resize(
static_cast<size_t>(numBins_));
331 rotIm_.resize(
static_cast<size_t>(numBins_));
332 peakBin_.resize(
static_cast<size_t>(numBins_ / 2 + 2));
341 fluxDetectorEnabled_ = spectralFluxDetector;
342 if (fluxDetectorEnabled_)
344 buildOnsetFilterBank();
345 bandCur_.assign(
static_cast<size_t>(numBands_), T(0));
346 bandPrev_.assign(
static_cast<size_t>(numBands_), T(0));
347 bandMaxPrev_.assign(
static_cast<size_t>(numBands_), T(0));
348 fluxHist_.assign(
static_cast<size_t>(kFluxWindow), 0.0);
349 fluxScratch_.assign(
static_cast<size_t>(kFluxWindow), 0.0);
356 odfScale_ =
static_cast<T
>(kOdfRefFrame /
static_cast<double>(fftSize_));
367 bandMaxPrev_.clear();
369 fluxScratch_.clear();
376 lockEnabled_ = transientLockedHop;
377 lockFrames_ = lockEnabled_ ? (fftSize_ + synthHop_ - 1) / synthHop_ : 0;
390 separationEnabled_ = separationSupport;
391 if (separationEnabled_)
393 const double hopSeconds =
static_cast<double>(synthHop_) / sampleRate_;
394 const double binHz = sampleRate_ /
static_cast<double>(fftSize_);
395 timeMedian_ = makeOdd(
static_cast<int>(std::lround(0.2 / hopSeconds)), 3, 31);
396 freqMedian_ = makeOdd(
static_cast<int>(std::lround(500.0 / binHz)), 3, 63);
397 freqMedian_ = std::min(freqMedian_, makeOdd(numBins_, 3, 63));
399 magHist_.assign(
static_cast<size_t>(timeMedian_) *
static_cast<size_t>(numBins_), T(0));
400 medianScratch_.resize(
static_cast<size_t>(std::max(timeMedian_, freqMedian_)));
401 maskHarm_.resize(
static_cast<size_t>(numBins_));
402 maskPerc_.resize(
static_cast<size_t>(numBins_));
403 specRaw_.resize(
static_cast<size_t>(fftSize_ + 2));
410 medianScratch_.clear();
424 if (!prepared_)
return;
425 for (
auto& r : inputRing_) std::fill(r.begin(), r.end(), T(0));
426 for (
auto& a : accum_) std::fill(a.begin(), a.end(), T(0));
427 std::fill(prevAnalysis_.begin(), prevAnalysis_.end(), T(0));
428 std::fill(prevSynth_.begin(), prevSynth_.end(), T(0));
429 std::fill(prevMag_.begin(), prevMag_.end(), T(0));
430 std::fill(magHist_.begin(), magHist_.end(), T(0));
434 std::fill(bandPrev_.begin(), bandPrev_.end(), T(0));
435 std::fill(bandMaxPrev_.begin(), bandMaxPrev_.end(), T(0));
436 std::fill(fluxHist_.begin(), fluxHist_.end(), 0.0);
444 writeHead_ =
static_cast<int64_t
>(accumSize_);
446 adoptParamsIfDirty();
447 stActive_ = std::clamp(targetSemitones_, -12.0, 12.0);
448 ratioActive_ = std::exp2(stActive_ / 12.0);
451 analysisHop_ = computeAnalysisHop();
467 seq_.fetch_add(1, std::memory_order_acq_rel);
472 std::atomic_thread_fence(std::memory_order_release);
473 stgSemitones_.store(p.targetSemitones, std::memory_order_relaxed);
474 stgTransient_.store(p.transientPreserve, std::memory_order_relaxed);
475 stgFormant_.store(p.formantPreserve, std::memory_order_relaxed);
476 stgPhaseLock_.store(p.phaseLock, std::memory_order_relaxed);
477 stgSplit_.store(p.percussiveSplit, std::memory_order_relaxed);
478 seq_.fetch_add(1, std::memory_order_release);
479 dirty_.store(
true, std::memory_order_release);
487 return analysisHop_ - inputSinceHop_;
492 void pushInput(
int ch,
const T* src,
int count)
noexcept
494 auto& ring = inputRing_[
static_cast<size_t>(ch)];
496 for (
int k = 0; k < count; ++k)
498 ring[
static_cast<size_t>(wp)] = src[k];
499 wp = (wp + 1) & ringMask_;
513 inputPos_ = (inputPos_ + count) & ringMask_;
514 inputSinceHop_ += count;
515 if (inputSinceHop_ >= analysisHop_)
518 processHop(numActiveChannels);
523 [[nodiscard]]
double activeRatio() const noexcept {
return ratioActive_; }
540 [[nodiscard]]
const T*
olaData(
int ch)
const noexcept
542 return accum_[
static_cast<size_t>(ch)].data();
546 [[nodiscard]] int64_t
olaMask() const noexcept
548 return static_cast<int64_t
>(accumMask_);
552 [[nodiscard]]
int olaSize() const noexcept {
return accumSize_; }
555 [[nodiscard]] int64_t
writeHead() const noexcept {
return writeHead_; }
558 [[nodiscard]]
int fftSize() const noexcept {
return prepared_ ? fftSize_ : 0; }
561 [[nodiscard]]
int synthHop() const noexcept {
return synthHop_; }
564 static constexpr double kTwoPi = 2.0 * std::numbers::pi;
571 static constexpr int kSeqlockMaxAttempts = 3;
578 static constexpr double kSeparationFactor = 2.0;
584 static constexpr double kOnsetFMin = 27.5;
585 static constexpr double kOnsetFMax = 16000.0;
586 static constexpr int kOnsetBandsPerOctave = 24;
590 static constexpr double kOdfRefFrame = 2048.0;
595 static constexpr int kFluxWindow = 12;
603 static constexpr double kFluxDelta = 0.02;
612 void adoptParamsIfDirty() noexcept
614 if (!dirty_.exchange(
false, std::memory_order_acquire))
return;
619 bool adopted =
false;
621 bool tr =
true, fo =
false, pl =
true, sp =
false;
622 for (
int attempt = 0; attempt < kSeqlockMaxAttempts; ++attempt)
624 const unsigned s0 = seq_.load(std::memory_order_acquire);
625 if ((s0 & 1u) != 0u)
continue;
626 st = stgSemitones_.load(std::memory_order_relaxed);
627 tr = stgTransient_.load(std::memory_order_relaxed);
628 fo = stgFormant_.load(std::memory_order_relaxed);
629 pl = stgPhaseLock_.load(std::memory_order_relaxed);
630 sp = stgSplit_.load(std::memory_order_relaxed);
635 std::atomic_thread_fence(std::memory_order_acquire);
636 if (s0 == seq_.load(std::memory_order_relaxed)) { adopted =
true;
break; }
642 dirty_.store(
true, std::memory_order_release);
647 targetSemitones_ = st;
651 splitOn_ = sp && separationEnabled_;
655 [[nodiscard]]
static double princArg(
double x)
noexcept
657 return x - kTwoPi * std::round(x / kTwoPi);
682 [[nodiscard]]
int computeAnalysisHop() noexcept
688 debt_ +=
static_cast<double>(synthHop_) / ratioActive_
689 -
static_cast<double>(synthHop_);
690 ideal =
static_cast<double>(synthHop_) + hopCarry_;
694 const double cap = 0.5 *
static_cast<double>(synthHop_);
695 const double repay = std::clamp(debt_, -cap, cap);
697 ideal =
static_cast<double>(synthHop_) / ratioActive_ + repay + hopCarry_;
699 int hop =
static_cast<int>(ideal);
700 hop = std::clamp(hop, 1, fftSize_);
701 hopCarry_ = ideal -
static_cast<double>(hop);
719 void armTransientLock(
bool transient)
noexcept
721 if (!lockEnabled_)
return;
722 if (!transient)
return;
723 if (firstFrame_)
return;
724 if (lockLeft_ > 0)
return;
725 if (std::abs(debt_) >= 1.0)
return;
726 lockLeft_ = lockFrames_;
730 void processHop(
int nCh)
noexcept
735 adoptParamsIfDirty();
738 const double stTarget = std::clamp(targetSemitones_, -12.0, 12.0);
739 stActive_ += std::clamp(stTarget - stActive_, -0.5, 0.5);
740 ratioActive_ = std::exp2(stActive_ / 12.0);
745 const bool transient = detectTransient();
746 armTransientLock(transient);
747 buildRotations(transient);
751 if (separationEnabled_)
753 pushMagnitudeHistory();
754 if (splitOn_) updateSeparationMasks();
759 std::copy(spec_.begin(), spec_.end(), prevAnalysis_.begin());
760 std::copy(mag_.begin(), mag_.end(), prevMag_.begin());
763 synthesizeChannel(0,
true);
764 for (
int ch = 1; ch < nCh; ++ch)
767 synthesizeChannel(ch,
false);
770 analysisHop_ = computeAnalysisHop();
771 writeHead_ += synthHop_;
776 void analyzeChannel(
int ch)
noexcept
778 const auto& ring = inputRing_[
static_cast<size_t>(ch)];
779 const int readPos = inputPos_;
780 for (
int k = 0; k < fftSize_; ++k)
782 const int idx = (readPos + k) & ringMask_;
783 fftIn_[
static_cast<size_t>(k)] = ring[
static_cast<size_t>(idx)]
784 * window_[
static_cast<size_t>(k)];
786 fft_->forward(fftIn_.data(), spec_.data());
790 for (
int k = 0; k < numBins_; ++k)
792 const T re = spec_[
static_cast<size_t>(2 * k)];
793 const T im = spec_[
static_cast<size_t>(2 * k + 1)];
794 mag_[
static_cast<size_t>(k)] = std::sqrt(re * re + im * im);
806 [[nodiscard]]
bool detectTransient() noexcept
808 return fluxDetectorEnabled_ ? detectTransientByFlux()
809 : detectTransientByEnergy();
822 [[nodiscard]]
bool detectTransientByEnergy() noexcept
825 for (
int k = 0; k < numBins_; ++k)
827 const double m =
static_cast<double>(mag_[
static_cast<size_t>(k)]);
830 const double prevEnv = onsetEnv_;
831 onsetEnv_ = std::max(energy, onsetEnv_ * 0.7);
834 return firstFrame_ || (energy > 4.0 * prevEnv && energy > 1e-12);
854 [[nodiscard]]
bool detectTransientByFlux() noexcept
856 const double flux = computeSpectralFlux();
862 if (fluxFilled_ >= kFluxWindow)
864 std::copy(fluxHist_.begin(), fluxHist_.end(), fluxScratch_.begin());
865 const auto mid = fluxScratch_.begin() + kFluxWindow / 2;
866 std::nth_element(fluxScratch_.begin(), mid, fluxScratch_.end());
867 rise = (flux - *mid) >= kFluxDelta;
870 fluxHist_[
static_cast<size_t>(fluxPos_)] = flux;
871 fluxPos_ = (fluxPos_ + 1) % kFluxWindow;
872 if (fluxFilled_ < kFluxWindow) ++fluxFilled_;
876 return firstFrame_ || rise;
881 [[nodiscard]]
double computeSpectralFlux() noexcept
886 for (
int b = 0; b < numBands_; ++b)
888 const double d =
static_cast<double>(bandCur_[
static_cast<size_t>(b)])
889 -
static_cast<double>(bandMaxPrev_[
static_cast<size_t>(b)]);
890 if (d > 0.0) flux += d;
892 flux /=
static_cast<double>(std::max(1, numBands_));
896 bandPrev_ = bandCur_;
904 void filterLogBands() noexcept
906 for (
int b = 0; b < numBands_; ++b)
908 const int start = fbStart_[
static_cast<size_t>(b)];
909 const int off = fbOffset_[
static_cast<size_t>(b)];
910 const int cnt = fbCount_[
static_cast<size_t>(b)];
912 for (
int i = 0; i < cnt; ++i)
914 const int k = start + i;
915 if (k >= 0 && k < numBins_)
916 acc += mag_[
static_cast<size_t>(k)]
917 * fbWeights_[
static_cast<size_t>(off + i)];
919 bandCur_[
static_cast<size_t>(b)] = std::log10(acc * odfScale_ + T(1));
924 void maxFilterBands() noexcept
926 for (
int b = 0; b < numBands_; ++b)
928 T mx = bandPrev_[
static_cast<size_t>(b)];
929 if (b > 0) mx = std::max(mx, bandPrev_[
static_cast<size_t>(b - 1)]);
930 if (b + 1 < numBands_) mx = std::max(mx, bandPrev_[
static_cast<size_t>(b + 1)]);
931 bandMaxPrev_[
static_cast<size_t>(b)] = mx;
944 void buildOnsetFilterBank()
952 const double binHz = sampleRate_ /
static_cast<double>(fftSize_);
953 const double fMax = std::min(kOnsetFMax, sampleRate_ * 0.5 * 0.999);
955 std::vector<int> centres;
956 for (
int i = 0; ; ++i)
958 const double f = kOnsetFMin * std::pow(2.0,
static_cast<double>(i)
959 /
static_cast<double>(kOnsetBandsPerOctave));
961 int bin =
static_cast<int>(std::lround(f / binHz));
962 bin = std::clamp(bin, 0, numBins_ - 1);
963 if (centres.empty() || bin > centres.back())
964 centres.push_back(bin);
967 for (
size_t j = 1; j + 1 < centres.size(); ++j)
969 const int lo = centres[j - 1];
970 const int ce = centres[j];
971 const int hi = centres[j + 1];
972 if (!(lo < ce && ce < hi))
continue;
974 fbStart_.push_back(lo);
975 fbOffset_.push_back(
static_cast<int>(fbWeights_.size()));
976 for (
int k = lo; k <= hi; ++k)
978 const T wv = (k <= ce)
979 ?
static_cast<T
>(
static_cast<double>(k - lo)
980 /
static_cast<double>(ce - lo))
981 :
static_cast<T
>(
static_cast<double>(hi - k)
982 /
static_cast<double>(hi - ce));
983 fbWeights_.push_back(wv);
987 for (
int b = 0; b < numBands_; ++b)
989 const int off = fbOffset_[
static_cast<size_t>(b)];
990 const int nextOff = (b + 1 < numBands_)
991 ? fbOffset_[
static_cast<size_t>(b + 1)]
992 :
static_cast<int>(fbWeights_.size());
993 fbCount_.push_back(nextOff - off);
998 if (numBands_ < 1) numBands_ = 0;
1008 void buildRotations(
bool transient)
noexcept
1012 std::fill(rotRe_.begin(), rotRe_.end(), T(1));
1013 std::fill(rotIm_.begin(), rotIm_.end(), T(0));
1019 buildPerBinRotations();
1025 for (
int k = 0; k < numBins_; ++k)
1026 maxMag = std::max(maxMag, mag_[
static_cast<size_t>(k)]);
1029 if (maxMag > T(1e-9))
1031 const T floorMag = maxMag * T(1e-4);
1037 const int last = numBins_ - 3;
1038 for (
int k = 2; k <= last; ++k)
1040 const T m = mag_[
static_cast<size_t>(k)];
1041 if (m < floorMag)
continue;
1042 if (m > mag_[
static_cast<size_t>(k - 1)] && m >= mag_[
static_cast<size_t>(k + 1)]
1043 && m > mag_[
static_cast<size_t>(k - 2)] && m >= mag_[
static_cast<size_t>(k + 2)])
1045 peakBin_[
static_cast<size_t>(numPeaks_++)] = k;
1053 std::fill(rotRe_.begin(), rotRe_.end(), T(1));
1054 std::fill(rotIm_.begin(), rotIm_.end(), T(0));
1061 const double Ra =
static_cast<double>(analysisHop_);
1062 const double Rs =
static_cast<double>(synthHop_);
1063 const double binW = kTwoPi /
static_cast<double>(fftSize_);
1065 int regionStart = 0;
1066 for (
int p = 0; p < numPeaks_; ++p)
1068 const int bin = peakBin_[
static_cast<size_t>(p)];
1069 const double re =
static_cast<double>(spec_[
static_cast<size_t>(2 * bin)]);
1070 const double im =
static_cast<double>(spec_[
static_cast<size_t>(2 * bin + 1)]);
1071 const double pre =
static_cast<double>(prevAnalysis_[
static_cast<size_t>(2 * bin)]);
1072 const double pim =
static_cast<double>(prevAnalysis_[
static_cast<size_t>(2 * bin + 1)]);
1074 double rotR = 1.0, rotI = 0.0;
1077 if (prevMag_[
static_cast<size_t>(bin)] > T(0.1) * mag_[
static_cast<size_t>(bin)])
1080 const double deltaPhi = std::atan2(im * pre - re * pim, re * pre + im * pim);
1081 const double omegaK = binW *
static_cast<double>(bin);
1082 const double omegaInst = omegaK + princArg(deltaPhi - omegaK * Ra) / Ra;
1084 const double psiPrev = std::atan2(
1085 static_cast<double>(prevSynth_[
static_cast<size_t>(2 * bin + 1)]),
1086 static_cast<double>(prevSynth_[
static_cast<size_t>(2 * bin)]));
1087 const double phi = std::atan2(im, re);
1088 const double theta = psiPrev + Rs * omegaInst - phi;
1089 rotR = std::cos(theta);
1090 rotI = std::sin(theta);
1094 int regionEnd = numBins_;
1095 if (p + 1 < numPeaks_)
1097 const int nextBin = peakBin_[
static_cast<size_t>(p + 1)];
1098 int valley = bin + 1;
1099 T valleyMag = mag_[
static_cast<size_t>(valley)];
1100 for (
int k = bin + 2; k < nextBin; ++k)
1102 if (mag_[
static_cast<size_t>(k)] < valleyMag)
1104 valleyMag = mag_[
static_cast<size_t>(k)];
1108 regionEnd = valley + 1;
1111 for (
int k = regionStart; k < regionEnd; ++k)
1113 rotRe_[
static_cast<size_t>(k)] =
static_cast<T
>(rotR);
1114 rotIm_[
static_cast<size_t>(k)] =
static_cast<T
>(rotI);
1116 regionStart = regionEnd;
1120 rotRe_[0] = T(1); rotIm_[0] = T(0);
1121 rotRe_[
static_cast<size_t>(numBins_ - 1)] = T(1); rotIm_[
static_cast<size_t>(numBins_ - 1)] = T(0);
1133 void buildPerBinRotations() noexcept
1135 const double Ra =
static_cast<double>(analysisHop_);
1136 const double Rs =
static_cast<double>(synthHop_);
1137 const double binW = kTwoPi /
static_cast<double>(fftSize_);
1140 rotRe_[0] = T(1); rotIm_[0] = T(0);
1141 rotRe_[
static_cast<size_t>(numBins_ - 1)] = T(1);
1142 rotIm_[
static_cast<size_t>(numBins_ - 1)] = T(0);
1144 for (
int k = 1; k < numBins_ - 1; ++k)
1146 double rotR = 1.0, rotI = 0.0;
1148 if (prevMag_[
static_cast<size_t>(k)] > T(0.1) * mag_[
static_cast<size_t>(k)])
1150 const double re =
static_cast<double>(spec_[
static_cast<size_t>(2 * k)]);
1151 const double im =
static_cast<double>(spec_[
static_cast<size_t>(2 * k + 1)]);
1152 const double pre =
static_cast<double>(prevAnalysis_[
static_cast<size_t>(2 * k)]);
1153 const double pim =
static_cast<double>(prevAnalysis_[
static_cast<size_t>(2 * k + 1)]);
1155 const double deltaPhi = std::atan2(im * pre - re * pim, re * pre + im * pim);
1156 const double omegaK = binW *
static_cast<double>(k);
1157 const double omegaInst = omegaK + princArg(deltaPhi - omegaK * Ra) / Ra;
1159 const double psiPrev = std::atan2(
1160 static_cast<double>(prevSynth_[
static_cast<size_t>(2 * k + 1)]),
1161 static_cast<double>(prevSynth_[
static_cast<size_t>(2 * k)]));
1162 const double phi = std::atan2(im, re);
1163 const double theta = psiPrev + Rs * omegaInst - phi;
1164 rotR = std::cos(theta);
1165 rotI = std::sin(theta);
1167 rotRe_[
static_cast<size_t>(k)] =
static_cast<T
>(rotR);
1168 rotIm_[
static_cast<size_t>(k)] =
static_cast<T
>(rotI);
1173 [[nodiscard]]
static int makeOdd(
int v,
int lo,
int hi)
noexcept
1175 v = std::clamp(v, lo, hi);
1176 if ((v & 1) == 0) --v;
1177 return std::max(v, lo | 1);
1181 void pushMagnitudeHistory() noexcept
1183 T* slot = magHist_.data() +
static_cast<size_t>(histPos_) *
static_cast<size_t>(numBins_);
1184 std::copy(mag_.begin(), mag_.end(), slot);
1185 histPos_ = (histPos_ + 1) % timeMedian_;
1186 if (histFilled_ < timeMedian_) ++histFilled_;
1206 void updateSeparationMasks() noexcept
1208 const int nHist = histFilled_;
1209 const int half = freqMedian_ / 2;
1211 for (
int k = 0; k < numBins_; ++k)
1214 for (
int f = 0; f < nHist; ++f)
1215 medianScratch_[
static_cast<size_t>(f)] =
1216 magHist_[
static_cast<size_t>(f) *
static_cast<size_t>(numBins_)
1217 +
static_cast<size_t>(k)];
1218 auto mid = medianScratch_.begin() + nHist / 2;
1219 std::nth_element(medianScratch_.begin(), mid,
1220 medianScratch_.begin() + nHist);
1221 const double h =
static_cast<double>(*mid);
1225 const int lo = std::max(0, k - half);
1226 const int hi = std::min(numBins_ - 1, k + half);
1227 const int n = hi - lo + 1;
1228 for (
int j = 0; j < n; ++j)
1229 medianScratch_[
static_cast<size_t>(j)] = mag_[
static_cast<size_t>(lo + j)];
1230 auto midF = medianScratch_.begin() + n / 2;
1231 std::nth_element(medianScratch_.begin(), midF, medianScratch_.begin() + n);
1232 const double p =
static_cast<double>(*midF);
1243 const bool percussive = (p > kSeparationFactor * h);
1244 maskHarm_[
static_cast<size_t>(k)] = percussive ? T(0) : T(1);
1245 maskPerc_[
static_cast<size_t>(k)] = percussive ? T(1) : T(0);
1251 void synthesizeChannel(
int ch,
bool isReference)
noexcept
1256 std::copy(spec_.begin(), spec_.end(), specRaw_.begin());
1260 for (
int k = 1; k < numBins_ - 1; ++k)
1262 const T re = spec_[
static_cast<size_t>(2 * k)];
1263 const T im = spec_[
static_cast<size_t>(2 * k + 1)];
1264 const T rr = rotRe_[
static_cast<size_t>(k)];
1265 const T ri = rotIm_[
static_cast<size_t>(k)];
1266 spec_[
static_cast<size_t>(2 * k)] = re * rr - im * ri;
1267 spec_[
static_cast<size_t>(2 * k + 1)] = re * ri + im * rr;
1275 if (formantOn_ && std::abs(ratioActive_ - 1.0) > 1e-6)
1278 computeFormantGains();
1279 for (
int k = 1; k < numBins_ - 1; ++k)
1281 const T g = formantGain_[
static_cast<size_t>(k)];
1282 spec_[
static_cast<size_t>(2 * k)] *= g;
1283 spec_[
static_cast<size_t>(2 * k + 1)] *= g;
1290 if (resampleCompensation_ && ratioActive_ > 1.0)
1292 const int cut =
static_cast<int>(
static_cast<double>(fftSize_ / 2) / ratioActive_);
1293 const int taperStart = std::max(1, cut - 4);
1294 for (
int k = taperStart; k < numBins_; ++k)
1298 g =
static_cast<T
>(cut - k + 1) /
static_cast<T
>(cut - taperStart + 1);
1299 spec_[
static_cast<size_t>(2 * k)] *= g;
1300 spec_[
static_cast<size_t>(2 * k + 1)] *= g;
1305 std::copy(spec_.begin(), spec_.end(), prevSynth_.begin());
1313 for (
int k = 0; k < numBins_; ++k)
1315 const T mh = maskHarm_[
static_cast<size_t>(k)];
1316 const T mp = maskPerc_[
static_cast<size_t>(k)];
1317 const auto re =
static_cast<size_t>(2 * k);
1318 const auto im =
static_cast<size_t>(2 * k + 1);
1319 spec_[re] = spec_[re] * mh + specRaw_[re] * mp;
1320 spec_[im] = spec_[im] * mh + specRaw_[im] * mp;
1324 fft_->inverse(spec_.data(), fftResult_.data());
1337 void computeFormantGains() noexcept
1340 for (
int k = 0; k < numBins_; ++k)
1342 const T re = spec_[
static_cast<size_t>(2 * k)];
1343 const T im = spec_[
static_cast<size_t>(2 * k + 1)];
1344 cepsTime_[
static_cast<size_t>(k)] =
1345 std::log(std::sqrt(re * re + im * im) + T(1e-9));
1347 for (
int k = numBins_; k < fftSize_; ++k)
1348 cepsTime_[
static_cast<size_t>(k)] =
1349 cepsTime_[
static_cast<size_t>(fftSize_ - k)];
1351 fft_->forward(cepsTime_.data(), cepsSpec_.data());
1356 const int qCut = std::max(8,
static_cast<int>(0.001 * sampleRate_));
1357 const int keep = std::min(qCut, numBins_ - 1);
1358 for (
int k = keep + 1; k < numBins_; ++k)
1360 cepsSpec_[
static_cast<size_t>(2 * k)] = T(0);
1361 cepsSpec_[
static_cast<size_t>(2 * k + 1)] = T(0);
1364 fft_->inverse(cepsSpec_.data(), cepsTime_.data());
1365 for (
int k = 0; k < numBins_; ++k)
1366 envLog_[
static_cast<size_t>(k)] = cepsTime_[
static_cast<size_t>(k)];
1369 for (
int k = 0; k < numBins_; ++k)
1371 const double pos = std::min(
static_cast<double>(k) * ratioActive_,
1372 static_cast<double>(numBins_ - 1));
1373 const auto i0 =
static_cast<int>(pos);
1374 const auto frac =
static_cast<T
>(pos - i0);
1375 const T target = envLog_[
static_cast<size_t>(i0)]
1376 + (envLog_[
static_cast<size_t>(std::min(i0 + 1, numBins_ - 1))]
1377 - envLog_[
static_cast<size_t>(i0)]) * frac;
1378 const T delta = std::clamp(target - envLog_[
static_cast<size_t>(k)],
1380 formantGain_[
static_cast<size_t>(k)] = std::exp(delta);
1385 void synthesizeTail(
int ch)
noexcept
1389 auto& acc = accum_[
static_cast<size_t>(ch)];
1390 const auto m =
static_cast<int64_t
>(accumMask_);
1394 for (
int k = fftSize_ - synthHop_; k < fftSize_; ++k)
1395 acc[
static_cast<size_t>((writeHead_ + k) & m)] = T(0);
1397 constexpr T kNorm = T(0.5);
1398 for (
int k = 0; k < fftSize_; ++k)
1400 const auto idx =
static_cast<size_t>((writeHead_ + k) & m);
1401 acc[idx] += fftResult_[
static_cast<size_t>(k)]
1402 * window_[
static_cast<size_t>(k)] * kNorm;
1407 double sampleRate_ = 48000.0;
1408 int numChannels_ = 0;
1409 bool prepared_ =
false;
1410 bool resampleCompensation_ =
false;
1412 int fftSize_ = 2048;
1413 int numBins_ = 1025;
1414 int synthHop_ = 512;
1415 int ringMask_ = 2047;
1416 int accumSize_ = 8192;
1417 int accumMask_ = 8191;
1419 std::unique_ptr<FFTReal<T>> fft_;
1420 std::vector<T> window_;
1422 std::vector<std::vector<T>> inputRing_;
1423 std::vector<std::vector<T>> accum_;
1425 std::vector<T> fftIn_, spec_, fftResult_;
1426 std::vector<T> prevAnalysis_, prevSynth_;
1427 std::vector<T> mag_, prevMag_;
1428 std::vector<T> cepsTime_, cepsSpec_;
1429 std::vector<T> envLog_, formantGain_;
1430 std::vector<T> rotRe_, rotIm_;
1431 std::vector<int> peakBin_;
1440 bool fluxDetectorEnabled_ =
false;
1441 double onsetEnv_ = 0.0;
1442 std::vector<int> fbStart_, fbOffset_, fbCount_;
1443 std::vector<T> fbWeights_;
1446 std::vector<T> bandCur_, bandPrev_, bandMaxPrev_;
1447 std::vector<double> fluxHist_;
1448 std::vector<double> fluxScratch_;
1450 int fluxFilled_ = 0;
1453 bool lockEnabled_ =
false;
1454 int lockFrames_ = 0;
1459 bool separationEnabled_ =
false;
1460 int timeMedian_ = 0;
1461 int freqMedian_ = 0;
1462 std::vector<T> magHist_;
1463 std::vector<T> medianScratch_;
1464 std::vector<T> maskHarm_, maskPerc_;
1465 std::vector<T> specRaw_;
1467 int histFilled_ = 0;
1470 int64_t writeHead_ = 0;
1474 double stActive_ = 0.0;
1475 double ratioActive_ = 1.0;
1476 double hopCarry_ = 0.0;
1477 int analysisHop_ = 512;
1478 int inputSinceHop_ = 0;
1479 bool firstFrame_ =
true;
1483 double targetSemitones_ = 0.0;
1484 bool transientOn_ =
true;
1485 bool formantOn_ =
false;
1486 bool phaseLockOn_ =
true;
1487 bool splitOn_ =
false;
1491 std::atomic<double> stgSemitones_ { 0.0 };
1492 std::atomic<bool> stgTransient_ {
true };
1493 std::atomic<bool> stgFormant_ {
false };
1494 std::atomic<bool> stgPhaseLock_ {
true };
1495 std::atomic<bool> stgSplit_ {
false };
1496 std::atomic<unsigned> seq_ { 0 };
1497 std::atomic<bool> dirty_ {
false };