101 static_assert(std::atomic<double>::is_always_lock_free,
102 "StudioVocoder requires lock-free std::atomic<double>");
103 static_assert(std::atomic<unsigned>::is_always_lock_free,
104 "StudioVocoder requires lock-free std::atomic<unsigned>");
105 static_assert(std::atomic<bool>::is_always_lock_free,
106 "StudioVocoder requires lock-free std::atomic<bool>");
129 bool prepare(
double sampleRate,
int numChannels,
int fftSize,
bool resampleCompensation)
131 if (!(sampleRate > 0.0) || !std::isfinite(sampleRate) || numChannels < 1
132 || (
fftSize & (
fftSize - 1)) != 0 || fftSize < 256 || fftSize > (1 << 20))
136 sampleRate_ = sampleRate;
137 numChannels_ = numChannels;
141 resample_ = resampleCompensation;
145 while (No_ < 8192 &&
static_cast<double>(No_) * 1.5 < 0.0213 * sampleRate_) No_ <<= 1;
152 delta_ = std::min(16, std::max(1, N_ / 64));
155 while (ring < 2 * N_ + 2 * D_ + No_) ring <<= 1;
157 ringMask_ = ring - 1;
159 accumMask_ = accumSize_ - 1;
161 fft_ = std::make_unique<FFTReal<T>>(
static_cast<size_t>(N_));
162 fftO_ = std::make_unique<FFTReal<T>>(
static_cast<size_t>(No_));
164 window_.resize(
static_cast<size_t>(N_));
165 for (
int k = 0; k < N_; ++k)
166 window_[
static_cast<size_t>(k)] =
static_cast<T
>(std::sqrt(
167 0.5 - 0.5 * std::cos(2.0 * std::numbers::pi * k / N_)));
168 windowO_.resize(
static_cast<size_t>(No_));
169 for (
int k = 0; k < No_; ++k)
170 windowO_[
static_cast<size_t>(k)] =
static_cast<T
>(
171 0.5 - 0.5 * std::cos(2.0 * std::numbers::pi * k / No_));
173 const auto nb =
static_cast<size_t>(bins_);
174 const auto spec =
static_cast<size_t>(N_ + 2);
175 ring_.assign(
static_cast<size_t>(numChannels_), std::vector<T>(
static_cast<size_t>(ringSize_), T(0)));
176 accum_.assign(
static_cast<size_t>(numChannels_), std::vector<T>(
static_cast<size_t>(accumSize_), T(0)));
177 spec_.assign(
static_cast<size_t>(numChannels_), std::vector<T>(spec, T(0)));
178 prevSpec_.assign(
static_cast<size_t>(numChannels_), std::vector<T>(spec, T(0)));
180 frame_.resize(
static_cast<size_t>(N_));
183 crossRe_.resize(nb); crossIm_.resize(nb);
184 dRe_.resize(nb); dIm_.resize(nb);
189 rotRe_.resize(nb); rotIm_.resize(nb);
191 heapKey_.resize(2 * nb + 2);
192 heapCode_.resize(2 * nb + 2);
193 cepsTime_.resize(
static_cast<size_t>(N_));
194 cepsSpec_.resize(spec);
199 frameO_.resize(
static_cast<size_t>(No_));
200 specO_.resize(
static_cast<size_t>(No_ + 2));
201 powO_.resize(
static_cast<size_t>(No_ / 2 + 1));
202 bandCur_.assign(
static_cast<size_t>(numBands_), T(0));
203 bandPrevMax_.assign(
static_cast<size_t>(numBands_), T(0));
204 bandPrev_.assign(
static_cast<size_t>(numBands_), T(0));
214 if (!prepared_)
return;
215 for (
auto& r : ring_) std::fill(r.begin(), r.end(), T(0));
216 for (
auto& a : accum_) std::fill(a.begin(), a.end(), T(0));
217 for (
auto& s : prevSpec_) std::fill(s.begin(), s.end(), T(0));
218 std::fill(prevMag_.begin(), prevMag_.end(), T(0));
219 std::fill(theta_.begin(), theta_.end(), 0.0);
220 for (
int k = 0; k < bins_; ++k)
221 omPrev_[
static_cast<size_t>(k)] = kTwoPi * k / N_;
223 adoptParamsIfDirty();
224 stActive_ = std::clamp(targetSemitones_, -12.0, 12.0);
225 ratio_ = std::exp2(stActive_ / 12.0);
228 frameStart_ =
static_cast<int64_t
>(accumSize_);
232 aNext_ =
static_cast<double>(Rs_) - 0.5 * N_;
235 cCur_ =
static_cast<double>(frameStart_) + 0.5 * N_ - Rs_;
237 originStream_ =
static_cast<double>(frameStart_) + 0.5 * N_;
238 originRatio_ = ratio_;
239 prevAnalysisStart_ = 0;
242 lockHead_ = lockCount_ = 0;
243 rejoinE_ = 0.0; rejoinC_ = 0.0; rejoinJ_ = 1.0;
244 lastLockS_ = -1e18; lastLockT_ = -1e18;
248 std::fill(bandPrev_.begin(), bandPrev_.end(), T(0));
249 std::fill(bandPrevMax_.begin(), bandPrevMax_.end(), T(0));
250 fluxFill_ = 0; fluxPos_ = 0;
251 fluxPrev_ = 0.0; fluxPrev2_ = 0.0; candPrev_ =
false; candFrame_ = -1;
252 lastOnsetFrame_ = -1000;
253 onsetRing_.fill(-1e18);
262 seq_.fetch_add(1, std::memory_order_acq_rel);
263 std::atomic_thread_fence(std::memory_order_release);
264 stgSemitones_.store(p.targetSemitones, std::memory_order_relaxed);
265 stgTransient_.store(p.transientPreserve, std::memory_order_relaxed);
266 stgFormant_.store(p.formantPreserve, std::memory_order_relaxed);
267 seq_.fetch_add(1, std::memory_order_release);
268 dirty_.store(
true, std::memory_order_release);
277 const int64_t need = analysisStart() + N_ + D_ - inCount_;
278 return static_cast<int>(std::max<int64_t>(0, need));
283 void pushInput(
int ch,
const T* src,
int count)
noexcept
285 auto& r = ring_[
static_cast<size_t>(ch)];
286 int64_t p = inCount_;
287 for (
int k = 0; k < count; ++k, ++p)
288 r[
static_cast<size_t>(p & ringMask_)] = src[k];
300 runDetector(numActiveChannels);
301 if (inCount_ >= analysisStart() + N_ + D_)
302 processFrame(numActiveChannels);
309 steerValue_ = nominalCentre;
326 [[nodiscard]]
double activeRatio() const noexcept {
return ratio_; }
334 return static_cast<double>(frameStart_) + 0.5 * N_;
340 return originStream_ + originRatio_ * (x - origin_);
343 [[nodiscard]]
const T*
olaData(
int ch)
const noexcept {
return accum_[
static_cast<size_t>(ch)].data(); }
344 [[nodiscard]] int64_t
olaMask() const noexcept {
return static_cast<int64_t
>(accumMask_); }
345 [[nodiscard]]
int olaSize() const noexcept {
return accumSize_; }
347 [[nodiscard]] int64_t
writeHead() const noexcept {
return frameStart_; }
348 [[nodiscard]]
int fftSize() const noexcept {
return prepared_ ? N_ : 0; }
349 [[nodiscard]]
int synthHop() const noexcept {
return Rs_; }
351 [[nodiscard]]
int lookahead() const noexcept {
return D_; }
353 [[nodiscard]] int64_t
inputCount() const noexcept {
return inCount_; }
356 static constexpr double kTwoPi = 2.0 * std::numbers::pi;
357 static constexpr int kSeqlockMaxAttempts = 3;
358 static constexpr int kMaxLocks = 32;
361 static constexpr double kHeapFloorDb = 100.0;
364 static constexpr T kRiseRatio =
static_cast<T
>(1.4142135623730951);
366 static constexpr double kResetCover = 0.375;
368 static constexpr double kFluxDelta = 0.025;
369 static constexpr int kFluxWindow = 12;
371 struct Lock {
double s, target, h; };
373 [[nodiscard]]
static double princArg(
double x)
noexcept
375 return x - kTwoPi * std::round(x / kTwoPi);
378 [[nodiscard]] int64_t analysisStart() const noexcept
380 return static_cast<int64_t
>(std::llround(aNext_)) - N_ / 2;
387 void adoptParamsIfDirty() noexcept
389 if (!dirty_.exchange(
false, std::memory_order_acquire))
return;
390 for (
int attempt = 0; attempt < kSeqlockMaxAttempts; ++attempt)
392 const unsigned s0 = seq_.load(std::memory_order_acquire);
393 if ((s0 & 1u) != 0u)
continue;
394 const double st = stgSemitones_.load(std::memory_order_relaxed);
395 const bool tr = stgTransient_.load(std::memory_order_relaxed);
396 const bool fo = stgFormant_.load(std::memory_order_relaxed);
399 std::atomic_thread_fence(std::memory_order_acquire);
400 if (s0 == seq_.load(std::memory_order_relaxed))
402 targetSemitones_ = st; transientOn_ = tr; formantOn_ = fo;
406 dirty_.store(
true, std::memory_order_release);
411 void analyse(
int ch, int64_t start, T* out)
noexcept
413 const auto& r = ring_[
static_cast<size_t>(ch)];
414 for (
int k = 0; k < N_; ++k)
415 frame_[
static_cast<size_t>(k)] = r[
static_cast<size_t>((start + k) & ringMask_)]
416 * window_[
static_cast<size_t>(k)];
417 fft_->forward(frame_.data(), out);
420 void processFrame(
int nCh)
noexcept
422 nCh = std::clamp(nCh, 1, numChannels_);
423 const int64_t A = analysisStart();
424 const double Ra = firstFrame_ ?
static_cast<double>(Rs_)
425 : static_cast<double>(A - prevAnalysisStart_);
426 const double binW = kTwoPi / N_;
427 const double Rs =
static_cast<double>(Rs_);
430 std::fill(mag_.begin(), mag_.end(), T(0));
431 std::fill(crossRe_.begin(), crossRe_.end(), 0.0);
432 std::fill(crossIm_.begin(), crossIm_.end(), 0.0);
433 std::fill(dRe_.begin(), dRe_.end(), 0.0);
434 std::fill(dIm_.begin(), dIm_.end(), 0.0);
435 for (
int ch = 0; ch < nCh; ++ch)
437 T* X = spec_[
static_cast<size_t>(ch)].data();
438 const T* P = prevSpec_[
static_cast<size_t>(ch)].data();
440 analyse(ch, A + delta_, specD_.data());
441 for (
int k = 0; k < bins_; ++k)
443 const double xr = X[2 * k], xi = X[2 * k + 1];
444 const double pr = P[2 * k],
pi = P[2 * k + 1];
445 const double dr = specD_[
static_cast<size_t>(2 * k)];
446 const double di = specD_[
static_cast<size_t>(2 * k + 1)];
447 mag_[
static_cast<size_t>(k)] +=
static_cast<T
>(xr * xr + xi * xi);
448 crossRe_[
static_cast<size_t>(k)] += xr * pr + xi *
pi;
449 crossIm_[
static_cast<size_t>(k)] += xi * pr - xr *
pi;
450 dRe_[
static_cast<size_t>(k)] += dr * xr + di * xi;
451 dIm_[
static_cast<size_t>(k)] += di * xr - dr * xi;
455 for (
int k = 0; k < bins_; ++k)
457 auto& m = mag_[
static_cast<size_t>(k)];
459 maxMag = std::max(maxMag, m);
463 const double dl =
static_cast<double>(delta_);
464 for (
int k = 0; k < bins_; ++k)
466 const double wk = binW * k;
467 const double omNow = wk + princArg(std::atan2(dIm_[
static_cast<size_t>(k)],
468 dRe_[
static_cast<size_t>(k)]) - wk * dl) / dl;
472 const double dphi = std::atan2(crossIm_[
static_cast<size_t>(k)],
473 crossRe_[
static_cast<size_t>(k)]);
474 const double prior = 0.5 * (omNow + omPrev_[
static_cast<size_t>(k)]);
475 const double om = Ra >= 1.0 ? prior + princArg(dphi - prior * Ra) / Ra : prior;
476 inc = princArg(theta_[
static_cast<size_t>(k)] + Rs * om - dphi);
478 tinc_[
static_cast<size_t>(k)] = inc;
479 omPrev_[
static_cast<size_t>(k)] = omNow;
484 std::fill(theta_.begin(), theta_.end(), 0.0);
505 if (transientOn_ && !firstFrame_ && lockCount_ > 0)
507 const Lock& L = locks_[
static_cast<size_t>(lockHead_)];
508 const double c =
static_cast<double>(frameStart_) + 0.5 * N_;
509 bool alone = L.h >= kResetCover * N_;
510 for (
const double o : onsetRing_)
511 if (std::abs(o - L.s) > 1.0 && std::abs(o - L.s) < static_cast<double>(N_))
513 if (alone && c >= L.target - L.h && c <= L.target + L.h)
514 for (
int k = 0; k < bins_; ++k)
515 if (mag_[
static_cast<size_t>(k)] > kRiseRatio * prevMag_[
static_cast<size_t>(k)])
516 theta_[
static_cast<size_t>(k)] = 0.0;
519 theta_[
static_cast<size_t>(bins_ - 1)] = 0.0;
520 for (
int k = 0; k < bins_; ++k)
522 rotRe_[
static_cast<size_t>(k)] =
static_cast<T
>(std::cos(theta_[
static_cast<size_t>(k)]));
523 rotIm_[
static_cast<size_t>(k)] =
static_cast<T
>(std::sin(theta_[
static_cast<size_t>(k)]));
527 const bool formant = resample_ && formantOn_ && std::abs(ratio_ - 1.0) > 1e-6;
528 const bool taper = resample_ && ratio_ > 1.0;
529 if (formant) computeFormantGains();
530 else std::fill(gain_.begin(), gain_.end(), T(1));
533 const int cut =
static_cast<int>(
static_cast<double>(N_ / 2) / ratio_);
534 const int taperStart = std::max(1, cut - 4);
535 for (
int k = taperStart; k < bins_; ++k)
536 gain_[
static_cast<size_t>(k)] *= (k <= cut)
537 ?
static_cast<T
>(cut - k + 1) /
static_cast<T
>(cut - taperStart + 1) : T(0);
541 const T norm =
static_cast<T
>(Rs / (0.5 * N_));
542 for (
int ch = 0; ch < nCh; ++ch)
544 T* X = spec_[
static_cast<size_t>(ch)].data();
545 std::copy(X, X + N_ + 2, prevSpec_[
static_cast<size_t>(ch)].data());
546 for (
int k = 0; k < bins_; ++k)
548 const T re = X[2 * k], im = X[2 * k + 1];
549 const T rr = rotRe_[
static_cast<size_t>(k)] * gain_[
static_cast<size_t>(k)];
550 const T ri = rotIm_[
static_cast<size_t>(k)] * gain_[
static_cast<size_t>(k)];
551 X[2 * k] = re * rr - im * ri;
552 X[2 * k + 1] = re * ri + im * rr;
556 fft_->inverse(X, frame_.data());
557 auto& acc = accum_[
static_cast<size_t>(ch)];
558 for (
int k = N_ - Rs_; k < N_; ++k)
559 acc[
static_cast<size_t>((frameStart_ + k) & accumMask_)] = T(0);
560 for (
int k = 0; k < N_; ++k)
561 acc[
static_cast<size_t>((frameStart_ + k) & accumMask_)]
562 += frame_[
static_cast<size_t>(k)] * window_[
static_cast<size_t>(k)] * norm;
564 for (
int ch = nCh; ch < numChannels_; ++ch)
565 std::fill(prevSpec_[
static_cast<size_t>(ch)].begin(),
566 prevSpec_[
static_cast<size_t>(ch)].end(), T(0));
567 std::copy(mag_.begin(), mag_.end(), prevMag_.begin());
569 prevAnalysisStart_ = A;
572 cCur_ =
static_cast<double>(frameStart_) + 0.5 * N_;
582 void propagate(T maxMag)
noexcept
584 const T tol = maxMag *
static_cast<T
>(std::pow(10.0, -kHeapFloorDb / 20.0));
586 for (
int k = 0; k < bins_; ++k)
588 const bool quiet = mag_[
static_cast<size_t>(k)] < tol;
589 done_[
static_cast<size_t>(k)] = quiet ? 1 : 0;
590 theta_[
static_cast<size_t>(k)] = tinc_[
static_cast<size_t>(k)];
594 for (
int k = 0; k < bins_; ++k)
595 if (prevMag_[
static_cast<size_t>(k)] > tol)
596 heapPush(prevMag_[
static_cast<size_t>(k)], 2 * k);
602 int best = -1; T bm = T(-1);
603 for (
int k = 0; k < bins_; ++k)
604 if (!done_[
static_cast<size_t>(k)] && mag_[
static_cast<size_t>(k)] > bm)
605 { bm = mag_[
static_cast<size_t>(k)]; best = k; }
606 done_[
static_cast<size_t>(best)] = 1; --todo;
607 heapPush(mag_[
static_cast<size_t>(best)], 2 * best + 1);
610 const int code = heapPop();
611 const int k = code >> 1;
614 if (!done_[
static_cast<size_t>(k)])
616 done_[
static_cast<size_t>(k)] = 1; --todo;
617 heapPush(mag_[
static_cast<size_t>(k)], 2 * k + 1);
622 const double th = theta_[
static_cast<size_t>(k)];
623 if (k + 1 < bins_ && !done_[
static_cast<size_t>(k + 1)])
625 done_[
static_cast<size_t>(k + 1)] = 1; --todo;
626 theta_[
static_cast<size_t>(k + 1)] = th;
627 heapPush(mag_[
static_cast<size_t>(k + 1)], 2 * (k + 1) + 1);
629 if (k > 0 && !done_[
static_cast<size_t>(k - 1)])
631 done_[
static_cast<size_t>(k - 1)] = 1; --todo;
632 theta_[
static_cast<size_t>(k - 1)] = th;
633 heapPush(mag_[
static_cast<size_t>(k - 1)], 2 * (k - 1) + 1);
639 void heapPush(T key,
int code)
noexcept
644 const int p = (i - 1) >> 1;
645 if (heapKey_[
static_cast<size_t>(p)] >= key)
break;
646 heapKey_[
static_cast<size_t>(i)] = heapKey_[
static_cast<size_t>(p)];
647 heapCode_[
static_cast<size_t>(i)] = heapCode_[
static_cast<size_t>(p)];
650 heapKey_[
static_cast<size_t>(i)] = key;
651 heapCode_[
static_cast<size_t>(i)] = code;
654 int heapPop() noexcept
656 const int top = heapCode_[0];
657 const T key = heapKey_[
static_cast<size_t>(heapSize_ - 1)];
658 const int code = heapCode_[
static_cast<size_t>(heapSize_ - 1)];
664 if (c >= heapSize_)
break;
665 if (c + 1 < heapSize_ && heapKey_[
static_cast<size_t>(c + 1)] > heapKey_[
static_cast<size_t>(c)]) ++c;
666 if (heapKey_[
static_cast<size_t>(c)] <= key)
break;
667 heapKey_[
static_cast<size_t>(i)] = heapKey_[
static_cast<size_t>(c)];
668 heapCode_[
static_cast<size_t>(i)] = heapCode_[
static_cast<size_t>(c)];
673 heapKey_[
static_cast<size_t>(i)] = key;
674 heapCode_[
static_cast<size_t>(i)] = code;
681 void computeFormantGains() noexcept
683 for (
int k = 0; k < bins_; ++k)
684 cepsTime_[
static_cast<size_t>(k)] = std::log(mag_[
static_cast<size_t>(k)] + T(1e-9));
685 for (
int k = bins_; k < N_; ++k)
686 cepsTime_[
static_cast<size_t>(k)] = cepsTime_[
static_cast<size_t>(N_ - k)];
687 fft_->forward(cepsTime_.data(), cepsSpec_.data());
688 const int keep = std::min(std::max(8,
static_cast<int>(0.001 * sampleRate_)), bins_ - 1);
689 for (
int k = keep + 1; k < bins_; ++k)
691 cepsSpec_[
static_cast<size_t>(2 * k)] = T(0);
692 cepsSpec_[
static_cast<size_t>(2 * k + 1)] = T(0);
694 fft_->inverse(cepsSpec_.data(), cepsTime_.data());
695 for (
int k = 0; k < bins_; ++k)
696 envLog_[
static_cast<size_t>(k)] = cepsTime_[
static_cast<size_t>(k)];
697 for (
int k = 0; k < bins_; ++k)
699 const double pos = std::min(
static_cast<double>(k) * ratio_,
static_cast<double>(bins_ - 1));
700 const auto i0 =
static_cast<int>(pos);
701 const auto fr =
static_cast<T
>(pos - i0);
702 const int i1 = std::min(i0 + 1, bins_ - 1);
703 const T target = envLog_[
static_cast<size_t>(i0)]
704 + (envLog_[
static_cast<size_t>(i1)] - envLog_[
static_cast<size_t>(i0)]) * fr;
705 gain_[
static_cast<size_t>(k)] = std::exp(std::clamp(
706 target - envLog_[
static_cast<size_t>(k)], T(-4.6), T(4.6)));
713 void planNext() noexcept
715 adoptParamsIfDirty();
716 const double stTarget = std::clamp(targetSemitones_, -12.0, 12.0);
717 stActive_ += std::clamp(stTarget - stActive_, -0.5, 0.5);
718 ratio_ = std::exp2(stActive_ / 12.0);
719 const double r = ratio_;
721 const double c =
static_cast<double>(frameStart_) + 0.5 * N_;
722 if (steer_) { aNom_ = steerValue_; steer_ =
false; }
723 else aNom_ +=
static_cast<double>(Rs_) / r;
727 while (lockCount_ > 0)
729 const Lock& L = locks_[
static_cast<size_t>(lockHead_)];
730 if (c < L.target + L.h)
break;
731 rejoinC_ = L.target + L.h;
732 rejoinE_ = (L.s + L.h) - (aNom_ + (rejoinC_ - c) / r);
733 rejoinJ_ = std::max(
static_cast<double>(N_), 2.0 * L.h * std::abs(r - 1.0));
734 lockHead_ = (lockHead_ + 1) % kMaxLocks;
741 const Lock& L = locks_[
static_cast<size_t>(lockHead_)];
742 if (c >= L.target - L.h)
743 a = L.s + (c - L.target);
746 const double span = (L.target - L.h) - cCur_;
747 const double f = span > 0.0 ? (c - cCur_) / span : 1.0;
748 a = aCur_ + ((L.s - L.h) - aCur_) * std::clamp(f, 0.0, 1.0);
753 const double f = std::clamp(1.0 - (c - rejoinC_) / rejoinJ_, 0.0, 1.0);
754 a = aNom_ + rejoinE_ * f;
756 aNext_ = std::max(a, aCur_);
760 void addOnset(
double s)
noexcept
762 onsetRing_[
static_cast<size_t>(onsetPos_)] = s;
763 onsetPos_ = (onsetPos_ + 1) % kOnsetRing;
764 if (!transientOn_ || lockCount_ >= kMaxLocks)
return;
765 const double r = ratio_;
766 const double cNext =
static_cast<double>(frameStart_) + 0.5 * N_;
767 const double target = cNext + (s - aNom_) * r;
769 h = std::min(h, s - aCur_ - 1.0);
770 h = std::min(h, target - cNext);
771 if (r < 1.0) h = std::min(h, leadLimit_ / (1.0 / r - 1.0));
772 else if (r > 1.0) h = std::min(h, leadLimit_ / (1.0 - 1.0 / r));
773 double prevS = lastLockS_, prevT = lastLockT_, prevH = 0.0;
776 const Lock& P = locks_[
static_cast<size_t>((lockHead_ + lockCount_ - 1) % kMaxLocks)];
777 prevS = P.s; prevT = P.target; prevH = P.h;
779 h = std::min(h, (s - prevS) - prevH -
static_cast<double>(Rs_));
780 h = std::min(h, (target - prevT) - prevH -
static_cast<double>(Rs_));
781 if (!(h >= 0.5 * Rs_))
return;
782 locks_[
static_cast<size_t>((lockHead_ + lockCount_) % kMaxLocks)] = Lock { s, target, h };
784 lastLockS_ = s; lastLockT_ = target;
786 const double keepNom = aNom_;
787 replanNextFrame(keepNom);
790 void replanNextFrame(
double nom)
noexcept
792 const double c =
static_cast<double>(frameStart_) + 0.5 * N_;
793 const Lock& L = locks_[
static_cast<size_t>(lockHead_)];
795 if (c >= L.target - L.h)
796 a = L.s + (c - L.target);
799 const double span = (L.target - L.h) - cCur_;
800 const double f = span > 0.0 ? (c - cCur_) / span : 1.0;
801 a = aCur_ + ((L.s - L.h) - aCur_) * std::clamp(f, 0.0, 1.0);
806 aNext_ = std::max(a, aCur_);
811 void buildOnsetBank()
813 fbStart_.clear(); fbCount_.clear(); fbOffset_.clear(); fbW_.clear();
814 const double binHz = sampleRate_ / No_;
815 const int nb = No_ / 2 + 1;
816 const double fMax = std::min(16000.0, 0.5 * sampleRate_ * 0.999);
817 std::vector<int> centres;
818 for (
int i = 0; ; ++i)
820 const double f = 27.5 * std::pow(2.0, i / 24.0);
822 const int b = std::clamp(
static_cast<int>(std::lround(f / binHz)), 0, nb - 1);
823 if (centres.empty() || b > centres.back()) centres.push_back(b);
825 for (
size_t j = 1; j + 1 < centres.size(); ++j)
827 const int lo = centres[j - 1], ce = centres[j], hi = centres[j + 1];
828 fbStart_.push_back(lo);
829 fbOffset_.push_back(
static_cast<int>(fbW_.size()));
830 for (
int k = lo; k <= hi; ++k)
831 fbW_.push_back(
static_cast<T
>(k <= ce ?
double(k - lo) / (ce - lo)
832 : double(hi - k) / (hi - ce)));
833 fbCount_.push_back(hi - lo + 1);
835 numBands_ =
static_cast<int>(fbStart_.size());
836 odfScale_ =
static_cast<T
>(2048.0 / No_);
840 void runDetector(
int nCh)
noexcept
842 nCh = std::clamp(nCh, 1, numChannels_);
843 while (nextFrameO_ * hopO_ + No_ <= inCount_)
845 const int64_t start = nextFrameO_ * hopO_;
846 std::fill(powO_.begin(), powO_.end(), T(0));
847 for (
int ch = 0; ch < nCh; ++ch)
849 const auto& r = ring_[
static_cast<size_t>(ch)];
850 for (
int k = 0; k < No_; ++k)
851 frameO_[
static_cast<size_t>(k)] = r[
static_cast<size_t>((start + k) & ringMask_)]
852 * windowO_[
static_cast<size_t>(k)];
853 fftO_->forward(frameO_.data(), specO_.data());
854 for (
int k = 0; k <= No_ / 2; ++k)
856 const T re = specO_[
static_cast<size_t>(2 * k)], im = specO_[
static_cast<size_t>(2 * k + 1)];
857 powO_[
static_cast<size_t>(k)] += re * re + im * im;
860 const T inv = T(1) /
static_cast<T
>(nCh);
862 for (
int b = 0; b < numBands_; ++b)
865 const int s0 = fbStart_[
static_cast<size_t>(b)], off = fbOffset_[
static_cast<size_t>(b)];
866 for (
int i = 0; i < fbCount_[static_cast<size_t>(b)]; ++i)
867 acc += std::sqrt(powO_[
static_cast<size_t>(s0 + i)] * inv) * fbW_[
static_cast<size_t>(off + i)];
868 const T v = std::log10(acc * odfScale_ + T(1));
869 bandCur_[
static_cast<size_t>(b)] = v;
870 const double d =
static_cast<double>(v - bandPrevMax_[
static_cast<size_t>(b)]);
871 if (d > 0.0) flux += d;
873 flux /= std::max(1, numBands_);
874 for (
int b = 0; b < numBands_; ++b)
876 T mx = bandCur_[
static_cast<size_t>(b)];
877 if (b > 0) mx = std::max(mx, bandCur_[
static_cast<size_t>(b - 1)]);
878 if (b + 1 < numBands_) mx = std::max(mx, bandCur_[
static_cast<size_t>(b + 1)]);
879 bandPrevMax_[
static_cast<size_t>(b)] = mx;
881 if (nextFrameO_ == 0) flux = 0.0;
885 if (candPrev_ && fluxPrev_ >= flux && candFrame_ - lastOnsetFrame_ > 4)
887 lastOnsetFrame_ = candFrame_;
888 const double s = locateOnset(candFrame_, nCh);
889 if (s >= 0.0) addOnset(s);
895 const int n = std::min(fluxFill_, kFluxWindow);
896 for (
int i = 0; i < n; ++i) medScratch_[static_cast<size_t>(i)] = fluxHist_[
static_cast<size_t>(i)];
897 std::nth_element(medScratch_.begin(), medScratch_.begin() + n / 2, medScratch_.begin() + n);
898 med = medScratch_[
static_cast<size_t>(n / 2)];
901 const double lo = *std::max_element(medScratch_.begin(), medScratch_.begin() + n / 2);
902 med = 0.5 * (med + lo);
905 candPrev_ = (flux - med > kFluxDelta) && flux >= fluxPrev_;
906 candFrame_ = nextFrameO_;
907 fluxHist_[
static_cast<size_t>(fluxPos_)] = flux;
908 fluxPos_ = (fluxPos_ + 1) % kFluxWindow;
917 [[nodiscard]]
double locateOnset(int64_t i,
int nCh)
const noexcept
919 const int64_t lo = std::max<int64_t>({ 16, i * hopO_ - 2 * hopO_, inCount_ - ringSize_ + 64 });
920 const int64_t hi = std::min<int64_t>(i * hopO_ + (3 * No_) / 4 + hopO_, inCount_ - 16);
921 if (hi - lo < 8)
return -1.0;
922 auto pw = [&](int64_t n) {
924 for (
int ch = 0; ch < nCh; ++ch)
926 const double v = ring_[
static_cast<size_t>(ch)][
static_cast<size_t>(n & ringMask_)];
933 for (int64_t n = lo - 16; n < lo + 16; ++n) run += pw(n);
934 double peak = -1.0; int64_t pk = lo;
938 for (int64_t n = lo; n < hi; ++n)
940 if (rr > peak) { peak = rr; pk = n; }
941 rr += pw(n + 16) - pw(n - 16);
944 if (peak <= 0.0)
return -1.0;
948 for (int64_t n = pk - 16; n < pk + 16; ++n) rr += pw(n);
950 while (j > lo && rr > peak / 16.0)
952 rr += pw(j - 17) - pw(j + 15);
955 return static_cast<double>(j);
959 double sampleRate_ = 48000.0;
960 int numChannels_ = 0;
961 bool prepared_ =
false;
962 bool resample_ =
false;
963 int N_ = 4096, bins_ = 2049, Rs_ = 1024;
964 int D_ = 1536, delta_ = 16;
965 int ringSize_ = 0, ringMask_ = 0, accumSize_ = 0, accumMask_ = 0;
967 std::unique_ptr<FFTReal<T>> fft_, fftO_;
968 std::vector<T> window_, windowO_;
969 std::vector<std::vector<T>> ring_, accum_, spec_, prevSpec_;
970 std::vector<T> specD_, frame_, mag_, prevMag_, rotRe_, rotIm_, gain_;
971 std::vector<double> crossRe_, crossIm_, dRe_, dIm_, theta_, tinc_, omPrev_;
972 std::vector<unsigned char> done_;
973 std::vector<T> heapKey_;
974 std::vector<int> heapCode_;
976 std::vector<T> cepsTime_, cepsSpec_, envLog_;
979 int64_t inCount_ = 0;
980 int64_t frameStart_ = 0;
981 int64_t prevAnalysisStart_ = 0;
982 bool firstFrame_ =
true;
983 double stActive_ = 0.0, ratio_ = 1.0;
984 double aNext_ = 0.0, aNom_ = 0.0, aCur_ = 0.0, cCur_ = 0.0;
985 double origin_ = 0.0, originStream_ = 0.0, originRatio_ = 1.0;
987 double steerValue_ = 0.0;
988 double leadLimit_ = 1e18;
991 std::array<Lock, kMaxLocks> locks_ {};
992 int lockHead_ = 0, lockCount_ = 0;
993 double rejoinE_ = 0.0, rejoinC_ = 0.0, rejoinJ_ = 1.0;
994 double lastLockS_ = -1e18, lastLockT_ = -1e18;
997 int No_ = 1024, hopO_ = 128, numBands_ = 0;
999 std::vector<int> fbStart_, fbCount_, fbOffset_;
1000 std::vector<T> fbW_, frameO_, specO_, powO_, bandCur_, bandPrev_, bandPrevMax_;
1001 std::array<double, kFluxWindow> fluxHist_ {};
1002 std::array<double, kFluxWindow> medScratch_ {};
1003 int fluxFill_ = 0, fluxPos_ = 0;
1004 int64_t nextFrameO_ = 0, candFrame_ = -1, lastOnsetFrame_ = -1000;
1005 static constexpr int kOnsetRing = 8;
1006 std::array<double, kOnsetRing> onsetRing_ {};
1008 double fluxPrev_ = 0.0, fluxPrev2_ = 0.0;
1009 bool candPrev_ =
false;
1012 double targetSemitones_ = 0.0;
1013 bool transientOn_ =
true, formantOn_ =
false;
1014 std::atomic<double> stgSemitones_ { 0.0 };
1015 std::atomic<bool> stgTransient_ {
true };
1016 std::atomic<bool> stgFormant_ {
false };
1017 std::atomic<unsigned> seq_ { 0 };
1018 std::atomic<bool> dirty_ {
false };