Skip to content

Commit 3a32863

Browse files
committed
fix(diffsinger): drop tone_shift pitch-offset logic from acoustic stage
The editor now pre-shifts pitch/f0 before invoking the plugin, so the vocoder-acoustic pipeline renders f0 and note midi faithfully. Remove the tone_shift resample/shift application (semitone/cent) while keeping the tone_shift tag, parser, and result fields; VocoderConfiguration:: pitchControllable remains exposed to the editor.
1 parent bd13e0a commit 3a32863

2 files changed

Lines changed: 4 additions & 38 deletions

File tree

‎domains/ds-infer/unittests/catch2/tst_algorithm_realworld.cpp‎

Lines changed: 1 addition & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -105,8 +105,7 @@ TEST_CASE("resample upsampling from 20ms to 10ms with tail fill", "[algorithm][r
105105
REQUIRE(approxEqual(result[99], 42.0));
106106
}
107107

108-
TEST_CASE("resample tone_shift with fillLast=false zeros tail", "[algorithm][realworld]") {
109-
// AcousticInference uses fillLast=false for tone_shift
108+
TEST_CASE("resample with fillLast=false zeros tail", "[algorithm][realworld]") {
110109
std::vector<double> toneShift(50, 5.0);
111110
auto result = resampleParam(toneShift, 0.02, 0.01, 100, false);
112111
REQUIRE(result.size() == 100);

‎plugins/diffsinger/acoustic/AcousticInference.cpp‎

Lines changed: 3 additions & 36 deletions
Original file line numberDiff line numberDiff line change
@@ -283,7 +283,6 @@ namespace srt::svs {
283283

284284
const Co::InputParameterInfo *pPitchParam = nullptr;
285285
const Co::InputParameterInfo *pF0Param = nullptr;
286-
const Co::InputParameterInfo *pToneShiftParam = nullptr;
287286

288287
for (const auto &param : acousticInput->parameters) {
289288
if (param.tag == Co::Tags::F0) {
@@ -296,11 +295,6 @@ namespace srt::svs {
296295
continue;
297296
}
298297

299-
if (param.tag == Co::Tags::ToneShift) {
300-
pToneShiftParam = &param;
301-
continue;
302-
}
303-
304298
// Resample the parameters to target time step,
305299
// and resize to target frame length (fill with last value)
306300
auto resampled = ds::infer::inferutil::resample(param.values, param.interval, frameWidth,
@@ -409,33 +403,6 @@ namespace srt::svs {
409403
std::string(param.tag.name()) +
410404
" resample failed", {}, "acoustic");
411405
}
412-
// tone_shift (音区偏移) applies to the f0 that drives the acoustic
413-
// model only. The original (un-shifted) f0 is returned through
414-
// AcousticResult::f0 so callers can drive the vocoder / variance
415-
// stage with a pitch independent from the register-shifted one;
416-
// reusing the shifted f0 there double-transposes the output.
417-
auto acousticSamples = samples;
418-
if (pToneShiftParam) {
419-
const auto &toneShift = *pToneShiftParam;
420-
if (!toneShift.values.empty()) {
421-
auto toneShiftSamples = ds::infer::inferutil::resample(
422-
toneShift.values, toneShift.interval, frameWidth, targetLength, false);
423-
if (toneShiftSamples.size() != targetLength) {
424-
return srt::core::Error::inferenceError(srt::core::ErrorCode::InferenceInputInvalid,
425-
"[Acoustic] parameter " + std::string(toneShift.tag.name()) +
426-
" resample failed", {}, "acoustic");
427-
}
428-
if (convertToF0) {
429-
for (size_t i = 0; i < targetLength; ++i) {
430-
acousticSamples[i] += toneShiftSamples[i] / 100.0;
431-
}
432-
} else {
433-
for (size_t i = 0; i < targetLength; ++i) {
434-
acousticSamples[i] *= std::exp2(toneShiftSamples[i] / 1200.0);
435-
}
436-
}
437-
}
438-
}
439406

440407
// Convert midi note to hz
441408
const auto toHz = [](double note) -> float {
@@ -460,18 +427,18 @@ namespace srt::svs {
460427
}
461428
};
462429

463-
// f0 tensor for the acoustic model (tone-shifted)
430+
// f0 tensor for the acoustic model (un-shifted, faithful)
464431
auto expForAcoustic = ds::infer::inferutil::TensorHelper<float>::createFor1DArray(targetLength);
465432
if (!expForAcoustic) {
466433
Log.srtCritical("%1 start: failed to create f0 tensor: %2",
467434
kLogPrefix, expForAcoustic.error().message());
468435
return expForAcoustic.takeError();
469436
}
470437
auto &acousticHelper = expForAcoustic.value();
471-
fillF0(acousticHelper, acousticSamples);
438+
fillF0(acousticHelper, samples);
472439
sessionInput->inputs["f0"] = acousticHelper.take(); // ref count +1
473440

474-
// f0 tensor for the vocoder (original, un-shifted)
441+
// f0 tensor for the vocoder (un-shifted, faithful)
475442
auto expForVocoder = ds::infer::inferutil::TensorHelper<float>::createFor1DArray(targetLength);
476443
if (!expForVocoder) {
477444
Log.srtCritical("%1 start: failed to create vocoder f0 tensor: %2",

0 commit comments

Comments
 (0)