From 753b7cc622aadf802b3145d7bb8f7df4afa213c4 Mon Sep 17 00:00:00 2001 From: Kakaru <97896816+KakaruHayate@users.noreply.github.com> Date: Sun, 28 Jun 2026 15:35:56 +0800 Subject: [PATCH 01/23] fix: Roll back the changes related to onnxslim in #308 and #305 (#310) --- deployment/exporters/acoustic_exporter.py | 28 ++++++---- deployment/exporters/nsf_hifigan_exporter.py | 4 +- deployment/exporters/variance_exporter.py | 58 +++++++++----------- docs/GettingStarted.md | 2 +- requirements.txt | 4 +- utils/onnx_helper.py | 10 ---- 6 files changed, 50 insertions(+), 56 deletions(-) diff --git a/deployment/exporters/acoustic_exporter.py b/deployment/exporters/acoustic_exporter.py index b8d5f7a72..1645c8358 100644 --- a/deployment/exporters/acoustic_exporter.py +++ b/deployment/exporters/acoustic_exporter.py @@ -3,6 +3,7 @@ from typing import Union, List, Tuple, Dict import onnx +import onnxsim import torch import yaml @@ -346,7 +347,8 @@ def _perform_spk_mix(self, spk_mix: Dict[str, float]): def _optimize_fs2_aux_graph(self, fs2: onnx.ModelProto) -> onnx.ModelProto: print(f'Running ONNX Simplifier on {self.fs2_aux_class_name}...') - fs2 = onnx_helper.simplify_onnx(fs2) + fs2, check = onnxsim.simplify(fs2, include_subgraph=True) + assert check, 'Simplified ONNX model could not be validated' onnx_helper.model_reorder_io_list( fs2, 'input', target_name='languages', insert_after_name='tokens' @@ -358,13 +360,9 @@ def _optimize_diffusion_graph(self, diffusion: onnx.ModelProto) -> onnx.ModelPro onnx_helper.model_override_io_shapes(diffusion, output_shapes={ 'mel': (1, 'n_frames', hparams['audio_num_mel_bins']) }) - # Running simplify_onnx here used to be "Simplifier #1", but the - # subsequent graph_extract_conditioner_projections call mutates - # the topology in ways that can collide with simplifier output - # ordering (causing the merged model to fail topological-sort - # validation downstream). The second simplifier pass below handles - # everything that the dropped first pass did, so removing it is - # both safe and a fix for a latent merge bug. + print(f'Running ONNX Simplifier #1 on {self.diffusion_class_name}...') + diffusion, check = onnxsim.simplify(diffusion, include_subgraph=True) + assert check, 'Simplified ONNX model could not be validated' onnx_helper.graph_fold_back_to_squeeze(diffusion.graph) onnx_helper.graph_extract_conditioner_projections( graph=diffusion.graph, op_type='Conv', @@ -372,8 +370,12 @@ def _optimize_diffusion_graph(self, diffusion: onnx.ModelProto) -> onnx.ModelPro alias_prefix='/diffusion/backbone/cache' ) onnx_helper.graph_remove_unused_values(diffusion.graph) - print(f'Running ONNX Simplifier on {self.diffusion_class_name}...') - diffusion = onnx_helper.simplify_onnx(diffusion) + print(f'Running ONNX Simplifier #2 on {self.diffusion_class_name}...') + diffusion, check = onnxsim.simplify( + diffusion, + include_subgraph=True + ) + assert check, 'Simplified ONNX model could not be validated' print(f'| optimize graph: {self.diffusion_class_name}') return diffusion @@ -397,7 +399,11 @@ def _merge_fs2_aux_diffusion_graphs(self, fs2: onnx.ModelProto, diffusion: onnx. merged.graph.name = fs2.graph.name print(f'Running ONNX Simplifier on {self.model_class_name}...') - merged = onnx_helper.simplify_onnx(merged) + merged, check = onnxsim.simplify( + merged, + include_subgraph=True + ) + assert check, 'Simplified ONNX model could not be validated' print(f'| optimize graph: {self.model_class_name}') return merged diff --git a/deployment/exporters/nsf_hifigan_exporter.py b/deployment/exporters/nsf_hifigan_exporter.py index 17dfb6fc7..26f2aa4d3 100644 --- a/deployment/exporters/nsf_hifigan_exporter.py +++ b/deployment/exporters/nsf_hifigan_exporter.py @@ -3,6 +3,7 @@ from typing import Union import onnx +import onnxsim import torch import yaml from torch import nn @@ -121,5 +122,6 @@ def _torch_export_model(self): def _optimize_model_graph(self, model: onnx.ModelProto) -> onnx.ModelProto: print(f'Running ONNX simplifier for {self.model_class_name}...') - model = onnx_helper.simplify_onnx(model) + model, check = onnxsim.simplify(model, include_subgraph=True) + assert check, 'Simplified ONNX model could not be validated' return model diff --git a/deployment/exporters/variance_exporter.py b/deployment/exporters/variance_exporter.py index 428a440c5..e8832ff12 100644 --- a/deployment/exporters/variance_exporter.py +++ b/deployment/exporters/variance_exporter.py @@ -1,9 +1,9 @@ import json -import re from pathlib import Path from typing import Union, List, Tuple, Dict import onnx +import onnxsim import torch import yaml @@ -659,7 +659,8 @@ def _optimize_linguistic_graph(self, linguistic: onnx.ModelProto) -> onnx.ModelP } ) print(f'Running ONNX Simplifier on {self.fs2_class_name}...') - linguistic = onnx_helper.simplify_onnx(linguistic) + linguistic, check = onnxsim.simplify(linguistic, include_subgraph=True) + assert check, 'Simplified ONNX model could not be validated' onnx_helper.model_reorder_io_list( linguistic, 'input', target_name='languages', insert_after_name='tokens' @@ -675,7 +676,8 @@ def _optimize_dur_predictor_graph(self, dur_predictor: onnx.ModelProto) -> onnx. } ) print(f'Running ONNX Simplifier on {self.dur_predictor_class_name}...') - dur_predictor = onnx_helper.simplify_onnx(dur_predictor) + dur_predictor, check = onnxsim.simplify(dur_predictor, include_subgraph=True) + assert check, 'Simplified ONNX model could not be validated' print(f'| optimize graph: {self.dur_predictor_class_name}') return dur_predictor @@ -685,16 +687,15 @@ def _optimize_merge_pitch_predictor_graph( onnx_helper.model_override_io_shapes( pitch_pre, output_shapes={'pitch_cond': (1, 'n_frames', hparams['hidden_size'])} ) - pitch_pre = onnx_helper.simplify_onnx(pitch_pre) + pitch_pre, check = onnxsim.simplify(pitch_pre, include_subgraph=True) + assert check, 'Simplified ONNX model could not be validated' onnx_helper.model_override_io_shapes( pitch_predictor, output_shapes={'pitch_pred': (1, 'n_frames')} ) - # See acoustic_exporter._optimize_diffusion_graph: the pre-surgery - # simplifier pass was dropped because the conditioner-projection - # extraction below can produce a topology that collides with the - # simplifier's output, causing merge_models to fail topological-sort - # validation. The post-surgery simplifier handles the same work. + print(f'Running ONNX Simplifier #1 on {self.pitch_predictor_class_name}...') + pitch_predictor, check = onnxsim.simplify(pitch_predictor, include_subgraph=True) + assert check, 'Simplified ONNX model could not be validated' onnx_helper.graph_fold_back_to_squeeze(pitch_predictor.graph) onnx_helper.graph_extract_conditioner_projections( graph=pitch_predictor.graph, op_type='Conv', @@ -702,15 +703,13 @@ def _optimize_merge_pitch_predictor_graph( alias_prefix='/pitch_predictor/backbone/cache' ) onnx_helper.graph_remove_unused_values(pitch_predictor.graph) - print(f'Running ONNX Simplifier on {self.pitch_predictor_class_name}...') - pitch_predictor = onnx_helper.simplify_onnx(pitch_predictor) + print(f'Running ONNX Simplifier #2 on {self.pitch_predictor_class_name}...') + pitch_predictor, check = onnxsim.simplify(pitch_predictor, include_subgraph=True) + assert check, 'Simplified ONNX model could not be validated' onnx_helper.model_add_prefixes(pitch_pre, node_prefix='/pre', ignored_pattern=r'.*embed.*') onnx_helper.model_add_prefixes(pitch_pre, dim_prefix='pre.', ignored_pattern='(n_tokens)|(n_notes)|(n_frames)') - onnx_helper.model_add_prefixes( - pitch_post, node_prefix='/post', value_info_prefix='/post', initializer_prefix='/post', - ignored_pattern=r'.*(pitch_pred).*' - ) + onnx_helper.model_add_prefixes(pitch_post, node_prefix='/post', ignored_pattern=None) onnx_helper.model_add_prefixes(pitch_post, dim_prefix='post.', ignored_pattern='n_frames') pitch_pre_diffusion = onnx.compose.merge_models( pitch_pre, pitch_predictor, io_map=[('pitch_cond', 'pitch_cond')], @@ -737,7 +736,8 @@ def _optimize_merge_variance_predictor_graph( onnx_helper.model_override_io_shapes( var_pre, output_shapes={'variance_cond': (1, 'n_frames', hparams['hidden_size'])} ) - var_pre = onnx_helper.simplify_onnx(var_pre) + var_pre, check = onnxsim.simplify(var_pre, include_subgraph=True) + assert check, 'Simplified ONNX model could not be validated' onnx_helper.model_override_io_shapes( var_diffusion, output_shapes={ @@ -746,10 +746,9 @@ def _optimize_merge_variance_predictor_graph( else (1, len(self.model.variance_prediction_list), 'n_frames') } ) - # See acoustic_exporter._optimize_diffusion_graph: pre-surgery - # simplifier dropped to avoid a latent topology collision with the - # conditioner-projection extraction; the post-surgery pass covers - # the same simplifications. + print(f'Running ONNX Simplifier #1 on {self.multi_var_predictor_class_name}...') + var_diffusion, check = onnxsim.simplify(var_diffusion, include_subgraph=True) + assert check, 'Simplified ONNX model could not be validated' onnx_helper.graph_fold_back_to_squeeze(var_diffusion.graph) onnx_helper.graph_extract_conditioner_projections( graph=var_diffusion.graph, op_type='Conv', @@ -757,25 +756,22 @@ def _optimize_merge_variance_predictor_graph( alias_prefix='/variance_predictor/backbone/cache' ) onnx_helper.graph_remove_unused_values(var_diffusion.graph) - print(f'Running ONNX Simplifier on {self.multi_var_predictor_class_name}...') - var_diffusion = onnx_helper.simplify_onnx(var_diffusion) + print(f'Running ONNX Simplifier #2 on {self.multi_var_predictor_class_name}...') + var_diffusion, check = onnxsim.simplify(var_diffusion, include_subgraph=True) + assert check, 'Simplified ONNX model could not be validated' - var_post = onnx_helper.simplify_onnx(var_post) + var_post, check = onnxsim.simplify(var_post, include_subgraph=True) + assert check, 'Simplified ONNX model could not be validated' - ignored_variance_names = '|'.join( - f'({re.escape(v_name)})' for v_name in self.model.variance_prediction_list - ) if self.model.variance_prediction_list else '(?!)' - ignored_variance_pred_names = '|'.join( - f'({re.escape(v_name)}_pred)' for v_name in self.model.variance_prediction_list - ) if self.model.variance_prediction_list else '(?!)' + ignored_variance_names = '|'.join([f'({v_name})' for v_name in self.model.variance_prediction_list]) onnx_helper.model_add_prefixes( var_pre, node_prefix='/pre', value_info_prefix='/pre', initializer_prefix='/pre', - ignored_pattern=fr'.*((embed)|(variance_cond)|{ignored_variance_names}).*' + ignored_pattern=fr'.*((embed)|{ignored_variance_names}).*' ) onnx_helper.model_add_prefixes(var_pre, dim_prefix='pre.', ignored_pattern='(n_tokens)|(n_frames)') onnx_helper.model_add_prefixes( var_post, node_prefix='/post', value_info_prefix='/post', initializer_prefix='/post', - ignored_pattern=fr'.*({ignored_variance_pred_names}).*' + ignored_pattern=None ) onnx_helper.model_add_prefixes(var_post, dim_prefix='post.', ignored_pattern='n_frames') diff --git a/docs/GettingStarted.md b/docs/GettingStarted.md index 02144af3f..1f7ef8f94 100644 --- a/docs/GettingStarted.md +++ b/docs/GettingStarted.md @@ -6,7 +6,7 @@ DiffSinger requires Python 3.10 or later. We strongly recommend you create a virtual environment via Conda, venv or uv before installing dependencies. -1. Install The latest PyTorch following the [official instructions](https://pytorch.org/get-started/locally/) according to your OS and hardware. We recommend using version >= 2.4.0, preferably <= 2.8.0. +1. Install The latest PyTorch following the [official instructions](https://pytorch.org/get-started/locally/) according to your OS and hardware. We recommend using the latest stable release that is >= 2.4.0. 2. Install other dependencies via the following command: diff --git a/requirements.txt b/requirements.txt index 1db04e83f..4645417f7 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,5 +1,5 @@ # It is recommended to install PyTorch manually. -# PyTorch >= 2.4, preferably <= 2.8 is recommended. +# PyTorch >= 2.4 is recommended. # See instructions at https://pytorch.org/get-started/locally/ click @@ -11,7 +11,7 @@ matplotlib MonkeyType>=23.3.0 numpy<2.0.0 onnx>=1.21.0 -onnxslim>=0.1.93 +onnxsim>=0.6.5 praat-parselmouth==0.4.3 pyworld==0.3.4 PyYAML diff --git a/utils/onnx_helper.py b/utils/onnx_helper.py index 522aa39b1..5fddcfe49 100644 --- a/utils/onnx_helper.py +++ b/utils/onnx_helper.py @@ -28,16 +28,6 @@ ) -def simplify_onnx(model: ModelProto) -> ModelProto: - """Simplify an ONNX ModelProto and return the simplified model. - - Unlike ``onnxslim.slim``, this function returns a single ``ModelProto`` - (not a tuple); validation failures raise instead of returning a flag. - """ - import onnxslim - return onnxslim.slim(model) - - def _verbose(self, *args, sep=' ', end='\n', file=None): if __verbose__: print(self, *args, sep=sep, end=end, file=file) From 36da95303e049a1a5c89bbf26a7551d0d1106c23 Mon Sep 17 00:00:00 2001 From: Kakaru <97896816+KakaruHayate@users.noreply.github.com> Date: Sun, 2 Aug 2026 20:55:07 +0800 Subject: [PATCH 02/23] fix: pool frame-level speaker mixes for Mix-LN (#317) --- deployment/modules/fastspeech2.py | 16 +--------------- modules/fastspeech/acoustic_encoder.py | 20 +++++++++++++++++++- 2 files changed, 20 insertions(+), 16 deletions(-) diff --git a/deployment/modules/fastspeech2.py b/deployment/modules/fastspeech2.py index 55f4b284d..f70848bee 100644 --- a/deployment/modules/fastspeech2.py +++ b/deployment/modules/fastspeech2.py @@ -6,7 +6,7 @@ import torch.nn.functional as F from modules.commons.common_layers import NormalInitEmbedding as Embedding -from modules.fastspeech.acoustic_encoder import FastSpeech2Acoustic +from modules.fastspeech.acoustic_encoder import FastSpeech2Acoustic, uniform_attention_pooling from modules.fastspeech.variance_encoder import FastSpeech2Variance from utils.hparams import hparams from utils.phoneme_utils import PAD_INDEX @@ -18,20 +18,6 @@ f0_mel_max = 1127 * np.log(1 + f0_max / 700) -def uniform_attention_pooling(spk_embed, durations): - _, T_mel, _ = spk_embed.shape - ph_starts = torch.cumsum(torch.cat([torch.zeros_like(durations[:, :1]), durations[:, :-1]], dim=1), dim=1) - ph_ends = ph_starts + durations - mel_indices = torch.arange(T_mel, device=spk_embed.device).view(1, 1, T_mel) - phoneme_to_mel_mask = (mel_indices >= ph_starts.unsqueeze(-1)) & (mel_indices < ph_ends.unsqueeze(-1)) - uniform_scores = phoneme_to_mel_mask.float() - sum_scores = uniform_scores.sum(dim=2, keepdim=True) - attn_weights = uniform_scores / (sum_scores + (sum_scores == 0).float()) # [B, T_ph, T_mel] - ph_spk_embed = torch.bmm(attn_weights, spk_embed) - - return ph_spk_embed - - def f0_to_coarse(f0): f0_mel = 1127 * (1 + f0 / 700).log() a = (f0_bin - 2) / (f0_mel_max - f0_mel_min) diff --git a/modules/fastspeech/acoustic_encoder.py b/modules/fastspeech/acoustic_encoder.py index cbd048008..241f9871c 100644 --- a/modules/fastspeech/acoustic_encoder.py +++ b/modules/fastspeech/acoustic_encoder.py @@ -12,6 +12,20 @@ from utils.phoneme_utils import PAD_INDEX +def uniform_attention_pooling(spk_embed, durations): + _, T_mel, _ = spk_embed.shape + ph_starts = torch.cumsum(torch.cat([torch.zeros_like(durations[:, :1]), durations[:, :-1]], dim=1), dim=1) + ph_ends = ph_starts + durations + mel_indices = torch.arange(T_mel, device=spk_embed.device).view(1, 1, T_mel) + phoneme_to_mel_mask = (mel_indices >= ph_starts.unsqueeze(-1)) & (mel_indices < ph_ends.unsqueeze(-1)) + uniform_scores = phoneme_to_mel_mask.float() + sum_scores = uniform_scores.sum(dim=2, keepdim=True) + attn_weights = uniform_scores / (sum_scores + (sum_scores == 0).float()) # [B, T_ph, T_mel] + ph_spk_embed = torch.bmm(attn_weights, spk_embed) + + return ph_spk_embed + + class FastSpeech2Acoustic(nn.Module): def __init__(self, vocab_size): super().__init__() @@ -143,7 +157,11 @@ def forward( extra_embed = dur_embed + lang_embed else: extra_embed = dur_embed - encoder_out = self.encoder(txt_embed, extra_embed, txt_tokens == 0, spk_embed) + if self.use_mix_ln and spk_embed is not None and spk_embed.shape[1] > 1: + ph_spk_embed = uniform_attention_pooling(spk_embed, dur) + else: + ph_spk_embed = spk_embed + encoder_out = self.encoder(txt_embed, extra_embed, txt_tokens == 0, ph_spk_embed) encoder_out = F.pad(encoder_out, [0, 0, 1, 0]) mel2ph_ = mel2ph[..., None].repeat([1, 1, encoder_out.shape[-1]]) From ee7463f551dac8f5f36caa30e4e61fdbbed6d3cd Mon Sep 17 00:00:00 2001 From: yxlllc <33565655+yxlllc@users.noreply.github.com> Date: Mon, 3 Aug 2026 19:39:05 +0800 Subject: [PATCH 03/23] some minor fixes (#318) --- inference/ds_variance.py | 5 +++-- preprocessing/acoustic_binarizer.py | 2 +- preprocessing/variance_binarizer.py | 15 +++++++++++---- utils/infer_utils.py | 8 +------- 4 files changed, 16 insertions(+), 14 deletions(-) diff --git a/inference/ds_variance.py b/inference/ds_variance.py index da3d6e94d..cb682fd93 100644 --- a/inference/ds_variance.py +++ b/inference/ds_variance.py @@ -241,14 +241,15 @@ def preprocess_input( batch['midi'] = ph_midi if load_pitch: + # Interpolate unvoiced parts before resampling. f0 = resample_align_curve( - np.array(param['f0_seq'].split(), np.float32), + interp_f0(np.array(param['f0_seq'].split(), np.float32))[0], original_timestep=float(param['f0_timestep']), target_timestep=self.timestep, align_length=T_s ) batch['pitch'] = torch.from_numpy( - librosa.hz_to_midi(interp_f0(f0)[0]).astype(np.float32) + librosa.hz_to_midi(f0).astype(np.float32) ).to(self.device)[None] if self.model.predict_dur: diff --git a/preprocessing/acoustic_binarizer.py b/preprocessing/acoustic_binarizer.py index 9301f14bc..16ad953a0 100644 --- a/preprocessing/acoustic_binarizer.py +++ b/preprocessing/acoustic_binarizer.py @@ -337,7 +337,7 @@ def arrange_data_augmentation(self, data_iterator): aug_list.append(aug_task) elif aug_type == 1: aug_task = { - 'name': aug_item, + 'name': aug_item['name'], 'func': aug_item['func'], 'kwargs': deepcopy(aug_item['kwargs']) } diff --git a/preprocessing/variance_binarizer.py b/preprocessing/variance_binarizer.py index 3d2990fe4..589bf571a 100644 --- a/preprocessing/variance_binarizer.py +++ b/preprocessing/variance_binarizer.py @@ -314,14 +314,21 @@ def process_item(self, item_name, meta_data, binarization_args): if self.prefer_ds: f0_seq = self.load_attr_from_ds(ds_id, name, 'f0_seq', idx=ds_seg_idx) if f0_seq is not None: + f0_timestep = float(self.load_attr_from_ds(ds_id, name, 'f0_timestep', idx=ds_seg_idx)) + # Interpolate unvoiced parts before resampling. + f0_points, uv_points = interp_f0(np.array(f0_seq.split(), np.float32)) f0 = resample_align_curve( - np.array(f0_seq.split(), np.float32), - original_timestep=float(self.load_attr_from_ds(ds_id, name, 'f0_timestep', idx=ds_seg_idx)), + f0_points, + original_timestep=f0_timestep, target_timestep=self.timestep, align_length=length ) - uv = f0 == 0 - f0, _ = interp_f0(f0, uv) + uv = resample_align_curve( + uv_points.astype(np.float32), + original_timestep=f0_timestep, + target_timestep=self.timestep, + align_length=length + ) > 0.5 if f0 is None: f0, uv = pitch_extractor.get_pitch( waveform, samplerate=hparams['audio_sample_rate'], length=length, diff --git a/utils/infer_utils.py b/utils/infer_utils.py index 7dc32c2ee..ec649b7a8 100644 --- a/utils/infer_utils.py +++ b/utils/infer_utils.py @@ -39,17 +39,11 @@ def trans_key(raw_data, key): def resample_align_curve(points: np.ndarray, original_timestep: float, target_timestep: float, align_length: int): - t_max = (len(points) - 1) * original_timestep curve_interp = np.interp( - np.arange(0, t_max, target_timestep), + np.arange(align_length) * target_timestep, original_timestep * np.arange(len(points)), points ).astype(points.dtype) - delta_l = align_length - len(curve_interp) - if delta_l < 0: - curve_interp = curve_interp[:align_length] - elif delta_l > 0: - curve_interp = np.concatenate((curve_interp, np.full(delta_l, fill_value=curve_interp[-1])), axis=0) return curve_interp From 8a07f76596d8f61334fa7c950c043f6fd802df15 Mon Sep 17 00:00:00 2001 From: Kakaru <97896816+KakaruHayate@users.noreply.github.com> Date: Tue, 4 Aug 2026 13:37:48 +0800 Subject: [PATCH 04/23] fix: pad mel with spec_min instead of 0.0 in acoustic collater (#313) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix: pad mel with spec_min instead of 0.0 in acoustic collater Raw log-mel 0.0 is not a neutral padding value: norm_spec maps it to +1.0 — the very top of the normalized range, i.e. maximum loudness. Every batch therefore filled the padded tail of shorter samples with full-loudness garbage. Two consequences: 1. The diffusion backbones (WaveNet / LYNXNet / LYNXNet2) receive no padding mask, so their receptive field (~181 frames for LYNXNet2 with kernel_size=31 x 6 layers) leaks the fake signal into the trailing valid frames. The loss mask (mel2ph > 0) hides this from the loss on padding frames, but the contaminated valid frames near the boundary are fully counted — a systematic bias on utterance tails, exactly where breathy endings and vibrato decay live. 2. The aux decoder loss is not masked at all: with zero-padding the aux decoder was actively trained to predict maximum loudness on padding frames from near-zero condition. Padding with spec_min (-12 by default) maps to -1.0 (silence) and sits next to the mel extractor's true silence floor log(1e-5) = -11.51, so padded regions now look like ordinary trailing silence — consistent with what the model sees at inference time. Note: this changes the training data distribution slightly; models trained before/after this fix are checkpoint-compatible but their padded-region behavior differs. * Update mel padding strategy to use spec_min Replace zero-padding with spec_min for mel padding to avoid full-loudness garbage in shorter samples. * Change mel padding to use log scale --- training/acoustic_task.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/training/acoustic_task.py b/training/acoustic_task.py index ca6a71c65..eb052bce0 100644 --- a/training/acoustic_task.py +++ b/training/acoustic_task.py @@ -1,3 +1,5 @@ +import math + import matplotlib import torch import torch.distributions @@ -45,7 +47,7 @@ def collater(self, samples): tokens = utils.collate_nd([s['tokens'] for s in samples], 0) f0 = utils.collate_nd([s['f0'] for s in samples], 0.0) mel2ph = utils.collate_nd([s['mel2ph'] for s in samples], 0) - mel = utils.collate_nd([s['mel'] for s in samples], 0.0) + mel = utils.collate_nd([s['mel'] for s in samples], math.log(1e-5)) batch.update({ 'tokens': tokens, 'mel2ph': mel2ph, From 96bb14dae9135dea98369fab0cad170673b11767 Mon Sep 17 00:00:00 2001 From: yxlllc <33565655+yxlllc@users.noreply.github.com> Date: Tue, 11 Aug 2026 15:15:43 +0800 Subject: [PATCH 05/23] some minor fixes (#321) * some minor fixes * some minor fixes * fix --- inference/ds_variance.py | 25 ++++++++++++------------- modules/toplevel.py | 5 ++++- preprocessing/acoustic_binarizer.py | 3 ++- utils/decomposed_waveform.py | 2 +- 4 files changed, 19 insertions(+), 16 deletions(-) diff --git a/inference/ds_variance.py b/inference/ds_variance.py index cb682fd93..f2821e2e9 100644 --- a/inference/ds_variance.py +++ b/inference/ds_variance.py @@ -443,19 +443,18 @@ def run_inference( param_copy[f'{v_name}_timestep'] = str(self.timestep) # Restore ph_spk_mix and spk_mix - if 'ph_spk_mix' in param_copy and 'spk_mix' in param_copy: - if 'ph_spk_mix_backup' in param_copy: - if param_copy['ph_spk_mix_backup'] is None: - del param_copy['ph_spk_mix'] - else: - param_copy['ph_spk_mix'] = param_copy['ph_spk_mix_backup'] - del param['ph_spk_mix_backup'] - if 'spk_mix_backup' in param_copy: - if param_copy['ph_spk_mix_backup'] is None: - del param_copy['spk_mix'] - else: - param_copy['spk_mix'] = param_copy['spk_mix_backup'] - del param['spk_mix_backup'] + if 'ph_spk_mix_backup' in param_copy: + if param_copy['ph_spk_mix_backup'] is None: + param_copy.pop('ph_spk_mix', None) + else: + param_copy['ph_spk_mix'] = param_copy['ph_spk_mix_backup'] + del param_copy['ph_spk_mix_backup'] + if 'spk_mix_backup' in param_copy: + if param_copy['spk_mix_backup'] is None: + param_copy.pop('spk_mix', None) + else: + param_copy['spk_mix'] = param_copy['spk_mix_backup'] + del param_copy['spk_mix_backup'] results.append(param_copy) diff --git a/modules/toplevel.py b/modules/toplevel.py index 4a97b3c91..ebe9a9bc8 100644 --- a/modules/toplevel.py +++ b/modules/toplevel.py @@ -341,7 +341,10 @@ def forward( return dur_pred_out, pitch_pred_out, ({} if infer else None) if pitch is None: - pitch = base_pitch + pitch_pred_out + if pitch_pred_out is not None: + pitch = base_pitch + pitch_pred_out + else: + pitch = base_pitch if self.use_variance_scaling: var_cond = condition + self.pitch_embed(pitch[:, :, None] / 12) else: diff --git a/preprocessing/acoustic_binarizer.py b/preprocessing/acoustic_binarizer.py index 16ad953a0..dff5e679b 100644 --- a/preprocessing/acoustic_binarizer.py +++ b/preprocessing/acoustic_binarizer.py @@ -319,7 +319,8 @@ def arrange_data_augmentation(self, data_iterator): k_from_aug = int(total_scale * scale / (1 + total_scale) * len(all_item_names)) k_mutate = int(total_scale * scale / (1 + scale) * len(all_item_names)) aug_types = [0] * k_from_raw + [1] * k_from_aug + [2] * k_mutate - aug_items = random.choices(all_item_names, k=k_from_raw) + random.choices(aug_list, k=k_from_aug + k_mutate) + aug_items = random.choices(all_item_names, k=k_from_raw) + \ + random.choices(aug_list, k=k_from_aug) + random.sample(aug_list, k=min(k_mutate, len(aug_list))) for aug_type, aug_item in zip(aug_types, aug_items): # Uniform distribution in log domain diff --git a/utils/decomposed_waveform.py b/utils/decomposed_waveform.py index cfccb0e17..593bf868f 100644 --- a/utils/decomposed_waveform.py +++ b/utils/decomposed_waveform.py @@ -75,7 +75,7 @@ def _init( # extraction parameters self._hop_size = hop_size self._fft_size = fft_size if fft_size is not None else win_size - self._win_size = win_size if win_size is not None else win_size + self._win_size = win_size if win_size is not None else fft_size self._time_step = hop_size / samplerate self._half_width = base_harmonic_radius self._device = ('cuda' if torch.cuda.is_available() else 'cpu') if device is None else device From e2307b1080b00f3999702ce9017cfd75c7f862fe Mon Sep 17 00:00:00 2001 From: Kakaru <97896816+KakaruHayate@users.noreply.github.com> Date: Fri, 14 Aug 2026 00:06:49 +0800 Subject: [PATCH 06/23] feat: Triton fused SoftSignGLU kernel for LYNXNet2 (#312) --- configs/acoustic.yaml | 3 + configs/variance.yaml | 3 + modules/backbones/lynxnet2.py | 6 +- modules/commons/common_layers.py | 41 ++ modules/kernels/__init__.py | 1 + modules/kernels/fused_linear_softsign_glu.py | 482 +++++++++++++++++++ modules/kernels/integration.py | 352 ++++++++++++++ training/acoustic_task.py | 32 ++ training/variance_task.py | 47 ++ 9 files changed, 966 insertions(+), 1 deletion(-) create mode 100644 modules/kernels/__init__.py create mode 100644 modules/kernels/fused_linear_softsign_glu.py create mode 100644 modules/kernels/integration.py diff --git a/configs/acoustic.yaml b/configs/acoustic.yaml index fad75600e..6ae4c7e78 100644 --- a/configs/acoustic.yaml +++ b/configs/acoustic.yaml @@ -3,6 +3,9 @@ base_config: task_cls: training.acoustic_task.AcousticTask +# Enable Triton-fused Linear+SoftSignGLU kernels for LYNXNet2 backbones. +use_fused_kernels: false + dictionaries: {} extra_phonemes: [] merged_phoneme_groups: [] diff --git a/configs/variance.yaml b/configs/variance.yaml index d4e203670..10f90d0f9 100644 --- a/configs/variance.yaml +++ b/configs/variance.yaml @@ -3,6 +3,9 @@ base_config: task_cls: training.variance_task.VarianceTask +# Enable Triton-fused Linear+SoftSignGLU kernels for LYNXNet2 backbones. +use_fused_kernels: false + dictionaries: {} extra_phonemes: [] merged_phoneme_groups: [] diff --git a/modules/backbones/lynxnet2.py b/modules/backbones/lynxnet2.py index 6e55d5f87..e2c717462 100644 --- a/modules/backbones/lynxnet2.py +++ b/modules/backbones/lynxnet2.py @@ -2,7 +2,9 @@ import torch.nn as nn import torch.nn.functional as F -from modules.commons.common_layers import SinusoidalPosEmb, SwiGLU, ATanGLU, Transpose, AdamWLinear +from modules.commons.common_layers import ( + SinusoidalPosEmb, SwiGLU, ATanGLU, SoftSignGLU, Transpose, AdamWLinear +) from utils.hparams import hparams @@ -14,6 +16,8 @@ def __init__(self, dim, expansion_factor, kernel_size=31, dropout=0., glu_type=' _glu = SwiGLU() elif glu_type == 'atanglu': _glu = ATanGLU() + elif glu_type == 'softsign_glu': + _glu = SoftSignGLU() else: raise ValueError(f'{glu_type} is not a valid activation') if float(dropout) > 0.: diff --git a/modules/commons/common_layers.py b/modules/commons/common_layers.py index 4da65693a..10852e20e 100644 --- a/modules/commons/common_layers.py +++ b/modules/commons/common_layers.py @@ -175,6 +175,47 @@ def forward(self, x): return out * torch.atan(gate) +class SoftSignGLUFunction(torch.autograd.Function): + """ATanGLUFunction-style memory trick for SoftSignGLU. + + softsign'(x) = 1/(1+|x|)^2 = (1-|softsign(x)|)^2, so both partial + derivatives of y = out * softsign(gate) are precomputable in forward: + dy/dout = softsign(gate) + dy/dgate = out * (1-|softsign(gate)|)^2 + Saves 2 tensors (vs 3 for naive autograd) and backward is two pure + multiplies with no softsign recompute. + """ + @staticmethod + def forward(ctx, out, gate): + ss_gate = torch.nn.functional.softsign(gate) + decay_out = out * (1.0 - ss_gate.abs()).square() + ctx.save_for_backward(ss_gate, decay_out) + return out * ss_gate + + @staticmethod + def backward(ctx, grad_output): + ss_gate, decay_out = ctx.saved_tensors + return grad_output * ss_gate, grad_output * decay_out + + +class SoftSignGLU(nn.Module): + """Gated Linear Unit with SoftSign gate: out * softsign(gate). + + More numerically stable than ATanGLU (no approximation needed in + Triton kernels) while providing similar gating behavior. + """ + def __init__(self, dim=-1): + super().__init__() + self.dim = dim + + def forward(self, x): + out, gate = torch.split(x, x.size(self.dim) // 2, dim=self.dim) + if self.training: + return SoftSignGLUFunction.apply(out, gate) + else: + return out * torch.nn.functional.softsign(gate) + + class AdamWConv1d(torch.nn.Conv1d): def __init__(self, *args, **kwargs): super().__init__(*args, **kwargs) diff --git a/modules/kernels/__init__.py b/modules/kernels/__init__.py new file mode 100644 index 000000000..d58de6003 --- /dev/null +++ b/modules/kernels/__init__.py @@ -0,0 +1 @@ +# Fused kernels for LYNXNet2 optimization \ No newline at end of file diff --git a/modules/kernels/fused_linear_softsign_glu.py b/modules/kernels/fused_linear_softsign_glu.py new file mode 100644 index 000000000..b66ce2ebb --- /dev/null +++ b/modules/kernels/fused_linear_softsign_glu.py @@ -0,0 +1,482 @@ +"""Fused Linear + SoftSignGLU for LYNXNet2. + +Computes ``left * softsign(gate)`` directly from a Linear layer whose output +weights are split into left and gate halves. The forward kernel uses fp32 +accumulators, and the backward path uses a Triton element-wise kernel followed +by cuBLAS matrix multiplications. +""" +import torch +import torch.nn.functional as F + +try: + import triton + import triton.language as tl + _TRITON_AVAILABLE = True +except ImportError: # no triton installed (e.g. Windows without the build) + _TRITON_AVAILABLE = False + + +# Minimum CUDA compute capability for Triton tl.dot (tensor cores). +# Volta (sm_70) is the floor; Turing sm_75 has fp16 tensor cores, Ampere +# sm_80 adds bf16. Anything older cannot run the fused kernel. +_MIN_CAPABILITY = (7, 0) + +_FUSED_CAPABLE = None + + +def is_triton_available(): + """Whether the Triton Python package imported successfully.""" + return _TRITON_AVAILABLE + + +def _fused_capable(): + """True only if the current CUDA device can run the fused kernel. + + Caches once per process. Returns False when Triton is missing, the + device is CPU, or the device's compute capability predates tensor cores. + """ + global _FUSED_CAPABLE + if _FUSED_CAPABLE is None: + _FUSED_CAPABLE = False + if _TRITON_AVAILABLE and torch.cuda.is_available(): + cap = torch.cuda.get_device_capability() + if cap >= _MIN_CAPABILITY: + _FUSED_CAPABLE = True + return _FUSED_CAPABLE + + +# --------------------------------------------------------------------------- +# Forward kernel +# --------------------------------------------------------------------------- + +if _TRITON_AVAILABLE: + + @triton.autotune( + configs=[ + # Small tiles + triton.Config({'BLOCK_M': 32, 'BLOCK_N': 32, 'BLOCK_K': 32, 'GROUP_M': 8}, num_warps=4, num_stages=3), + triton.Config({'BLOCK_M': 32, 'BLOCK_N': 64, 'BLOCK_K': 32, 'GROUP_M': 8}, num_warps=4, num_stages=3), + triton.Config({'BLOCK_M': 64, 'BLOCK_N': 32, 'BLOCK_K': 32, 'GROUP_M': 8}, num_warps=4, num_stages=3), + triton.Config({'BLOCK_M': 64, 'BLOCK_N': 64, 'BLOCK_K': 32, 'GROUP_M': 8}, num_warps=4, num_stages=3), + # Larger tiles; unsupported shared-memory sizes are pruned by Triton. + triton.Config({'BLOCK_M': 64, 'BLOCK_N': 64, 'BLOCK_K': 64, 'GROUP_M': 8}, num_warps=4, num_stages=4), + triton.Config({'BLOCK_M': 128, 'BLOCK_N': 64, 'BLOCK_K': 32, 'GROUP_M': 8}, num_warps=4, num_stages=4), + triton.Config({'BLOCK_M': 64, 'BLOCK_N': 128, 'BLOCK_K': 32, 'GROUP_M': 8}, num_warps=4, num_stages=4), + triton.Config({'BLOCK_M': 128, 'BLOCK_N': 128, 'BLOCK_K': 32, 'GROUP_M': 8}, num_warps=8, num_stages=3), + triton.Config({'BLOCK_M': 128, 'BLOCK_N': 64, 'BLOCK_K': 64, 'GROUP_M': 8}, num_warps=8, num_stages=3), + ], + key=['M_BUCKET', 'N', 'K'], + ) + @triton.jit + def _fused_linear_softsign_glu_fwd_kernel( + x_ptr, w_left_ptr, w_right_ptr, b_left_ptr, b_right_ptr, + y_ptr, left_ptr, gate_ptr, + M, N, K, + M_BUCKET, # next_power_of_2(M) — autotune key only, not used in body + stride_x_b, stride_x_k, + stride_wl_n, stride_wl_k, + stride_wr_n, stride_wr_k, + stride_y_b, stride_y_n, + stride_l_b, stride_l_n, + stride_g_b, stride_g_n, + BLOCK_M: tl.constexpr, BLOCK_N: tl.constexpr, BLOCK_K: tl.constexpr, + GROUP_M: tl.constexpr, + ): + """ + y = (x @ W_left^T + b_left) * softsign(x @ W_right^T + b_right) + + N = output dim per GLU half (= inner_dim = dim × expansion_factor) + K = input feature dim (= dim for first Linear, inner_dim for second) + + 2D grid over (M // BLOCK_M, N // BLOCK_N) with grouped ordering: + programs are swizzled so that GROUP_M row-blocks share column tiles + while they are still hot in L2 (standard Triton matmul swizzle). + """ + pid = tl.program_id(0) + num_pid_m = tl.cdiv(M, BLOCK_M) + num_pid_n = tl.cdiv(N, BLOCK_N) + # Grouped pid swizzle for L2 reuse + num_pid_in_group = GROUP_M * num_pid_n + group_id = pid // num_pid_in_group + first_pid_m = group_id * GROUP_M + group_size_m = tl.minimum(num_pid_m - first_pid_m, GROUP_M) + pid_m = first_pid_m + ((pid % num_pid_in_group) % group_size_m) + pid_n = (pid % num_pid_in_group) // group_size_m + + offs_m = pid_m * BLOCK_M + tl.arange(0, BLOCK_M) + offs_n = pid_n * BLOCK_N + tl.arange(0, BLOCK_N) + offs_k = tl.arange(0, BLOCK_K) + + m_mask_2d = offs_m[:, None] < M + n_mask_nk = offs_n[:, None] < N # [BLOCK_N, 1] for N×K weight access + n_mask_mn = offs_n[None, :] < N # [1, BLOCK_N] for M×N output access + n_mask_1d = offs_n < N + + acc_left = tl.zeros([BLOCK_M, BLOCK_N], dtype=tl.float32) + acc_gate = tl.zeros([BLOCK_M, BLOCK_N], dtype=tl.float32) + + for k_start in range(0, K, BLOCK_K): + k_offs = k_start + offs_k + k_mask_2d = k_offs[None, :] < K + + x = tl.load( + x_ptr + offs_m[:, None] * stride_x_b + k_offs[None, :] * stride_x_k, + mask=m_mask_2d & k_mask_2d, other=0.0, + ) + wl = tl.load( + w_left_ptr + offs_n[:, None] * stride_wl_n + k_offs[None, :] * stride_wl_k, + mask=n_mask_nk & k_mask_2d, other=0.0, + ) + acc_left += tl.dot(x, wl.T) + + wr = tl.load( + w_right_ptr + offs_n[:, None] * stride_wr_n + k_offs[None, :] * stride_wr_k, + mask=n_mask_nk & k_mask_2d, other=0.0, + ) + acc_gate += tl.dot(x, wr.T) + + # Bias + b_left = tl.load(b_left_ptr + offs_n, mask=n_mask_1d, other=0.0) + b_right = tl.load(b_right_ptr + offs_n, mask=n_mask_1d, other=0.0) + acc_left += b_left + acc_gate += b_right + + # Computed in fp32 for numerical safety + gate_f32 = acc_gate.to(tl.float32) + ss_gate = gate_f32 / (1.0 + tl.abs(gate_f32)) + gated = acc_left * ss_gate + + # Write output y + tl.store( + y_ptr + offs_m[:, None] * stride_y_b + offs_n[None, :] * stride_y_n, + gated, mask=m_mask_2d & n_mask_mn, + ) + + # Save intermediates for backward + tl.store( + left_ptr + offs_m[:, None] * stride_l_b + offs_n[None, :] * stride_l_n, + acc_left, mask=m_mask_2d & n_mask_mn, + ) + tl.store( + gate_ptr + offs_m[:, None] * stride_g_b + offs_n[None, :] * stride_g_n, + acc_gate, mask=m_mask_2d & n_mask_mn, + ) + + + # --------------------------------------------------------------------------- + # Element-wise backward kernel — grad_left_pre, grad_gate + # + # GEMMs run through cuBLAS; Triton handles the element-wise gradient. + # --------------------------------------------------------------------------- + + @triton.autotune( + configs=[ + triton.Config({'BLOCK_M': 64, 'BLOCK_N': 64}, num_warps=4, num_stages=2), + triton.Config({'BLOCK_M': 128, 'BLOCK_N': 32}, num_warps=4, num_stages=2), + triton.Config({'BLOCK_M': 64, 'BLOCK_N': 128}, num_warps=4, num_stages=2), + triton.Config({'BLOCK_M': 128, 'BLOCK_N': 64}, num_warps=8, num_stages=2), + ], + key=['N'], # element-wise: tile choice is insensitive to M — never key on it + ) + @triton.jit + def _softsign_glu_bwd_elem_kernel( + left_ptr, gate_ptr, grad_y_ptr, + glp_ptr, gg_ptr, + M, N, + stride_l_b, stride_l_n, + stride_g_b, stride_g_n, + stride_gy_b, stride_gy_n, + stride_glp_b, stride_glp_n, + stride_gg_b, stride_gg_n, + BLOCK_M: tl.constexpr, BLOCK_N: tl.constexpr, + ): + """Element-wise SoftSignGLU backward. + + For y = l * softsign(g): + dy/dl = softsign(g) + dy/dg = l / (1+|g|)^2 + """ + pid = tl.program_id(0) + num_pid_m = tl.cdiv(M, BLOCK_M) + num_pid_n = tl.cdiv(N, BLOCK_N) + pid_m = pid // num_pid_n + pid_n = pid % num_pid_n + + offs_m = pid_m * BLOCK_M + tl.arange(0, BLOCK_M) + offs_n = pid_n * BLOCK_N + tl.arange(0, BLOCK_N) + + m_mask = offs_m[:, None] < M + n_mask = offs_n[None, :] < N + + left = tl.load(left_ptr + offs_m[:, None] * stride_l_b + offs_n[None, :] * stride_l_n, + mask=m_mask & n_mask, other=0.0) + gate = tl.load(gate_ptr + offs_m[:, None] * stride_g_b + offs_n[None, :] * stride_g_n, + mask=m_mask & n_mask, other=0.0) + gy = tl.load(grad_y_ptr + offs_m[:, None] * stride_gy_b + offs_n[None, :] * stride_gy_n, + mask=m_mask & n_mask, other=0.0) + + gate_f32 = gate.to(tl.float32) + left_f32 = left.to(tl.float32) + denom_g = 1.0 / (1.0 + tl.abs(gate_f32)) + denom_g2 = denom_g * denom_g + ss_gate = gate_f32 * denom_g + grad_left_pre = gy * ss_gate + grad_gate = gy * (left_f32 * denom_g2) + + tl.store(glp_ptr + offs_m[:, None] * stride_glp_b + offs_n[None, :] * stride_glp_n, + grad_left_pre, mask=m_mask & n_mask) + tl.store(gg_ptr + offs_m[:, None] * stride_gg_b + offs_n[None, :] * stride_gg_n, + grad_gate, mask=m_mask & n_mask) + + + # --------------------------------------------------------------------------- + # Python wrapper — torch.autograd.Function + # --------------------------------------------------------------------------- + + class FusedLinearSoftSignGLUFn(torch.autograd.Function): + """Fused Linear(2K, K) + SoftSignGLU.""" + + @staticmethod + def forward(ctx, x, weight, bias): + orig_shape = x.shape + K = weight.shape[1] # input feature dim (contraction dim) + N = weight.shape[0] // 2 # output dim per GLU half + x_2d = x.reshape(-1, K) + M = x_2d.shape[0] + + w_left, w_right = weight.split(N, dim=0) + if bias is not None: + b_left, b_right = bias.split(N, dim=0) + else: + b_left = b_right = None + + out = torch.empty(M, N, device=x.device, dtype=x.dtype) + left = torch.empty(M, N, device=x.device, dtype=x.dtype) + gate = torch.empty(M, N, device=x.device, dtype=x.dtype) + + def grid(meta): + return (triton.cdiv(M, meta['BLOCK_M']) * triton.cdiv(N, meta['BLOCK_N']),) + + _fused_linear_softsign_glu_fwd_kernel[grid]( + x_2d, w_left, w_right, b_left, b_right, + out, left, gate, + M, N, K, + triton.next_power_of_2(M), # M_BUCKET: bounds autotune re-runs under variable batch frame counts + x_2d.stride(0), x_2d.stride(1), + w_left.stride(0), w_left.stride(1), + w_right.stride(0), w_right.stride(1), + out.stride(0), out.stride(1), + left.stride(0), left.stride(1), + gate.stride(0), gate.stride(1), + ) + + if x.dim() != 2: + out = out.view(*orig_shape[:-1], N) + + ctx.save_for_backward(x_2d, weight, left, gate) + ctx.orig_x_shape = orig_shape + ctx.N = N + return out + + @staticmethod + def backward(ctx, grad_y): + x, weight, left, gate = ctx.saved_tensors + M, K = x.shape + N = ctx.N + w_left, w_right = weight.split(N, dim=0) + + if grad_y.dim() != 2: + grad_y = grad_y.reshape(-1, N) + if not grad_y.is_contiguous(): + grad_y = grad_y.contiguous() + + # Step 1: Fused element-wise GLU backward (single Triton kernel, + # grad_left_pre/grad_gate computed in registers, one HBM write each) + grad_left_pre = torch.empty(M, N, device=x.device, dtype=x.dtype) + grad_gate = torch.empty(M, N, device=x.device, dtype=x.dtype) + + def elem_grid(meta): + return (triton.cdiv(M, meta['BLOCK_M']) * triton.cdiv(N, meta['BLOCK_N']),) + + _softsign_glu_bwd_elem_kernel[elem_grid]( + left, gate, grad_y, + grad_left_pre, grad_gate, + M, N, + left.stride(0), left.stride(1), + gate.stride(0), gate.stride(1), + grad_y.stride(0), grad_y.stride(1), + grad_left_pre.stride(0), grad_left_pre.stride(1), + grad_gate.stride(0), grad_gate.stride(1), + ) + + # Step 2/3: All backward GEMMs on cuBLAS (faster than a Triton GEMM + # here, and preserves fp16/bf16/fp32 dtype without forced casts). + # grad_weight assembled without torch.cat: write both halves into one + # preallocated [2N, K] buffer via out= GEMMs. + grad_weight = torch.empty(2 * N, K, device=x.device, dtype=x.dtype) + torch.mm(grad_left_pre.T, x, out=grad_weight[:N]) + torch.mm(grad_gate.T, x, out=grad_weight[N:]) + grad_bias = torch.cat([grad_left_pre.sum(0), grad_gate.sum(0)], dim=0) + + # grad_x = grad_left_pre @ W_left + grad_gate @ W_right + grad_x = torch.mm(grad_left_pre, w_left) + grad_x.addmm_(grad_gate, w_right) + + if len(ctx.orig_x_shape) != 2: + grad_x = grad_x.view(*ctx.orig_x_shape) + + return grad_x, grad_weight, grad_bias + + +# --------------------------------------------------------------------------- +# Public API +# --------------------------------------------------------------------------- + +def _eager_linear_softsign_glu(x, weight, bias): + """Unfused reference path for unsupported dtypes and devices.""" + linear = F.linear(x, weight, bias) + left, gate = torch.split(linear, linear.shape[-1] // 2, dim=-1) + return left * F.softsign(gate) + + +_FUSED_SUPPORTED_DTYPES = None +_FUSED_FALLBACK_LOG = {} + + +def _fused_supported_dtypes(): + """Dtypes the fused kernel can run on the current GPU. + + fp16 tl.dot: all tensor-core GPUs (Volta sm_70 and newer). + bf16 tl.dot: Ampere (sm_80) and newer — RTX 4090 (sm_89) and + RTX 5090 (sm_120) are fine, but Turing debug GPUs (RTX 20xx) are not. + fp32 falls back to eager: training runs 16-mixed/bf16-mixed, and eager + fp32 keeps full precision without a tf32 surprise inside the kernel. + """ + global _FUSED_SUPPORTED_DTYPES + if _FUSED_SUPPORTED_DTYPES is None: + supported = {torch.float16} + if torch.cuda.get_device_capability() >= (8, 0): + supported.add(torch.bfloat16) + _FUSED_SUPPORTED_DTYPES = supported + return _FUSED_SUPPORTED_DTYPES + + +def fused_linear_softsign_glu(x, weight, bias): + """Fused Linear(C, 2*N) + SoftSignGLU. + + Supports expansion_factor != 1 by splitting weight at its midpoint. + Unsupported dtypes and devices use the eager implementation. + + Args: + x: Input [..., K]. + weight: Linear weight [2*N, K]. + bias: Linear bias [2*N] or None. + + Returns: + Output [..., N]. + """ + if not _TRITON_AVAILABLE or not _fused_capable(): + return _eager_linear_softsign_glu(x, weight, bias) + if not x.is_cuda: + return _eager_linear_softsign_glu(x, weight, bias) + if bias is None: + return _eager_linear_softsign_glu(x, weight, bias) + + fallback_key = (x.dtype, x.shape[-1]) + if x.dtype not in _fused_supported_dtypes(): + if _FUSED_FALLBACK_LOG.get(fallback_key, 0) < 1: + _FUSED_FALLBACK_LOG[fallback_key] = 1 + import warnings + warnings.warn( + f'Fused SoftSignGLU: dtype {x.dtype} not supported for this GPU; ' + f'falling back to eager. (This message is shown once per (dtype, K) pair.)', + stacklevel=2, + ) + return _eager_linear_softsign_glu(x, weight, bias) + # Match weight/bias dtype to input (handles 16-mixed precision where + # weights are fp32 but activations are autocast to fp16) + if weight.dtype != x.dtype: + weight = weight.to(x.dtype) + if bias.dtype != x.dtype: + bias = bias.to(x.dtype) + if not weight.is_contiguous(): + weight = weight.contiguous() + if not bias.is_contiguous(): + bias = bias.contiguous() + return FusedLinearSoftSignGLUFn.apply(x, weight, bias) + + +# --------------------------------------------------------------------------- +# Test +# --------------------------------------------------------------------------- + +def _test(): + """Numerical check against an fp32 reference.""" + import time + + torch.manual_seed(42) + device = 'cuda' + margin = 2.0 + + def rel_err(actual, reference): + return (actual.float() - reference).abs().max().item() / reference.abs().mean().item() + + for K in [256, 512, 1024]: + M = 4096 if K == 256 else (2048 if K == 512 else 1024) + x16 = torch.randn(M, K, device=device, dtype=torch.float16, requires_grad=True) + w16 = torch.randn(2 * K, K, device=device, dtype=torch.float16, requires_grad=True) + b16 = torch.randn(2 * K, device=device, dtype=torch.float16, requires_grad=True) + grad = torch.randn(M, K, device=device, dtype=torch.float16) + + x32 = x16.detach().float().requires_grad_(True) + w32 = w16.detach().float().requires_grad_(True) + b32 = b16.detach().float().requires_grad_(True) + l32, g32 = torch.split(F.linear(x32, w32, b32), K, dim=-1) + ref = l32 * F.softsign(g32) + ref.backward(grad.float()) + + l16, g16 = torch.split(F.linear(x16, w16, b16), K, dim=-1) + y_eager = l16 * F.softsign(g16) + y_eager.backward(grad) + eager_errors = ( + rel_err(y_eager, ref), + rel_err(x16.grad, x32.grad), + rel_err(w16.grad, w32.grad), + rel_err(b16.grad, b32.grad), + ) + + x16.grad = w16.grad = b16.grad = None + y_fused = fused_linear_softsign_glu(x16, w16, b16) + y_fused.backward(grad) + fused_errors = ( + rel_err(y_fused, ref), + rel_err(x16.grad, x32.grad), + rel_err(w16.grad, w32.grad), + rel_err(b16.grad, b32.grad), + ) + + for name, fused_error, eager_error in zip( + ('fwd', 'grad_x', 'grad_w', 'grad_b'), fused_errors, eager_errors + ): + assert fused_error <= eager_error * margin + 1e-6, ( + f'K={K} {name}: fused={fused_error:.4e} vs eager={eager_error:.4e}' + ) + + torch.cuda.synchronize() + t0 = time.time() + for _ in range(50): + fused_linear_softsign_glu(x16, w16, b16) + torch.cuda.synchronize() + fused_t = (time.time() - t0) / 50 + + print( + f"K={K:4d} fwd={fused_errors[0]:.2e} dx={fused_errors[1]:.2e} " + f"dw={fused_errors[2]:.2e} db={fused_errors[3]:.2e} " + f"t={fused_t * 1000:.2f}ms" + ) + + print("\nAll tests passed: fused error is within fp16 rounding of the eager path.") + + +if __name__ == '__main__': + _test() diff --git a/modules/kernels/integration.py b/modules/kernels/integration.py new file mode 100644 index 000000000..a30abf445 --- /dev/null +++ b/modules/kernels/integration.py @@ -0,0 +1,352 @@ +""" +Drop-in replacement for LYNXNet2Block with fused Linear+SoftSignGLU kernels. + +The fused kernel replaces: + nn.Linear(dim, inner_dim*2) + SoftSignGLU → one fused kernel call +(training mode only; eval mode uses the original nn.Sequential path). + +Only softsign_glu is supported — other GLU types are left unpatched +(warning at patch time, block runs the original forward). + +Numerical accuracy: + SoftSignGLU is exact in Triton (no approximation). Differences vs the + eager path are fp16 rounding only (~1e-3 max on unit-scale activations). + +HBM savings (per fused call, M=50000, N=1024, fp16): + Eager: Linear writes [M, 2N] (200 MB), GLU reads [M, 2N] + writes [M, N] + Fused: writes y/left/gate = 3×[M, N] — saves the [M, 2N] round-trip +Backward saves the softsign/denominator intermediates by fusing the +element-wise gradient into one kernel; all GEMMs stay on cuBLAS. + +ONNX export: + Use `model.eval()` → falls back to original path → ONNX export works +""" +import contextlib +import traceback +import warnings + +import torch +import torch.nn as nn + +from modules.backbones.lynxnet2 import LYNXNet2Block +from modules.commons.common_layers import Transpose +from modules.kernels.fused_linear_softsign_glu import ( + fused_linear_softsign_glu, + is_triton_available, +) + + +_FUSABLE_GLU_TYPES = ('softsign_glu',) + + +class _FusedLYNXNet2BlockMixin: + """Pickle-safe fused forward mixed into an existing LYNXNet2Block.""" + + def forward(self, x): + if not self.training: + return super().forward(x) + + residual = x + x = self.net[0](x) + x = self.net[1](x) + x = self.net[2](x) + x = self.net[3](x) + x = fused_linear_softsign_glu(x, self.net[4].weight, self.net[4].bias) + x = fused_linear_softsign_glu(x, self.net[6].weight, self.net[6].bias) + x = self.net[8](x) + x = self.net[9](x) + return x + residual + + +class FusedLYNXNet2Block(_FusedLYNXNet2BlockMixin, LYNXNet2Block): + """LYNXNet2Block variant with a pickle-safe fused training forward.""" + + +def wrap_lynxnet2_block(block, glu_type='softsign_glu'): + """Wrap an existing LYNXNet2Block to use fused forward. + + Keeps all weights in-place (state_dict compatible). + Only modifies the forward pass. + + Only 'softsign_glu' is fused. Other GLU types are returned unpatched. + + Args: + block: LYNXNet2Block instance + glu_type: GLU type configured for this block + + Returns: + The same block, with patched forward if glu_type is supported. + """ + if glu_type not in _FUSABLE_GLU_TYPES: + warnings.warn( + f"Fused kernels support only {_FUSABLE_GLU_TYPES}; leaving block " + f"with glu_type={glu_type!r} unpatched.", + stacklevel=2, + ) + return block + + net = block.net + if not ( + len(net) == 10 + and isinstance(net[1], Transpose) # channel-first + and isinstance(net[3], Transpose) # back to channel-last + and isinstance(net[2], nn.Conv1d) + and net[2].groups == net[2].in_channels # depthwise conv keeps `dim` channels + and isinstance(net[4], nn.Linear) + and isinstance(net[6], nn.Linear) + and isinstance(net[8], nn.Linear) + and net[4].out_features == 2 * net[6].in_features + and net[6].out_features == 2 * net[8].in_features + and net[8].out_features == net[4].in_features # round-trip to dim + ): + warnings.warn( + 'Unexpected LYNXNet2Block.net layout; leaving block unpatched.', + stacklevel=2, + ) + return block + + block.__class__ = FusedLYNXNet2Block + return block + + +def patch_lynxnet2_model(model, glu_type='softsign_glu'): + """Patch all LYNXNet2Blocks in a LYNXNet2 model. + + Args: + model: LYNXNet2 instance + glu_type: GLU type configured for the model (only softsign_glu fuses) + + Returns: + Number of blocks patched (0 if glu_type unsupported). + """ + if glu_type not in _FUSABLE_GLU_TYPES: + warnings.warn( + f"Fused kernels require glu_type in {_FUSABLE_GLU_TYPES}; " + f"got {glu_type!r}. Skipping patch.", + stacklevel=2, + ) + return 0 + if not is_triton_available(): + raise RuntimeError( + 'Fused kernels require a working Triton installation. ' + 'Install Triton for this platform or set use_fused_kernels=false.' + ) + patched = 0 + for i, layer in enumerate(model.residual_layers): + if isinstance(layer, LYNXNet2Block): + layer = wrap_lynxnet2_block(layer, glu_type=glu_type) + model.residual_layers[i] = layer + patched += isinstance(layer, FusedLYNXNet2Block) + return patched + + +# --------------------------------------------------------------------------- +# Safe patching — handles both DDPM (denoise_fn) and ReFlow (velocity_fn), +# and checks that the backbone is actually a LYNXNet2 before patching. +# --------------------------------------------------------------------------- + +def _patch_backbone_fn(backbone_fn, glu_type): + """Patch a single backbone function/module if it's a LYNXNet2. + + Args: + backbone_fn: The backbone module (e.g., diffusion.denoise_fn) + glu_type: GLU type (only softsign_glu fuses) + + Returns: + Number of blocks patched (0 if not a LYNXNet2). + """ + from modules.backbones.lynxnet2 import LYNXNet2 + if not isinstance(backbone_fn, LYNXNet2): + return 0 + return patch_lynxnet2_model(backbone_fn, glu_type=glu_type) + + +def _try_patch(module, attr, glu_type): + """Try to patch backbone at module.attr if it's a LYNXNet2. Safe to call + even if attr doesn't exist — returns 0 silently.""" + backbone = getattr(module, attr, None) + if backbone is None: + return 0 + return _patch_backbone_fn(backbone, glu_type) + + +def patch_diffusion_module(diffusion, glu_type='softsign_glu'): + """Patch a diffusion module's backbone (DDPM or ReFlow). + + Handles both: + GaussianDiffusion / PitchDiffusion / MultiVarianceDiffusion → .denoise_fn + RectifiedFlow / PitchRectifiedFlow / MultiVarianceRectifiedFlow → .velocity_fn + + Returns: + Number of blocks patched. + """ + return ( + _try_patch(diffusion, 'denoise_fn', glu_type) + + _try_patch(diffusion, 'velocity_fn', glu_type) + ) + + +# --------------------------------------------------------------------------- +# Warmup — trigger Triton autotune before training starts +# --------------------------------------------------------------------------- + +def warmup_fused_backbone(backbone, max_frames=None, autocast_dtype=None): + """Run dummy forward passes to trigger Triton autotune compilation + for all fused kernels (fwd + bwd elem). Call after patching, before + the first real training step (model must already be on its CUDA device). + + Only forward is executed (``torch.no_grad``) — the element-wise backward + kernel's autotune key depends on ``N`` (a single fixed value per model), + so its one-off compile cost is paid on the first real step instead. + + Autotune timings are cached in process memory only (Triton persists + compiled binaries to disk, but re-runs the config benchmark per process), + so this runs once per training process. The forward kernel's autotune + key buckets M by next_power_of_2, so we sweep the power-of-two buckets + a real run will hit: from a small bucket up to next_power_of_2(max_frames). + + Args: + backbone: LYNXNet2 model (already patched). + max_frames: max total frames per batch (hparams['max_batch_frames']). + If None, warms a single small bucket only. + autocast_dtype: torch.float16 for '16-mixed', torch.bfloat16 for + 'bf16-mixed'. If None, no autocast — with fp32 parameters the + fused path falls back to eager and the warmup is a no-op. + """ + device = next(backbone.parameters()).device + dtype = next(backbone.parameters()).dtype + + if device.type != 'cuda': + return 0 + if not is_triton_available(): + raise RuntimeError( + 'Fused kernel warmup requires a working Triton installation. ' + 'Install Triton for this platform or set use_fused_kernels=false.' + ) + + import triton + + # cond hidden size from the conditioner projection (Linear or Conv1d) + proj = backbone.conditioner_projection + hidden = getattr(proj, 'in_features', None) or proj.in_channels + + B = 4 + # Sweep M buckets: 2048 up to next_power_of_2(max_frames) + if max_frames is not None: + top = triton.next_power_of_2(int(max_frames)) + bucket = 2048 + t_list = [] + while bucket <= top: + # M = B * T lands in this bucket (M just above the previous bucket) + t_list.append(bucket // B // 2 + 1) + bucket *= 2 + else: + t_list = [500] + + ac_factory = ( + (lambda: torch.autocast(device_type=device.type, dtype=autocast_dtype)) + if autocast_dtype is not None else contextlib.nullcontext + ) + # Fork the RNG so dummy inputs do not advance the training noise stream. + with torch.random.fork_rng(devices=[device]): + for T in t_list: + # spec shape: [B, n_feats, in_dims, T] + spec = torch.randn(B, backbone.n_feats, backbone.in_dims, T, + device=device, dtype=dtype) + t = torch.randint(0, 1000, (B,), device=device).float() + cond = torch.randn(B, hidden, T, device=device, dtype=dtype) + + try: + with torch.no_grad(): + with ac_factory(): + backbone(spec, t, cond=cond) + except Exception as e: # noqa: BLE001 - warmup must remain non-fatal + # Autotune failure should not crash training — Triton cache + # can be built on the first real step instead. + warnings.warn( + f'Fused kernel warmup skipped at T={T} ' + f'({type(e).__name__}: {e})\n{traceback.format_exc()}', + stacklevel=2, + ) + break + finally: + del spec, cond + torch.cuda.empty_cache() + return len(t_list) + + +def warmup_fused_backbones(backbones, max_frames, precision): + """Warm all patched backbones using Lightning's effective precision.""" + precision = str(precision) + autocast_dtype = ( + torch.float16 if '16' in precision and 'bf16' not in precision + else torch.bfloat16 if 'bf16' in precision + else None + ) + if autocast_dtype is None: + from lightning.pytorch.utilities.rank_zero import rank_zero_info + rank_zero_info( + 'Fused kernels: precision=%s has no autocast dtype; ' + 'fused kernel will fall back to eager at runtime.', precision + ) + for backbone in backbones: + warmup_fused_backbone( + backbone, + max_frames=max_frames, + autocast_dtype=autocast_dtype, + ) + + +# --------------------------------------------------------------------------- +# Test +# --------------------------------------------------------------------------- + +def _test(): + import torch + from modules.backbones.lynxnet2 import LYNXNet2Block + + device = 'cuda' + torch.manual_seed(42) + + # Create a single block + block = LYNXNet2Block(dim=256, expansion_factor=1, glu_type='softsign_glu').to(device).half() + + # Copy weights + block_ref = LYNXNet2Block(dim=256, expansion_factor=1, glu_type='softsign_glu').to(device).half() + block_ref.load_state_dict(block.state_dict()) + + # Patch + wrap_lynxnet2_block(block, glu_type='softsign_glu') + + B, T = 2, 500 + x = torch.randn(B, T, 256, device=device, dtype=torch.float16) + + # Forward + out_orig = block_ref(x) + out_fused = block(x) + + fwd_diff = (out_fused - out_orig).abs().max().item() + print(f"Block forward max diff: {fwd_diff:.4e}") + + # Backward + grad = torch.randn_like(out_orig) + out_orig.backward(grad) + grads_ref = {n: p.grad.clone() for n, p in block_ref.named_parameters() if p.grad is not None} + + for p in block.parameters(): + p.grad = None + + out_fused = block(x) + out_fused.backward(grad) + grads_fused = {n: p.grad.clone() for n, p in block.named_parameters() if p.grad is not None} + + max_w_diff = max( + (grads_fused[n] - grads_ref[n]).abs().max().item() + for n in grads_ref + ) + print(f"Block weight grad max diff: {max_w_diff:.4e}") + print("\nIntegration works! Use model.eval() for ONNX export fallback.") + + +if __name__ == '__main__': + _test() diff --git a/training/acoustic_task.py b/training/acoustic_task.py index eb052bce0..a261097e4 100644 --- a/training/acoustic_task.py +++ b/training/acoustic_task.py @@ -96,6 +96,38 @@ def __init__(self): self.required_variances.append('tension') super()._finish_init() + # ── Fuse LYNXNet2 backbone kernels (in-place) ── + # Only SoftSignGLU backbones are patched. + self._fused_kernels_patched = 0 + if hparams.get('use_fused_kernels', False): + try: + from modules.kernels.integration import patch_diffusion_module + from lightning.pytorch.utilities.rank_zero import rank_zero_info + # NOTE: LYNXNet2 defaults to swiglu when glu_type is unset + self._fused_kernels_patched = patch_diffusion_module( + self.model.diffusion, + glu_type=hparams['backbone_args'].get('glu_type', 'swiglu'), + ) + rank_zero_info('Fused kernels: patched %d LYNXNet2 blocks', self._fused_kernels_patched) + except ImportError as e: + from lightning.pytorch.utilities.rank_zero import rank_zero_info + rank_zero_info('Fused kernels unavailable (ImportError: %s); running eager.', e) + + def on_fit_start(self): + # Warm Triton autotune caches after the model is on its CUDA device, + # so the first training steps don't pay the per-bucket benchmark cost. + if self._fused_kernels_patched > 0 and self.device.type == 'cuda': + from modules.kernels.integration import warmup_fused_backbones + backbones = [ + backbone for attr in ('denoise_fn', 'velocity_fn') + if (backbone := getattr(self.model.diffusion, attr, None)) is not None + ] + warmup_fused_backbones( + backbones, + max_frames=hparams['max_batch_frames'], + precision=self.trainer.precision, + ) + def _build_model(self): return DiffSingerAcoustic( vocab_size=len(self.phoneme_dictionary), diff --git a/training/variance_task.py b/training/variance_task.py index 646d9540a..032acfc72 100644 --- a/training/variance_task.py +++ b/training/variance_task.py @@ -115,6 +115,53 @@ def __init__(self): self.lambda_var_loss = hparams['lambda_var_loss'] super()._finish_init() + # ── Fuse LYNXNet2 backbone kernels (in-place) ── + self._fused_kernels_patched = 0 + self._fused_kernel_backbones = [] + if hparams.get('use_fused_kernels', False): + try: + from modules.backbones.lynxnet2 import LYNXNet2 + from modules.kernels.integration import patch_diffusion_module + from lightning.pytorch.utilities.rank_zero import rank_zero_info + # Each predictor has its own backbone config; patch only the ones + # actually configured with softsign_glu (others are skipped with + # a warning instead of silently changing their math). + # NOTE: LYNXNet2 defaults to swiglu when glu_type is unset. + for predictor_attr, args_key in ( + ('pitch_predictor', 'pitch_prediction_args'), + ('variance_predictor', 'variances_prediction_args'), + ): + predictor = getattr(self.model, predictor_attr, None) + if predictor is None: + continue + glu = (hparams.get(args_key) or {}).get('backbone_args', {}).get('glu_type', 'swiglu') + n = patch_diffusion_module(predictor, glu_type=glu) + self._fused_kernels_patched += n + if n > 0: + for attr in ('denoise_fn', 'velocity_fn'): + backbone = getattr(predictor, attr, None) + if isinstance(backbone, LYNXNet2): + self._fused_kernel_backbones.append(backbone) + rank_zero_info( + 'Fused kernels: patched %d LYNXNet2 blocks in %s (glu_type=%s)', + n, predictor_attr, glu + ) + except ImportError as e: + from lightning.pytorch.utilities.rank_zero import rank_zero_info + rank_zero_info('Fused kernels unavailable (ImportError: %s); running eager.', e) + + def on_fit_start(self): + # Warm Triton autotune caches after the model is on its CUDA device, + # so the first training steps don't pay the per-bucket benchmark cost. + # Mirrors AcousticTask.on_fit_start, but sweeps both predictors. + if self._fused_kernels_patched > 0 and self.device.type == 'cuda': + from modules.kernels.integration import warmup_fused_backbones + warmup_fused_backbones( + self._fused_kernel_backbones, + max_frames=hparams['max_batch_frames'], + precision=self.trainer.precision, + ) + def _build_model(self): return DiffSingerVariance( vocab_size=len(self.phoneme_dictionary), From 1802dfb87771ea57a173c37994e2693245f6d6db Mon Sep 17 00:00:00 2001 From: yxlllc <33565655+yxlllc@users.noreply.github.com> Date: Sun, 16 Aug 2026 14:45:51 +0800 Subject: [PATCH 07/23] some minor fixes (#322) * some minor fixes * some minor fixes * fix * correct the length calculation * fix preder_ds data loading * fix the off-by-one issue * fix naming error --- deployment/exporters/variance_exporter.py | 2 +- preprocessing/variance_binarizer.py | 11 ++++++----- utils/onnx_helper.py | 5 +++-- 3 files changed, 10 insertions(+), 8 deletions(-) diff --git a/deployment/exporters/variance_exporter.py b/deployment/exporters/variance_exporter.py index e8832ff12..381fe1173 100644 --- a/deployment/exporters/variance_exporter.py +++ b/deployment/exporters/variance_exporter.py @@ -691,7 +691,7 @@ def _optimize_merge_pitch_predictor_graph( assert check, 'Simplified ONNX model could not be validated' onnx_helper.model_override_io_shapes( - pitch_predictor, output_shapes={'pitch_pred': (1, 'n_frames')} + pitch_predictor, output_shapes={'x_pred': (1, 'n_frames')} ) print(f'Running ONNX Simplifier #1 on {self.pitch_predictor_class_name}...') pitch_predictor, check = onnxsim.simplify(pitch_predictor, include_subgraph=True) diff --git a/preprocessing/variance_binarizer.py b/preprocessing/variance_binarizer.py index 589bf571a..7b193d5ed 100644 --- a/preprocessing/variance_binarizer.py +++ b/preprocessing/variance_binarizer.py @@ -107,7 +107,7 @@ def load_attr_from_ds(self, ds_id, name, attr, idx=0): if not isinstance(ds, list): ds = [ds] self.cached_ds[cache_key] = ds - ds = ds[idx] + ds = ds[0] if cache_key == item_name_with_idx else ds[idx] return ds.get(attr) def load_meta_data(self, raw_data_dir: pathlib.Path, ds_id, spk, lang): @@ -117,11 +117,12 @@ def load_meta_data(self, raw_data_dir: pathlib.Path, ds_id, spk, lang): for utterance_label in csv.DictReader(f): utterance_label: dict item_name = utterance_label['name'] - item_idx = int(item_name.rsplit(DS_INDEX_SEP, maxsplit=1)[-1]) if DS_INDEX_SEP in item_name else 0 + item_base_name, *item_seg_idx = item_name.rsplit(DS_INDEX_SEP, maxsplit=1) + item_idx = int(item_seg_idx[0]) if item_seg_idx else 0 def require(attr, optional=False): if self.prefer_ds: - value = self.load_attr_from_ds(ds_id, item_name, attr, item_idx) + value = self.load_attr_from_ds(ds_id, item_base_name, attr, item_idx) else: value = None if value is None: @@ -275,7 +276,6 @@ def process_item(self, item_name, meta_data, binarization_args): ds_id = int(ds_id) ds_seg_idx = meta_data['ds_idx'] seconds = sum(meta_data['ph_dur']) - length = round(seconds / self.timestep) T_ph = len(meta_data['ph_seq']) processed_input = { 'name': item_name, @@ -283,7 +283,6 @@ def process_item(self, item_name, meta_data, binarization_args): 'spk_id': meta_data['spk_id'], 'spk_name': meta_data['spk_name'], 'seconds': seconds, - 'length': length, 'languages': np.array(meta_data['lang_seq'], dtype=np.int64), 'tokens': np.array(meta_data['ph_seq'], dtype=np.int64), 'ph_text': meta_data['ph_text'], @@ -292,7 +291,9 @@ def process_item(self, item_name, meta_data, binarization_args): ph_dur_sec = torch.FloatTensor(meta_data['ph_dur']).to(self.device) ph_acc = torch.round(torch.cumsum(ph_dur_sec, dim=0) / self.timestep + 0.5).long() ph_dur = torch.diff(ph_acc, dim=0, prepend=torch.LongTensor([0]).to(self.device)) + length = int(ph_acc[-1]) processed_input['ph_dur'] = ph_dur.cpu().numpy() + processed_input['length'] = length mel2ph = get_mel2ph_torch( self.lr, ph_dur_sec, length, self.timestep, device=self.device diff --git a/utils/onnx_helper.py b/utils/onnx_helper.py index 5fddcfe49..bcd2c6a18 100644 --- a/utils/onnx_helper.py +++ b/utils/onnx_helper.py @@ -80,7 +80,7 @@ def model_reorder_io_list( :param model: model to perform the operation on :param input_or_output: 'input' or 'output' to specify the list to reorder :param target_name: the name of the input to be reordered - :param insert_after_name: the name of the input to be inserted after (None for the first) + :param insert_after_name: the name of the input to be inserted after """ def _reorder_input(input_list: RepeatedCompositeFieldContainer[ValueInfoProto]): nonlocal input_or_output @@ -93,7 +93,8 @@ def _reorder_input(input_list: RepeatedCompositeFieldContainer[ValueInfoProto]): insert_after_idx = i if target_idx != -1 and insert_after_idx != -1: target = input_list.pop(target_idx) - input_list.insert(insert_after_idx + 1, target) + insert_at = insert_after_idx + (1 if target_idx > insert_after_idx else 0) + input_list.insert(insert_at, target) _verbose(f'| reorder {input_or_output}: \'{target_name}\' after \'{insert_after_name}\'') if input_or_output == 'input': From 1b0a6da195646cdc4f8051d7afcc40e16960446c Mon Sep 17 00:00:00 2001 From: yxlllc <33565655+yxlllc@users.noreply.github.com> Date: Wed, 19 Aug 2026 16:43:09 +0800 Subject: [PATCH 08/23] some minor fixes (#325) * some minor fixes * some minor fixes * fix * correct the length calculation * fix preder_ds data loading * fix the off-by-one issue * fix naming error * fill zero-frame phones with correct pitch values * fill zero-frame words with correct pitch values --- inference/ds_variance.py | 18 ++++++++++++++++++ preprocessing/variance_binarizer.py | 6 ++++++ 2 files changed, 24 insertions(+) diff --git a/inference/ds_variance.py b/inference/ds_variance.py index f2821e2e9..767c43a4b 100644 --- a/inference/ds_variance.py +++ b/inference/ds_variance.py @@ -229,12 +229,30 @@ def preprocess_input( ph_midi = frame_midi_pitch.new_zeros(1, T_ph + 1).scatter_add( 1, mel2ph, frame_midi_pitch / mel2pdur )[:, 1:] + # Phones collapsed to 0 frames receive no scatter contribution; + # fall back to the pitch value at their position on the time axis. + zero_dur = ph_dur <= 0 + if zero_dur.any(): + dur_cum = torch.cumsum(ph_dur, dim=1) + bound = torch.cat( + [dur_cum.new_zeros(dur_cum.shape[0], 1), dur_cum[:, :-1]], dim=1 + ).clamp(min=0, max=T_s - 1) + ph_midi = torch.where(zero_dur, torch.gather(frame_midi_pitch, 1, bound), ph_midi) else: # Phone durations are not available, calculate word-level MIDI instead. mel2wdur = torch.gather(F.pad(word_dur, [1, 0], value=1), 1, mel2word) w_midi = frame_midi_pitch.new_zeros(1, T_w + 1).scatter_add( 1, mel2word, frame_midi_pitch / mel2wdur )[:, 1:] + # Words collapsed to 0 frames receive no scatter contribution; + # fall back to the pitch value at their position on the time axis. + zero_dur = word_dur <= 0 + if zero_dur.any(): + dur_cum = torch.cumsum(word_dur, dim=1) + bound = torch.cat( + [dur_cum.new_zeros(dur_cum.shape[0], 1), dur_cum[:, :-1]], dim=1 + ).clamp(min=0, max=T_s - 1) + w_midi = torch.where(zero_dur, torch.gather(frame_midi_pitch, 1, bound), w_midi) # Convert word-level MIDI to phoneme-level MIDI ph_midi = torch.gather(F.pad(w_midi, [1, 0]), 1, ph2word) ph_midi = ph_midi.round().long() diff --git a/preprocessing/variance_binarizer.py b/preprocessing/variance_binarizer.py index 7b193d5ed..5cc14f3c8 100644 --- a/preprocessing/variance_binarizer.py +++ b/preprocessing/variance_binarizer.py @@ -349,6 +349,12 @@ def process_item(self, item_name, meta_data, binarization_args): ph_midi = pitch.new_zeros(T_ph + 1).scatter_add( 0, mel2ph, pitch / mel2dur )[1:] + # Phones collapsed to 0 frames receive no scatter contribution; + # fall back to the pitch value at their position on the time axis. + zero_dur = ph_dur <= 0 + if zero_dur.any(): + bound = torch.cat([ph_acc.new_zeros(1), ph_acc[:-1]]).clamp(min=0, max=length - 1) + ph_midi[zero_dur] = pitch[bound[zero_dur]] processed_input['midi'] = ph_midi.round().long().clamp(min=0, max=127).cpu().numpy() if hparams['predict_pitch']: From 1ee5eb8969ec4b9dc245666adf187e8b132e5edb Mon Sep 17 00:00:00 2001 From: yxlllc <33565655+yxlllc@users.noreply.github.com> Date: Tue, 25 Aug 2026 18:42:49 +0800 Subject: [PATCH 09/23] some minor fixes (#329) * some minor fixes * some minor fixes * fix * correct the length calculation * fix preder_ds data loading * fix the off-by-one issue * fix naming error * fill zero-frame phones with correct pitch values * fill zero-frame words with correct pitch values * union-find over phonemes * remove dead code from prefix matching --- basics/base_binarizer.py | 18 ----------------- utils/phoneme_utils.py | 42 +++++++++++++++++++++++++--------------- 2 files changed, 26 insertions(+), 34 deletions(-) diff --git a/basics/base_binarizer.py b/basics/base_binarizer.py index 397bd8305..ac24fa83a 100644 --- a/basics/base_binarizer.py +++ b/basics/base_binarizer.py @@ -121,15 +121,6 @@ def split_train_valid_set(self, prefixes: list): if prefix in self.item_names: valid_item_names[prefix] = 1 prefixes.pop(prefix) - # Add prefixes that exactly matches item name without speaker id to test set - for prefix in deepcopy(prefixes): - matched = False - for name in self.item_names: - if name.split(':')[-1] == prefix: - valid_item_names[name] = 1 - matched = True - if matched: - prefixes.pop(prefix) # Add names with one of the remaining prefixes to test set for prefix in deepcopy(prefixes): matched = False @@ -139,15 +130,6 @@ def split_train_valid_set(self, prefixes: list): matched = True if matched: prefixes.pop(prefix) - for prefix in deepcopy(prefixes): - matched = False - for name in self.item_names: - if name.split(':')[-1].startswith(prefix): - valid_item_names[name] = 1 - matched = True - if matched: - prefixes.pop(prefix) - if len(prefixes) != 0: warnings.warn( f'The following rules in test_prefixes have no matching names in the dataset: {", ".join(prefixes.keys())}', diff --git a/utils/phoneme_utils.py b/utils/phoneme_utils.py index 8b2a5ad2d..2d8be5518 100644 --- a/utils/phoneme_utils.py +++ b/utils/phoneme_utils.py @@ -79,19 +79,34 @@ def __init__( _merged_groups.append(_group) merged_groups = [set(phones) for phones in _merged_groups if len(phones) > 1] # Step 3: Build phoneme index - merged_phonemes_inverted_index = {} - for idx, group in enumerate(merged_groups): - other_idx = None + # Union-find over phonemes: groups must merge transitively. + _uf_parent = {} + + def _uf_find(p): + _uf_parent.setdefault(p, p) + while _uf_parent[p] != p: + _uf_parent[p] = _uf_parent[_uf_parent[p]] + p = _uf_parent[p] + return p + + for group in merged_groups: + anchor = next(iter(group)) for phoneme in group: - if phoneme in merged_phonemes_inverted_index: - other_idx = merged_phonemes_inverted_index[phoneme] - break - target_idx = idx if other_idx is None else other_idx + _uf_parent[_uf_find(phoneme)] = _uf_find(anchor) + + _uf_classes = {} + for group in merged_groups: for phoneme in group: - merged_phonemes_inverted_index[phoneme] = target_idx - if other_idx is not None: - merged_groups[other_idx] |= group - group.clear() + root = _uf_find(phoneme) + if root not in _uf_classes: + _uf_classes[root] = set() + _uf_classes[root].add(phoneme) + merged_groups = list(_uf_classes.values()) + merged_phonemes_inverted_index = { + phoneme: group_idx + for group_idx, group in enumerate(merged_groups) + for phoneme in group + } phone_to_id = {} id_to_phone = [] cross_lingual_phonemes = set() @@ -174,12 +189,7 @@ def dump(self, filename): json.dump(self._phone_to_id, fp, ensure_ascii=False, indent=2) -_dictionary = None - - def load_phoneme_dictionary() -> PhonemeDictionary: - if _dictionary is not None: - return _dictionary config_dicts = hparams.get('dictionaries') if config_dicts is not None: dicts = {} From d9f8d64666222eaad5a81bc05efaeaed9f92e9c0 Mon Sep 17 00:00:00 2001 From: Kakaru <97896816+KakaruHayate@users.noreply.github.com> Date: Sat, 29 Aug 2026 22:26:25 +0800 Subject: [PATCH 10/23] Expose RoPE theta via config API; cache fp32 cos/sin tables (#320) * feat: expose RoPE theta via config API * perf(rope): cache fp32 cos/sin tables, cast at use, drop einops * chore: remove unused einops dependency --- configs/acoustic.yaml | 1 + configs/templates/config_acoustic.yaml | 1 + configs/templates/config_variance.yaml | 1 + configs/variance.yaml | 1 + modules/commons/rotary_embedding_torch.py | 54 +++++++++++------------ modules/fastspeech/acoustic_encoder.py | 1 + modules/fastspeech/tts_modules.py | 6 ++- modules/fastspeech/variance_encoder.py | 6 ++- requirements.txt | 1 - 9 files changed, 39 insertions(+), 33 deletions(-) diff --git a/configs/acoustic.yaml b/configs/acoustic.yaml index 6ae4c7e78..901aeccfc 100644 --- a/configs/acoustic.yaml +++ b/configs/acoustic.yaml @@ -70,6 +70,7 @@ max_beta: 0.02 enc_ffn_kernel_size: 3 use_rope: true rope_interleaved: false +rope_theta: 10000 use_stretch_embed: true use_variance_scaling: true rel_pos: true diff --git a/configs/templates/config_acoustic.yaml b/configs/templates/config_acoustic.yaml index e344fb450..02ff55b6e 100644 --- a/configs/templates/config_acoustic.yaml +++ b/configs/templates/config_acoustic.yaml @@ -75,6 +75,7 @@ diffusion_type: reflow enc_ffn_kernel_size: 3 use_rope: true rope_interleaved: false +rope_theta: 10000 use_stretch_embed: true use_variance_scaling: true use_shallow_diffusion: true diff --git a/configs/templates/config_variance.yaml b/configs/templates/config_variance.yaml index 116154ac7..d3acd51a6 100644 --- a/configs/templates/config_variance.yaml +++ b/configs/templates/config_variance.yaml @@ -66,6 +66,7 @@ tension_logit_max: 10.0 enc_ffn_kernel_size: 3 use_rope: true rope_interleaved: false +rope_theta: 10000 use_stretch_embed: false use_variance_scaling: true hidden_size: 384 diff --git a/configs/variance.yaml b/configs/variance.yaml index 10f90d0f9..0ec4a63fe 100644 --- a/configs/variance.yaml +++ b/configs/variance.yaml @@ -40,6 +40,7 @@ predict_tension: false enc_ffn_kernel_size: 3 use_rope: true rope_interleaved: false +rope_theta: 10000 use_stretch_embed: false use_variance_scaling: true rel_pos: true diff --git a/modules/commons/rotary_embedding_torch.py b/modules/commons/rotary_embedding_torch.py index 1a1fa193e..703bf940f 100644 --- a/modules/commons/rotary_embedding_torch.py +++ b/modules/commons/rotary_embedding_torch.py @@ -1,29 +1,22 @@ import torch -from einops import rearrange, repeat -from torch import einsum, Tensor +from torch import Tensor from torch.nn import Module -def rotate_half(x: Tensor, interleaved=True) -> Tensor: - if not interleaved: - # x_half1, x_half2 = x.chunk(2, dim=-1) - # Using torch.split instead of chunk for ONNX export compatibility. - x1, x2 = torch.split(x, x.size(-1) // 2, dim=-1) - return torch.cat((-x2, x1), dim=-1) - else: - x = rearrange(x, '... (d r) -> ... d r', r=2) - x1, x2 = x.unbind(dim=-1) - x = torch.stack((-x2, x1), dim=-1) - return rearrange(x, '... d r -> ... (d r)') - - -def apply_rotary_emb(freqs: Tensor, t: Tensor, interleaved=True) -> Tensor: - rot_dim = freqs.shape[-1] +def apply_rotary_emb(freqs_cos: Tensor, freqs_sin: Tensor, t: Tensor, interleaved=True) -> Tensor: + rot_dim = freqs_cos.shape[-1] t_to_rotate = t[..., :rot_dim] t_pass_through = t[..., rot_dim:] - t_rotated = (t_to_rotate * freqs.cos()) + (rotate_half(t_to_rotate, interleaved) * freqs.sin()) + if interleaved: + x = t_to_rotate.view(*t_to_rotate.shape[:-1], t_to_rotate.size(-1) // 2, 2) + x1, x2 = x.unbind(dim=-1) + rotated_half = torch.stack((-x2, x1), dim=-1).reshape_as(t_to_rotate) + else: + x1, x2 = torch.split(t_to_rotate, t_to_rotate.size(-1) // 2, dim=-1) + rotated_half = torch.cat((-x2, x1), dim=-1) + t_rotated = (t_to_rotate * freqs_cos) + (rotated_half * freqs_sin) return torch.cat((t_rotated, t_pass_through), dim=-1) @@ -40,24 +33,29 @@ def __init__( self.cached_freqs_seq_len = max_seq_len inv_freq = 1. / (theta ** (torch.arange(0, dim, 2).float() / dim)) self.register_buffer('inv_freq', inv_freq, persistent=False) - self.register_buffer('cached_freqs', self._precompute_cache(max_seq_len), persistent=False) + cos, sin = self._precompute_cache(max_seq_len) + self.register_buffer('cached_cos', cos, persistent=False) + self.register_buffer('cached_sin', sin, persistent=False) def _precompute_cache(self, seq_len: int): - seq = torch.arange(seq_len, device=self.inv_freq.device, dtype=self.inv_freq.dtype) - freqs = einsum('i, j -> i j', seq, self.inv_freq) + # Cache fp32 cos/sin, cast only at use — fp16/bf16 training must not + # recompute trig on low-precision angles. + seq = torch.arange(seq_len, device=self.inv_freq.device, dtype=torch.float32) + freqs = torch.einsum('i, j -> i j', seq, self.inv_freq.float()) if self.interleaved: - freqs = repeat(freqs, '... n -> ... (n r)', r=2) + freqs = torch.repeat_interleave(freqs, 2, dim=-1) else: freqs = torch.cat((freqs, freqs), dim=-1) - return freqs + return torch.cos(freqs), torch.sin(freqs) - def forward(self, seq_len: int) -> Tensor: + def forward(self, seq_len: int): if seq_len > self.cached_freqs_seq_len: raise RuntimeError("sequence exceeds RoPE max_seq_len!") - return self.cached_freqs[0: seq_len].detach() + return self.cached_cos[0: seq_len].detach(), self.cached_sin[0: seq_len].detach() def rotate_queries_or_keys(self, t: Tensor) -> Tensor: device, dtype, seq_len = t.device, t.dtype, t.shape[-2] - freqs = self.forward(seq_len=seq_len) - - return apply_rotary_emb(freqs.to(device=device, dtype=dtype), t, self.interleaved) + freqs_cos, freqs_sin = self.forward(seq_len=seq_len) + freqs_cos = freqs_cos.to(device=device, dtype=dtype) + freqs_sin = freqs_sin.to(device=device, dtype=dtype) + return apply_rotary_emb(freqs_cos, freqs_sin, t, self.interleaved) diff --git a/modules/fastspeech/acoustic_encoder.py b/modules/fastspeech/acoustic_encoder.py index 241f9871c..b960c5a25 100644 --- a/modules/fastspeech/acoustic_encoder.py +++ b/modules/fastspeech/acoustic_encoder.py @@ -58,6 +58,7 @@ def __init__(self, vocab_size): dropout=hparams['dropout'], num_heads=hparams['num_heads'], use_pos_embed=hparams['use_pos_embed'], rel_pos=hparams.get('rel_pos', False), use_rope=hparams.get('use_rope', False), rope_interleaved=hparams.get('rope_interleaved', True), + rope_theta=hparams.get('rope_theta', 10000), mix_ln_layer=self.mix_ln_layer ) diff --git a/modules/fastspeech/tts_modules.py b/modules/fastspeech/tts_modules.py index 10f774156..2b1549956 100644 --- a/modules/fastspeech/tts_modules.py +++ b/modules/fastspeech/tts_modules.py @@ -373,7 +373,7 @@ def __init__( self, hidden_size, num_layers, ffn_kernel_size=9, ffn_act='gelu', dropout=None, num_heads=2, use_pos_embed=True, rel_pos=True, - use_rope=False, rope_interleaved=True, mix_ln_layer=None + use_rope=False, rope_interleaved=True, rope_theta=10000, mix_ln_layer=None ): super().__init__() self.num_layers = num_layers @@ -386,7 +386,9 @@ def __init__( "RoPE requires the hidden size to be multiple of " f"num_heads * 2 = {num_heads * 2}, but got {embed_dim}." ) - rotary_embed = RotaryEmbedding(dim=embed_dim // num_heads, interleaved=rope_interleaved) + rotary_embed = RotaryEmbedding( + dim=embed_dim // num_heads, theta=rope_theta, interleaved=rope_interleaved + ) else: rotary_embed = None self.layers = nn.ModuleList([ diff --git a/modules/fastspeech/variance_encoder.py b/modules/fastspeech/variance_encoder.py index 712964846..6e0aa79a2 100644 --- a/modules/fastspeech/variance_encoder.py +++ b/modules/fastspeech/variance_encoder.py @@ -34,7 +34,8 @@ def __init__(self, vocab_size): ffn_kernel_size=hparams['enc_ffn_kernel_size'], ffn_act=hparams['ffn_act'], dropout=hparams['dropout'], num_heads=hparams['num_heads'], use_pos_embed=hparams['use_pos_embed'], rel_pos=hparams.get('rel_pos', False), - use_rope=hparams.get('use_rope', False), rope_interleaved=hparams.get('rope_interleaved', True) + use_rope=hparams.get('use_rope', False), rope_interleaved=hparams.get('rope_interleaved', True), + rope_theta=hparams.get('rope_theta', 10000) ) dur_hparams = hparams['dur_prediction_args'] @@ -128,7 +129,8 @@ def get_hparam(key): ffn_kernel_size=get_hparam('enc_ffn_kernel_size'), ffn_act=get_hparam('ffn_act'), dropout=get_hparam('dropout'), num_heads=get_hparam('num_heads'), use_pos_embed=get_hparam('use_pos_embed'), rel_pos=get_hparam('rel_pos'), - use_rope=get_hparam('use_rope'), rope_interleaved=hparams.get('rope_interleaved', True) + use_rope=get_hparam('use_rope'), rope_interleaved=hparams.get('rope_interleaved', True), + rope_theta=hparams.get('rope_theta', 10000) ) self.out_proj = Linear(hidden_size, hparams['hidden_size']) diff --git a/requirements.txt b/requirements.txt index 4645417f7..1a98f7951 100644 --- a/requirements.txt +++ b/requirements.txt @@ -3,7 +3,6 @@ # See instructions at https://pytorch.org/get-started/locally/ click -einops>=0.7.0 h5py librosa<0.10.0 lightning~=2.3.0 From 4975e19fce99c9ccca67ce95100806fd67439c23 Mon Sep 17 00:00:00 2001 From: yxlllc <33565655+yxlllc@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:23:48 +0800 Subject: [PATCH 11/23] some minor fixes (#330) * some minor fixes * some minor fixes * fix * correct the length calculation * fix preder_ds data loading * fix the off-by-one issue * fix naming error * fill zero-frame phones with correct pitch values * fill zero-frame words with correct pitch values * union-find over phonemes * remove dead code from prefix matching * fix the behavior of lang_seq under prefer_ds * update BestPractices.md * update GettingStarted.md * remove unused shuffle operations from preprocessing * update the template vocoder_ckpt to match the parent * update ConfigurationSchemas.md * update README.md --- README.md | 8 +- basics/base_binarizer.py | 4 - configs/acoustic.yaml | 1 - configs/base.yaml | 1 - configs/templates/config_acoustic.yaml | 2 +- configs/variance.yaml | 1 - docs/BestPractices.md | 137 +++-- docs/ConfigurationSchemas.md | 719 +++++++++++++------------ docs/GettingStarted.md | 22 +- preprocessing/variance_binarizer.py | 2 +- 10 files changed, 461 insertions(+), 436 deletions(-) diff --git a/README.md b/README.md index 624673ac3..b7cd4dde0 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,6 @@ # DiffSinger (OpenVPI maintained version) -[![arXiv](https://img.shields.io/badge/arXiv-Paper-.svg)](https://arxiv.org/abs/2105.02446) +[![arXiv](https://img.shields.io/badge/arXiv-Paper-b31b1b.svg)](https://arxiv.org/abs/2105.02446) [![downloads](https://img.shields.io/github/downloads/openvpi/DiffSinger/total.svg)](https://github.com/openvpi/DiffSinger/releases) [![Bilibili](https://img.shields.io/badge/Bilibili-Demo-blue)](https://www.bilibili.com/video/BV1be411N7JA/) [![license](https://img.shields.io/badge/License-Apache%202.0-blue.svg)](https://github.com/openvpi/DiffSinger/blob/main/LICENSE) @@ -8,10 +8,10 @@ This is a refactored and enhanced version of _DiffSinger: Singing Voice Synthesis via Shallow Diffusion Mechanism_ based on the original [paper](https://arxiv.org/abs/2105.02446) and [implementation](https://github.com/MoonInTheRiver/DiffSinger), which provides: - Cleaner code structure: useless and redundant files are removed and the others are re-organized. -- Better sound quality: the sampling rate of synthesized audio are adapted to 44.1 kHz instead of the original 24 kHz. +- Better sound quality: the sampling rate of synthesized audio is adapted to 44.1 kHz instead of the original 24 kHz. - Higher fidelity: improved acoustic models and diffusion sampling acceleration algorithms are integrated. - More controllability: introduced variance models and parameters for prediction and control of pitch, energy, breathiness, etc. -- Production compatibility: functionalities are designed to match the requirements of production deployment and the SVS communities. +- Production compatibility: functionalities are designed to match the requirements of production deployment and the Singing Voice Synthesis (SVS) communities. | Overview | Variance Model | Acoustic Model | |:-------------------------------------------------------------------------------------:|:-------------------------------------------------------------------------------------:|:-------------------------------------------------------------------------------------:| @@ -69,7 +69,7 @@ TBD ## Disclaimer -Any organization or individual is prohibited from using any functionalities included in this repository to generate someone's speech without his/her consent, including but not limited to government leaders, political figures, and celebrities. If you do not comply with this item, you could be in violation of copyright laws. +Any organization or individual is prohibited from using any functionalities included in this repository to generate someone's voice without his/her consent, including but not limited to government leaders, political figures, and celebrities. If you do not comply with this item, you could be in violation of copyright laws. ## License diff --git a/basics/base_binarizer.py b/basics/base_binarizer.py index ac24fa83a..f7297b80f 100644 --- a/basics/base_binarizer.py +++ b/basics/base_binarizer.py @@ -1,7 +1,6 @@ import json import pathlib import pickle -import random import shutil import warnings from copy import deepcopy @@ -177,9 +176,6 @@ def process(self): self.item_names = sorted(list(self.items.keys())) self._train_item_names, self._valid_item_names = self.split_train_valid_set(test_prefixes) - if self.binarization_args['shuffle']: - random.shuffle(self.item_names) - self.binary_data_dir.mkdir(parents=True, exist_ok=True) # Copy spk_map, lang_map and dictionary to binary data dir diff --git a/configs/acoustic.yaml b/configs/acoustic.yaml index 901aeccfc..6d0ff747b 100644 --- a/configs/acoustic.yaml +++ b/configs/acoustic.yaml @@ -22,7 +22,6 @@ fmin: 40 fmax: 16000 binarization_args: - shuffle: true num_workers: 0 augmentation_args: random_pitch_shifting: diff --git a/configs/base.yaml b/configs/base.yaml index 72def325d..bd1dccac5 100644 --- a/configs/base.yaml +++ b/configs/base.yaml @@ -9,7 +9,6 @@ datasets: [] binary_data_dir: null binarizer_cls: null binarization_args: - shuffle: false num_workers: 0 audio_sample_rate: 44100 diff --git a/configs/templates/config_acoustic.yaml b/configs/templates/config_acoustic.yaml index 02ff55b6e..1040dca47 100644 --- a/configs/templates/config_acoustic.yaml +++ b/configs/templates/config_acoustic.yaml @@ -36,7 +36,7 @@ pe_ckpt: 'checkpoints/rmvpe/model.pt' hnsep: vr hnsep_ckpt: 'checkpoints/vr/model.pt' vocoder: NsfHifiGAN -vocoder_ckpt: checkpoints/nsf_hifigan_44.1k_hop512_128bin_2024.02/model.ckpt +vocoder_ckpt: checkpoints/pc_nsf_hifigan_44.1k_hop512_128bin_2025.02/model.ckpt use_lang_id: false num_lang: 1 diff --git a/configs/variance.yaml b/configs/variance.yaml index 0ec4a63fe..bac0f8154 100644 --- a/configs/variance.yaml +++ b/configs/variance.yaml @@ -18,7 +18,6 @@ win_size: 2048 # FFT size. midi_smooth_width: 0.06 # in seconds binarization_args: - shuffle: true num_workers: 0 prefer_ds: false diff --git a/docs/BestPractices.md b/docs/BestPractices.md index cc9c26dd9..79bf9f3e5 100644 --- a/docs/BestPractices.md +++ b/docs/BestPractices.md @@ -6,19 +6,19 @@ A configuration file is a YAML file that defines enabled features, model hyperparameters and controls the behavior of the binarizer, trainer and inference. Almost all settings and controls in this repository, including the practices in this guidance, are achieved through configuration files. -For more information of the configuration system and configurable attributes, see [Configuration Schemas](ConfigurationSchemas.md). +For more information about the configuration system and configurable attributes, see [Configuration Schemas](ConfigurationSchemas.md). ### Languages -Each language you are dealing with should have a unique tag in the configuration file. **We highly recommend using ISO 639 language codes as language tags.** For example, `zh` and `zho` stands for Chinese (`cmn` specifically for Mandarin Chinese), `ja` and `jpn` for Japanese, `en` and `eng` for English, `yue` for Cantonese (Yue). You can download a complete language code table from https://iso639-3.sil.org/code_tables/download_tables. +Each language you are dealing with should have a unique tag in the configuration file. **We highly recommend using ISO 639 language codes as language tags.** For example, `zh` and `zho` stand for Chinese (`cmn` specifically for Mandarin Chinese), `ja` and `jpn` for Japanese, `en` and `eng` for English, `yue` for Cantonese (Yue). You can download a complete language code table from https://iso639-3.sil.org/code_tables/download_tables. ### Phonemes Phonemes are the fundamental part of dictionaries and labels. There are two types of phonemes: language-specific phonemes and global phonemes. -**Language-specific phonemes:** If there are multiple languages, all language-specific phonemes will be prefixed with its language name. For example: `zh/a`, `ja/o`, `en/eh`. These are called the **full name** of the phonemes, while `a`, `o`, `eh` are called the **short name** which has definite meaning only in a specific language context. If there is only one language, the short names can be used to determine each phoneme. +**Language-specific phonemes:** If there are multiple languages, all language-specific phonemes will be prefixed with their language name. For example: `zh/a`, `ja/o`, `en/eh`. These are called the **full name** of the phonemes, while `a`, `o`, `eh` are called the **short name** which has definite meaning only in a specific language context. If there is only one language, the short names can be used to determine each phoneme. -**Global phonemes:** Some phonemes do not belong to any language. There are two reserved global phoneme tags: `SP` for space, and `AP` for aspiration. There can also be other user-defined tags (`EP`, `GS`, `VF`, etc.). These tags will not be prefixed with language, and are prior when identifying phoneme names. +**Global phonemes:** Some phonemes do not belong to any language. There are two reserved global phoneme tags: `SP` for space, and `AP` for aspiration. There can also be other user-defined tags (`EP`, `GS`, `VF`, etc.). These tags will not be prefixed with a language tag, and are prior when identifying phoneme names. Extra phonemes, including user-defined global phonemes and additional language-specific phonemes that are not present in the dictionaries, can be defined in a list in the configuration file (full names should be used): @@ -34,17 +34,17 @@ merged_phoneme_groups: - [zh/s, ja/s, en/s] - [ja/cl, SP] # global phonemes can also be merged # ... (other groups omitted for brevity) -use_lang_id: true # whether to use language embedding; only take effects if there are cross-lingual phonemes +use_lang_id: true # whether to use language embedding; only takes effect if there are cross-lingual phonemes ``` -Merging phonemes does not mean that they are exactly the same for the dictionary. For those cross-lingual merged phonemes, Setting `use_lang_id` to true will still distinguish them by language IDs. +Merging phonemes does not mean that they are exactly the same for the dictionary. For those cross-lingual merged phonemes, setting `use_lang_id` to true will still distinguish them by language IDs. #### Phoneme naming principles - Short names of language-specific phonemes should not conflict with global phoneme names, including reserved ones. - `/` cannot be used because it is already used for splitting the language tag and the short name. - `-` and `+` cannot be used because they are defined as slur tags in most singing voice synthesis editors. -- Other special characters, including but not limited to `@`, `#`, `&`, `|`, `<`, `>`, is not recommended because they may be used as special tags in the future format changes. +- Other special characters, including but not limited to `@`, `#`, `&`, `|`, `<`, `>`, are not recommended because they may be used as special tags in the future format changes. - ASCII characters are preferred for the best encoding compatibility, but all UTF-8 characters are acceptable. ### Dictionaries @@ -71,8 +71,8 @@ Each dictionary is a *.txt* file, in which each line represents a mapping rule f - `AP` and `SP` cannot be used because they are reserved tags when using DiffSinger in editors. - `/` cannot be used because it is already used for splitting the language tag and the short name. - `-` and `+` cannot be used because they are defined as slur tags in most singing voice synthesis editors. -- Syllable names is not recommended to start with `.` because this may have special meanings in the future editors. -- Other special characters, including but not limited to `@`, `#`, `&`, `|`, `<`, `>`, is not recommended because they may be used as special tags in the future format changes. +- Syllable names are not recommended to start with `.` because this may have special meanings in the future editors. +- Other special characters, including but not limited to `@`, `#`, `&`, `|`, `<`, `>`, are not recommended because they may be used as special tags in the future format changes. - ASCII characters are preferred for the best encoding compatibility, but all UTF-8 characters are acceptable. There are some example dictionaries in the [dictionaries/](../dictionaries) folder. @@ -119,24 +119,24 @@ datasets: # define all raw datasets - wav1 - wav2 # ... (other datasets omitted for brevity) -num_spk: 2 # number of languages; should be > maximum speaker ID +num_spk: 2 # number of speakers; should be > maximum speaker ID ``` ### DS files -DS files are JSON files with _.ds_ suffix that contains phoneme sequence, phoneme durations, music scores or curve parameters. They are mainly used to run inference on models for test and evaluation purposes, and they can be used as training data in some cases. There are some example DS files in the [samples/](../samples) folder. +DS files are JSON files with _.ds_ suffix that contain phoneme sequence, phoneme durations, music scores or curve parameters. They are mainly used to run inference on models for test and evaluation purposes, and they can be used as training data in some cases. There are some example DS files in the [samples/](../samples) folder. -The current recommended way of using a model for production purposes is to use [OpenUTAU for DiffSinger](https://github.com/xunmengshe/OpenUtau). It can export DS files as well. +The current recommended way of using a model for production purposes is to use [OpenUTAU for DiffSinger](https://github.com/stakira/OpenUtau). It can export DS files as well. ### Other fundamental assets #### Vocoders -A vocoder is a model that can reconstruct the audio waveform given the low-dimensional mel-spectrogram. The vocoder is the essential dependency if you want to train an acoustic model and hear the voice on the TensorBoard. +A vocoder is a model that can reconstruct the audio waveform given the low-dimensional mel-spectrogram. The vocoder is the essential dependency if you want to train an acoustic model and hear the voice on TensorBoard. The [DiffSinger Community Vocoders Project](https://openvpi.github.io/vocoders) provides a universal pre-trained NSF-HiFiGAN vocoder that can be used for starters of this repository. To use it, download the model (~50 MB size) from its releases and unzip it into the `checkpoints/` folder. -The pre-trained vocoder can be fine-tuned on your target dataset. It is highly recommended to do so because fine-tuned vocoder can generate much better results on specific (seen) datasets while does not need much computing resources. See the [vocoder training and fine-tuning repository](https://github.com/openvpi/SingingVocoders) for detailed instructions. After you get the fine-tuned vocoder checkpoint, you can configure it by `vocoder_ckpt` key in your configuration file. The fine-tuned NSF-HiFiGAN vocoder checkpoints can be exported to ONNX format like other DiffSinger user models for further production purposes. +The pre-trained vocoder can be fine-tuned on your target dataset. It is highly recommended to do so because fine-tuned vocoder can generate much better results on specific (seen) datasets while not requiring much computing resources. See the [vocoder training and fine-tuning repository](https://github.com/openvpi/SingingVocoders) for detailed instructions. After you get the fine-tuned vocoder checkpoint, you can configure it by `vocoder_ckpt` key in your configuration file. The fine-tuned NSF-HiFiGAN vocoder checkpoints can be exported to ONNX format like other DiffSinger user models for further production purposes. Another unrecommended option: train an ultra-lightweight [DDSP vocoder](https://github.com/yxlllc/pc-ddsp) first by yourself, then configure it according to the relevant [instructions](https://github.com/yxlllc/pc-ddsp/blob/master/DiffSinger.md). @@ -152,15 +152,15 @@ An acoustic model takes low-level singing information as input, including (but n ### Datasets -To train an acoustic model, you must have three columns in your transcriptions.csv: `name`, `ph_seq` and `ph_dur`, where `ph_seq` is the phoneme sequence and `ph_dur` is the phoneme duration sequence in seconds. You must have all corresponding recordings declared by the `name` column in mono, WAV format. +To train an acoustic model, you must have three columns in your transcriptions.csv: `name`, `ph_seq` and `ph_dur`, where `ph_seq` is the phoneme sequence and `ph_dur` is the phoneme duration sequence in seconds. You must have all corresponding recordings declared by the `name` column placed in the `wavs` folder, in WAV or FLAC format (`.wav` takes precedence if both exist). Although not mandatory, it is recommended to prepare all recordings in mono. -Training from multiple datasets in one model (so that the model is a multi-speaker model) is supported. See `speakers`, `spk_ids` and `use_spk_id` in the configuration schemas. +Training from multiple datasets in one model (so that the model is a multi-speaker model) is supported. See `datasets[].speaker`, `datasets[].spk_id`, `num_spk` and `use_spk_id` in the configuration schemas. ### Functionalities Functionalities of acoustic models are defined by their inputs. Acoustic models have three basic and fixed inputs: phoneme sequence, phoneme duration sequence and F0 (pitch) sequence. There are three categories of additional inputs (control parameters): -- speaker IDs: if your acoustic model is a multi-speaker model, you can use different speaker in the same model, or mix their timbre and style. +- speaker IDs: if your acoustic model is a multi-speaker model, you can use different speakers in the same model, or mix their timbre and style. - variance parameters: these curve parameters are features extracted from the recordings, and can control the timbre and style of the singing voice. See `use_energy_embed` and `use_breathiness_embed` in the configuration schemas. Please note that variance parameters **do not have default values**, so they are usually obtained from the variance model at inference time. - transition parameters: these values represent the transition of the mel-spectrogram, and are obtained by enabling data augmentation. They are scalars at training time and sequences at inference time. See `augmentation_args`, `use_key_shift_embed` and `use_speed_embed` in the configuration schemas. @@ -184,7 +184,7 @@ Variance models support multi-speaker settings like acoustic models do. ### Functionalities -Functionalities of variance models are defined by their outputs. There are three main prediction modules that can be enabled/disable independently: +Functionalities of variance models are defined by their outputs. There are three main prediction modules that can be enabled/disabled independently: - Duration Predictor: predicts the phoneme durations. See `predict_dur` in the configuration schemas. - Pitch Predictor: predicts the pitch curve. See `predict_pitch` in the configuration schemas. @@ -194,7 +194,7 @@ There may be some mutual influence between the modules above when they are enabl ## Build variance datasets with DS files -By default, the variance binarizer loads attributes from transcriptions.csv and searches for recording files (*.wav) to extract features and parameters. These attributes and parameters also exist in DS files, which are normally used for inference. This section introduces the required settings and important notes to build a variance dataset from DS files. +By default, the variance binarizer loads attributes from transcriptions.csv and searches for recording files (*.wav or *.flac) to extract features and parameters. These attributes and parameters also exist in DS files, which are normally used for inference. This section introduces the required settings and important notes to build a variance dataset from DS files. First of all, you should edit your configuration file to enable loading from DS files: @@ -218,12 +218,13 @@ The DS files should also use the same dictionary as that of your target model. T | `f0_seq` | ✓ | ✓ | ✓ | WAV | DS/WAV | | `energy`, `breathiness`, ... | | | ✓ | WAV | DS/WAV | -This means you only need one column in transcriptions.csv, the `name` column, to declare all DS files included in the dataset. The name pattern can be: +This means you only need the `name` column in transcriptions.csv to declare all DS files included in the dataset. The name can be a bare name or a 0-based segment index joined by `#`, and the resolution follows the rule of "the most specific file wins": -- Full name: `some-name` will firstly match the first segment in `some-name.ds`. -- Name with index: `some-name#0` and `some-name#1` will match segment 0 and segment 1 in `some-name.ds` if there are no match with full name. +- `some-name#1` first matches the first segment in file `some-name#1.ds` (a dedicated single-segment file); +- if there is no such file, it matches segment 1 in file `some-name.ds` (a multi-segment file). +- A bare name `some-name` is equivalent to `some-name#0`. -Though not recommended, the binarizer will still try to load attributes from transcriptions.csv or extract parameters from recordings if there are no matching DS files. In this case the full name matching logic is applied (the same as the normal binarization process). +If neither file exists, the binarizer falls back to transcriptions.csv or parameter extraction from the recording. ## Choosing variance parameters @@ -233,9 +234,9 @@ Variance parameters are a type of parameters that are significantly related to s #### Energy -> WARNING +> [!WARNING] > -> This parameter is no longer recommended in favor of the new voicing parameter. The latter are less coupled with breathiness than energy. +> This parameter is no longer recommended in favor of the new voicing parameter. The latter is less coupled with breathiness than energy. Energy is defined as the RMS curve of the singing, in dB, which can control the strength of voice to a certain extent. @@ -250,13 +251,17 @@ Voicing is defined as the RMS curve of the harmonic part of the singing, in dB, #### Tension Tension is mostly related to the ratio of the base harmonic to the full harmonics, which can be used to control the strength and timbre of the voice. The ratio is calculated as + $$ -r = \frac{\text{RMS}(H_{full}-H_{base})}{\text{RMS}(H_{full})} +r = \frac{\sqrt{\text{RMS}(H_{full})^2 - \text{RMS}(H_{base})^2}}{\text{RMS}(H_{full})} $$ -where $H_{full}$ is the full harmonics and $H_{base}$ is the base harmonic. The ratio is then mapped to the final domain via the inverse function of Sigmoid, that + +where $H_{full}$ is the full harmonics and $H_{base}$ is the base harmonic. The ratio is then mapped to the final domain via the inverse function of Sigmoid, that is + $$ T = \log{\frac{r}{1-r}} $$ + where $T$ is the tension value. ### Principles of choosing multiple parameters @@ -267,17 +272,17 @@ These three parameters should **NOT** be enabled together. Energy is the RMS of #### Energy, voicing and tension -When voicing (or energy) is enabled, it almost fixes the loudness. However, tension sometimes rely on the implicitly predicted loudness for more expressiveness, because when a person sings with higher tension, he/she always produces louder voice. For this reason, some people may find their models or datasets _less natural_ with tension control. To be specific, changing tension will change the timbre but keep the loudness, and changing voicing (or energy) will change the loudness but keep the timbre. This behavior can be suitable for some, but not all datasets and users. Therefore, it is highly recommended for everyone to conduct some experiments on the actual datasets used to train the model. +When voicing (or energy) is enabled, it almost fixes the loudness. However, tension sometimes relies on the implicitly predicted loudness for more expressiveness, because when a person sings with higher tension, he/she always produces a louder voice. For this reason, some people may find their models or datasets _less natural_ with tension control. To be specific, changing tension will change the timbre but keep the loudness, and changing voicing (or energy) will change the loudness but keep the timbre. This behavior can be suitable for some, but not all datasets and users. Therefore, it is highly recommended for everyone to conduct some experiments on the actual datasets used to train the model. ## Mutual influence between variance modules -In some recent experiments and researches, some mutual influence between the modules of variance models has been found. In practice, being aware of the influence and making use of it can improve accuracy and avoid instability of the model. +In some recent experiments and research, some mutual influence between the modules of variance models has been found. In practice, being aware of the influence and making use of it can improve accuracy and avoid instability of the model. ### Influence on the duration predictor The duration predictor benefits from its downstream modules, like the pitch predictor and the variance predictor. -The experiments were conducted on both manually refined datasets and automatically labeled datasets, and with pitch predictors driven by both base pitch and melody encoder. All the results have shown that when either of the pitch predictor and the variance predictor is enabled together with the duration predictor, its rhythm correctness and duration accuracy significantly outperforms those of a solely trained duration predictor. +The experiments were conducted on both manually refined datasets and automatically labeled datasets, and with pitch predictors driven by both base pitch and melody encoder. All the results have shown that when either of the pitch predictor and the variance predictor is enabled together with the duration predictor, its rhythm correctness and duration accuracy significantly outperform those of a solely trained duration predictor. Possible reason for this difference can be the lack of information carried by pure phoneme duration sequences, which may not fully represent the phoneme features in the real world. With the help of frame-level feature predictors, the encoder learns more knowledge about the voice features related to the phoneme types and durations, thus making the duration predictor produce better results. @@ -324,7 +329,7 @@ pe: parselmouth #### RMVPE (recommended) -[RMVPE](https://github.com/Dream-High/RMVPE) (Robust Model for Vocal Pitch Estimation) is the state-of-the-art NN-based pitch estimation model for singing voice. It runs slower than parselmouth, consumes more memory, however uses CUDA to accelerate computation (if available) and produce better results on noisy recordings and edge cases. +[RMVPE](https://github.com/Dream-High/RMVPE) (Robust Model for Vocal Pitch Estimation) is the state-of-the-art NN-based pitch estimation model for singing voice. It runs slower than parselmouth, consumes more memory, however uses CUDA to accelerate computation (if available) and produces better results on noisy recordings and edge cases. To enable RMVPE, download its pre-trained checkpoint from [here](https://github.com/yxlllc/RMVPE/releases), extract it into the `checkpoints/` folder and edit the configuration file: @@ -343,11 +348,11 @@ To use Harvest, simply include the following line in your configuration file: pe: harvest ``` -**Note:** It is also recommended to change the F0 detection range for Harvest with accordance to your dataset, as they are hard boundaries for this algorithm and the defaults might not suffice for most use cases. To change the F0 detection range, you may include or edit this part in the configuration file: +**Note:** It is also recommended to change the F0 detection range for Harvest in accordance with your dataset, as they are hard boundaries for this algorithm and the defaults might not suffice for most use cases. To change the F0 detection range, you may include or edit this part in the configuration file: ```yaml f0_min: 65 # Minimum F0 to detect -f0_max: 800 # Maximum F0 to detect +f0_max: 1100 # Maximum F0 to detect ``` ### Harmonic-noise separation @@ -356,7 +361,7 @@ Harmonic-noise separation is the process of separating the harmonic part and the #### WORLD -This algorithm uses Masanori Morise's [WORLD](https://github.com/mmorise/World), a free software for high-quality speech analysis, manipulation and synthesis. It uses CPU (no CUDA required) but runs relatively slow. +This algorithm uses Masanori Morise's [WORLD](https://github.com/mmorise/World), a free software for high-quality speech analysis, manipulation and synthesis. It uses CPU (no CUDA required) but runs relatively slowly. To use WORLD, simply include the following line in your configuration file: @@ -377,7 +382,7 @@ hnsep_ckpt: checkpoints/vr/model.pt ## Shallow diffusion -Shallow diffusion is a mechanism that can improve quality and save inference time for diffusion models that was first introduced in the original DiffSinger [paper](https://arxiv.org/abs/2105.02446). Instead of starting the diffusion process from purely gaussian noise as classic diffusion does, shallow diffusion adds a shallow gaussian noise on a low-quality results generated by a simple network (which is called the auxiliary decoder) to skip many unnecessary steps from the beginning. With the combination of shallow diffusion and sampling acceleration algorithms, we can get better results under the same inference speed as before, or achieve higher inference speed without quality deterioration. +Shallow diffusion is a mechanism that can improve quality and save inference time for diffusion models that was first introduced in the original DiffSinger [paper](https://arxiv.org/abs/2105.02446). Instead of starting the diffusion process from purely Gaussian noise as classic diffusion does, shallow diffusion adds shallow Gaussian noise to low-quality results generated by a simple network (which is called the auxiliary decoder) to skip many unnecessary steps from the beginning. With the combination of shallow diffusion and sampling acceleration algorithms, we can get better results under the same inference speed as before, or achieve higher inference speed without quality deterioration. Currently, acoustic models in this repository support shallow diffusion. The main switch of shallow diffusion is `use_shallow_diffusion` in the configuration file, and most arguments of shallow diffusion can be adjusted under `shallow_diffusion_args`. See [Configuration Schemas](ConfigurationSchemas.md) for more details. @@ -387,13 +392,15 @@ To train a full shallow diffusion model from scratch, simply introduce the follo ```yaml use_shallow_diffusion: true -K_step: 400 # adjust according to your needs -K_step_infer: 400 # should be <= K_step +T_start: 0.4 # adjust according to your needs +T_start_infer: 0.4 # should be >= T_start ``` -Please note that when shallow diffusion is enabled, only the last $K$ diffusion steps will be trained. Unlike classic diffusion models which are trained on full steps, the limit of `K_step` can make the training more efficient. However, `K_step` should not be set too small because without enough diffusion depth (steps), the low-quality auxiliary decoder results cannot be well refined. 200 ~ 400 should be the proper range of `K_step`. +With the default `diffusion_type: reflow`, shallow diffusion trains only the tail of the trajectory, $t \in (T_{start}, 1)$. This is more efficient than training on the full trajectory, but `T_start` should not be set too large, because without enough diffusion depth the low-quality auxiliary decoder results cannot be well refined. + +For DDPM models (currently not recommended), use `K_step` and `K_step_infer` (in number of steps, and `K_step_infer` should be <= `K_step`) instead; they are equivalent to the above under `T_start = 1 - K_step / timesteps`. See `T_start`, `T_start_infer`, `K_step` and `K_step_infer` in the configuration schemas. -The auxiliary decoder and the diffusion decoder shares the same linguistic encoder, which receives gradients from both the decoders. In some experiments, it was found that gradients from the auxiliary decoder will cause mismatching between the encoder and the diffusion decoder, resulting in the latter being unable to produce reasonable results. To prevent this case, a configuration item called `aux_decoder_grad` is introduced to apply a scale factor on the gradients from the auxiliary decoder during training. To adjust this factor, introduce the following in the configuration file: +The auxiliary decoder and the diffusion decoder share the same linguistic encoder, which receives gradients from both the decoders. In some experiments, it was found that gradients from the auxiliary decoder will cause mismatching between the encoder and the diffusion decoder, resulting in the latter being unable to produce reasonable results. To prevent this case, a configuration item called `aux_decoder_grad` is introduced to apply a scale factor on the gradients from the auxiliary decoder during training. To adjust this factor, introduce the following in the configuration file: ```yaml shallow_diffusion_args: @@ -402,11 +409,11 @@ shallow_diffusion_args: ### Train auxiliary decoder and diffusion decoder separately -Training a full shallow diffusion model can consume more memory because the auxiliary decoder is also in the training graph. In limited situations, the two decoders can be trained separately, i.e. train one decoder after another. +Training a full shallow diffusion model can consume more memory because the auxiliary decoder is also in the training graph. In resource-limited situations, the two decoders can be trained separately, i.e. train one decoder after another. **STEP 1: train the diffusion decoder** -In the first stage, the linguistic encoder and the diffusion decoder is trained together, while the auxiliary decoder is left unchanged. Edit your configuration file like this: +In the first stage, the linguistic encoder and the diffusion decoder are trained together, while the auxiliary decoder is left unchanged. Edit your configuration file like this: ```yaml use_shallow_diffusion: true # make sure the main option is turned on @@ -416,11 +423,11 @@ shallow_diffusion_args: val_gt_start: true # should be true because the auxiliary decoder is not trained yet ``` -Start training until `max_updates` is reached, or until you get satisfactory results on the TensorBoard. +Start training until `max_updates` is reached, or until you get satisfactory results on TensorBoard. **STEP 2: train the auxiliary decoder** -In the second stage, the auxiliary decoder is trained besides the linguistic encoder and the diffusion decoder. Edit your configuration file like this: +In the second stage, only the auxiliary decoder is trained; the diffusion decoder is excluded from the training graph. Edit your configuration file like this: ```yaml shallow_diffusion_args: @@ -437,14 +444,11 @@ frozen_params: - model.fs2 # the linguistic encoder ``` -You should also manually reset your learning rate scheduler because this is a new training process for the auxiliary decoder. Possible ways are: - -1. Rename the latest checkpoint to `model_ckpt_steps_0.ckpt` and remove the other checkpoints from the directory. -2. Increase the initial learning rate (if you use a scheduler that decreases the LR over training steps) so that the auxiliary decoder gets proper learning rate. +You should also manually reset your learning rate scheduler because this is a new training process for the auxiliary decoder. For example, increase the initial learning rate (if you use a scheduler that decreases the LR over training steps) so that the auxiliary decoder gets a proper learning rate. Additionally, `max_updates` should be adjusted to ensure enough training steps for the auxiliary decoder. -Once you finished the configurations above, you can resume the training. The auxiliary decoder normally does not need many steps to train, and you can stop training when you get stable results on the TensorBoard. Because this step is much more complicated than the previous step, it is recommended to run some inference to verify if the model is trained properly after everything is finished. +Once you have finished the configurations above, you can resume the training. The auxiliary decoder normally does not need many steps to train, and you can stop training when you get stable results on TensorBoard. Because this step is much more complicated than the previous step, it is recommended to run some inference to verify if the model is trained properly after everything is finished. ### Add shallow diffusion to classic diffusion models @@ -458,7 +462,7 @@ finetune_ckpt_path: xxx.ckpt # path to your old checkpoint finetune_ignored_params: [] # do not ignore any parameters ``` -Then you can follow the instructions in STEP 2 of the [previous section](#add-shallow-diffusion-to-classic-diffusion-models) to finish your training. +Then you can follow the instructions in STEP 2 of the [previous section](#train-auxiliary-decoder-and-diffusion-decoder-separately) to finish your training. ## Performance tuning @@ -485,30 +489,24 @@ For more details of the batch sampler algorithm and this configuration key, see ### Automatic mixed precision -Enabling automatic mixed precision (AMP) can accelerate training and save GPU memory. DiffSinger have adapted the latest version of PyTorch Lightning for AMP functionalities. - -By default, the training runs in FP32 precision. To enable AMP, edit your configuration file: - -```yaml -pl_trainer_precision: 16-mixed # FP16 precision -``` +Enabling automatic mixed precision (AMP) can accelerate training and save GPU memory. DiffSinger has adapted the latest version of PyTorch Lightning for AMP functionalities. -or +By default, the training runs in 16-mixed precision. To disable AMP, edit your configuration file: ```yaml -pl_trainer_precision: bf16-mixed # BF16 precision +pl_trainer_precision: 32-true # FP32 precision ``` For more precision options, please check out the [official documentation](https://lightning.ai/docs/pytorch/stable/common/trainer.html#precision). ### Training on multiple GPUs -Using distributed data parallel (DDP) can divide training tasks to multiple GPUs and synchronize gradients and weights between them. DiffSinger have adapted the latest version of PyTorch Lightning for DDP functionalities. +Using distributed data parallel (DDP) can divide training tasks to multiple GPUs and synchronize gradients and weights between them. DiffSinger has adapted the latest version of PyTorch Lightning for DDP functionalities. -By default, the trainer will utilize all CUDA devices defined in the `CUDA_VISIBLE_DEVICES` environment variable (empty means using all available devices). If you want to specify which GPUs to use, edit your configuration file: +By default, the trainer will utilize all CUDA devices defined in the `CUDA_VISIBLE_DEVICES` environment variable (or all available devices if this variable is not set). If you want to specify which GPUs to use, edit your configuration file: ```yaml -pl_trainer_devices: [0, 1, 2, 3] # use the first 4 GPUs defined in CUDA_VISIBLE_DEVICES +pl_trainer_devices: [0, 1, 2, 3] # use the first 4 GPUs among the visible devices ``` Please note that `max_batch_size` and `max_batch_frames` are values for **each** GPU. @@ -541,7 +539,7 @@ Please note that enabling gradient accumulation will slow down training because ## Optimizers and learning rate schedulers -The optimizer and the learning rate scheduler can take an important role in the training process. DiffSinger uses a flexible configuration logic for these two modules. +The optimizer and the learning rate scheduler can play an important role in the training process. DiffSinger uses a flexible configuration logic for these two modules. ### Basic configurations @@ -560,19 +558,18 @@ and for the learning rate scheduler: ```yaml lr_scheduler_args: - scheduler_cls: torch.optim.lr_scheduler.StepLR # class name of learning rate schedule - warmup_steps: 2000 + scheduler_cls: torch.optim.lr_scheduler.StepLR # class name of learning rate scheduler step_size: 50000 gamma: 0.5 ``` -Note that `optimizer_args` and `lr_scheduler_args` will be filtered by needed parameters and passed to `__init__` as keyword arguments (`kwargs`) when constructing the optimizer and scheduler. Therefore, you could specify all arguments according to your need in the configuration file to directly control the behavior of optimization and LR scheduling. It will also tolerate parameters existing in the configuration but not needed in `__init__`. +Note that `optimizer_args` and `lr_scheduler_args` will be filtered by needed parameters and passed to `__init__` as keyword arguments (`kwargs`) when constructing the optimizer and scheduler. Therefore, you could specify all arguments according to your needs in the configuration file to directly control the behavior of optimization and LR scheduling. It will also tolerate parameters existing in the configuration but not needed in `__init__`. Also, note that the LR scheduler performs scheduling on the granularity of steps, not epochs. The special case applies when a tuple is needed in `__init__`: `beta1` and `beta2` are treated separately and form a tuple in the code. You could try to pass in an array instead. (And as an experiment, AdamW does accept `[beta1, beta2]`). If there is another special treatment required, please submit an issue. -For PyTorch built-in optimizers and LR schedulers, see official [documentation](https://pytorch.org/docs/stable/optim.html) of the `torch.optim` package. If you found other optimizer and learning rate scheduler useful, you can raise a topic in [Discussions](https://github.com/openvpi/DiffSinger/discussions), raise [Issues](https://github.com/openvpi/DiffSinger/issues) or submit [PRs](https://github.com/openvpi/DiffSinger/pulls) if it introduces new codes or dependencies. +For PyTorch built-in optimizers and LR schedulers, see official [documentation](https://pytorch.org/docs/stable/optim.html) of the `torch.optim` package. If you find other optimizers and learning rate scheduler useful, you can raise a topic in [Discussions](https://github.com/openvpi/DiffSinger/discussions), raise [Issues](https://github.com/openvpi/DiffSinger/issues) or submit [PRs](https://github.com/openvpi/DiffSinger/pulls) if it introduces new code or dependencies. ### Composite LR schedulers @@ -594,7 +591,7 @@ lr_scheduler_args: - 20 ``` -The LR scheduler objects will be recursively construct objects if `cls` is present in sub-arguments. Please note that `cls` must be a scheduler class because this is a special design. +The LR scheduler objects will be recursively constructed if `cls` is present in sub-arguments. Please note that `cls` must be a scheduler class because this is a special design. **WARNING:** Nested `SequentialLR` and `ChainedScheduler` have unexpected behavior. **DO NOT** nest them. Also, make sure the scheduler is _chainable_ before using it in `ChainedScheduler`. @@ -602,7 +599,7 @@ The LR scheduler objects will be recursively construct objects if `cls` is prese ### Fine-tuning from existing checkpoints -By default, the training starts from a model from scratch with randomly initialized parameters. However, if you already have some pre-trained checkpoints, and you need to adapt them to other datasets with their functionalities unchanged, fine-tuning may save training steps and time. In general, you need to add the following structure into the configuration file: +By default, the training starts from a model with randomly initialized parameters. However, if you already have some pre-trained checkpoints, and you need to adapt them to other datasets with their functionalities unchanged, fine-tuning may save training steps and time. In general, you need to add the following structure into the configuration file: ```yaml # take acoustic models as an example @@ -625,7 +622,7 @@ For the pre-trained checkpoint, it must be a file saved with `torch.save`, conta "model.fs2.pitch_embed.bias": null, // torch.Tensor // ... (other parameters) } - // ... (other possible keys + // ... (other possible keys) } ``` @@ -648,4 +645,4 @@ frozen_params: # prefix rules to freeze specific parameters during training - model.fs2.pitch_embed ``` -You may interrupt the training and change the settings above at any time. Sometimes this will cause mismatching optimizer state - and it will be discarded silently. +You may interrupt the training and change the settings above at any time. Sometimes this can lead to optimizer state mismatch, resulting in errors. diff --git a/docs/ConfigurationSchemas.md b/docs/ConfigurationSchemas.md index b7c8df68d..449e9e3c0 100644 --- a/docs/ConfigurationSchemas.md +++ b/docs/ConfigurationSchemas.md @@ -2,33 +2,35 @@ ## The configuration system -DiffSinger uses a cascading configuration system based on YAML files. All configuration files originally inherit and override [configs/base.yaml](../configs/base.yaml), and each file directly override another file by setting the `base_config` attribute. The overriding rules are: +DiffSinger uses a cascading configuration system based on YAML files. Inheritance is completely explicit: a configuration file inherits from other files by listing them in its `base_config` attribute. Sources are applied in the following order, with later sources overriding earlier ones: -- Configuration keys with the same path and the same name will be replaced. Other paths and names will be merged. -- All configurations in the inheritance chain will be squashed (via the rule above) as the final configuration. -- The trainer will save the final configuration in the experiment directory, which is detached from the chain and made independent from other configuration files. +1. **The `base_config` chain** (from `--config`): base files are loaded depth-first, and configurations are merged recursively: when the overriding value is a mapping and the key already exists in the inherited configuration, it is merged key by key into the existing mapping instead of replacing the whole mapping; non-mapping values (scalars, lists, etc.) simply replace whatever was there before. Keys that exist in only one configuration are kept. All configurations in the inheritance chain are squashed as the final configuration of this source. +2. **The saved experiment configuration**: when `--exp_name` is given, the final configuration is saved to `checkpoints//config.yaml` (with `base_config` emptied), which is detached from the chain and independent of other configuration files. When the same `--exp_name` is used again (e.g., when resuming training), every key present in the saved file replaces the chain's value wholesale, including nested mappings, while keys that exist only in the chain are kept. Pass `--reset` to discard the saved configuration and rebuild it from `--config` (the rebuilt configuration is then saved again). +3. **Command-line overrides** (from `--hparams key=value,key=value`): applied last, taking precedence over both sources above. The argument string is split on `,` and then on `=`, so values must not contain either character. The override syntax addresses top-level keys only (it does not interpret dotted paths). For an existing key, conversion is only reliable for scalar values whose current type is `bool`, `int`, `float` or `str`; list/mapping values and `None` are not reliably convertible. Boolean overrides currently require Python's `True`/`False` spellings, and the parser uses `eval()` for overrides, so do not use it with untrusted input. + +The final configuration is saved to the experiment directory at startup only when `checkpoints//config.yaml` does not exist yet or `--reset` is given; resuming an existing experiment does *not* re-save it. The saving step is skipped when `--infer` is given, which also marks the run as inference (`hparams['infer']` set to `true`). Only the main process performs the saving. ## Configurable parameters -This following are the meaning and usages of all editable keys in a configuration file. +The following are the meanings and usages of all editable keys in a configuration file. -Each configuration key (including nested keys) are described with a brief explanation and several attributes listed as follows: +Each configuration key (including nested keys) is described with a brief explanation and several attributes listed as follows: -| Attribute | Explanation | -|:---------------:|:---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| -| visibility | Represents what kind(s) of models and tasks this configuration belongs to. | -| scope | The scope of effects of the configuration, indicating what it can influence within the whole pipeline. Possible values are:
**nn** - This configuration is related to how the neural networks are formed and initialized. Modifying it will result in failure when loading or resuming from checkpoints.
**preprocessing** - This configuration controls how raw data pieces or inference inputs are converted to inputs of neural networks. Binarizers should be re-run if this configuration is modified.
**training** - This configuration describes the training procedures. Most training configurations can affect training performance, memory consumption, device utilization and loss calculation. Modifying training-only configurations will not cause severe inconsistency or errors in most situations.
**inference** - This configuration describes the calculation logic through the model graph. Changing it can lead to inconsistent or wrong outputs of inference or validation.
**others** - Other configurations not discussed above. Will have different effects according to the descriptions. | -| customizability | The level of customizability of the configuration. Possible values are:
**required** - This configuration **must** be set or modified according to the actual situation or condition, otherwise errors can be raised.
**recommended** - It is recommended to adjust this configuration according to the dataset, requirements, environment and hardware. Most functionality-related and feature-related configurations are at this level, and all configurations in this level are widely tested with different values. However, leaving it unchanged will not cause problems.
**normal** - There is no need to modify it as the default value is carefully tuned and widely validated. However, one can still use another value if there are some special requirements or situations.
**not recommended** - No other values except the default one of this configuration are tested. Modifying it will not cause errors, but may cause unpredictable or significant impacts to the pipelines.
**reserved** - This configuration **must not** be modified. It appears in the configuration file only for future scalability, and currently changing it will result in errors. | -| type | Value type of the configuration. Follows the syntax of Python type hints. | -| constraints | Value constraints of the configuration. | -| default | Default value of the configuration. Uses YAML value syntax. | +| Attribute | Explanation | +| :-: | :-- | +| visibility | Represents which kinds of models and tasks this configuration applies to. Possible values are:
**acoustic** - This configuration applies to the acoustic model and task.
**variance** - This configuration applies to the variance model and task. | +| scope | The scope of the configuration's effects, indicating what it can influence within the whole pipeline. Possible values are:
**nn** - This configuration determines the presence or shapes of parameters and persistent buffers of the neural networks. Modifying it will result in failure when loading or resuming from checkpoints. Configurations that are read at model construction but do **not** change any saved key or shape are not **nn**.
**preprocessing** - This configuration controls how raw data pieces or inference inputs are converted to inputs of neural networks. Binarizers should be re-run if this configuration is modified.
**training** - This configuration describes the training procedures. Most training configurations can affect training performance, memory consumption, device utilization and loss calculation. Modifying training-only configurations will not cause severe inconsistency or errors in most situations.
**inference** - This configuration describes the calculation logic through the model graph. Changing it can lead to inconsistent or wrong outputs of inference or validation. | +| customizability | The level of customizability of the configuration. Possible values are:
**required** - This configuration **must** be set or modified according to the actual situation or condition, otherwise errors can be raised.
**recommended** - It is recommended to adjust this configuration according to the dataset, requirements, environment and hardware. Most functionality-related and feature-related configurations are at this level, and all configurations at this level are widely tested with different values. However, leaving it unchanged will not cause problems.
**normal** - There is no need to modify it as the default value is carefully tuned and widely validated. However, one can still use another value if there are some special requirements or situations.
**not recommended** - No values other than the default one are tested for this configuration. Modifying it will not cause errors, but may cause unpredictable or significant impacts on the pipelines.
**reserved** - This configuration **must not** be modified. It appears in the configuration file only for future scalability, and currently changing it will result in errors. | +| type | Value type of the configuration. Follows the syntax of Python type hints. Optional omission and fallback behavior are stated in the field description, while explicit `null` is included in the type only when it is accepted. | +| default | Default value of the configuration. Uses YAML value syntax. | +| constraints | Value constraints of the configuration. | ### accumulate_grad_batches -Indicates that gradients of how many training steps are accumulated before each `optimizer.step()` call. 1 means no gradient accumulation. +Indicates how many training steps' gradients are accumulated before each `optimizer.step()` call. 1 means no gradient accumulation. - + @@ -53,7 +55,7 @@ Sampling rate of waveforms.
visibilityall
visibilityacoustic, variance
scopetraining
customizabilityrecommended
typeint
- + @@ -64,7 +66,7 @@ Sampling rate of waveforms. Arguments for data augmentation.
visibilityacoustic, variance
scopepreprocessing
scopepreprocessing, inference
customizabilityreserved
typeint
default44100
- +
typedict
typedict[str, Any]
### augmentation_args.fixed_pitch_shifting @@ -72,7 +74,7 @@ Arguments for data augmentation. Arguments for fixed pitch shifting augmentation. - +
typedict
typedict[str, Any]
### augmentation_args.fixed_pitch_shifting.enabled @@ -85,7 +87,7 @@ Whether to apply fixed pitch shifting augmentation. customizabilityrecommended typebool defaultfalse -constraintsMust be false if augmentation_args.random_pitch_shifting.enabled is set to true. +constraintsMust be false if augmentation_args.random_pitch_shifting.enabled is set to true. Enabling it requires use_spk_id to be true, and num_spk ≥ (1 + number of targets) × (max spk_id + 1). ### augmentation_args.fixed_pitch_shifting.scale @@ -96,8 +98,9 @@ Scale ratio of each target in fixed pitch shifting augmentation. visibilityacoustic scopepreprocessing customizabilityrecommended -typetuple +typefloat default0.5 +constraintsMust be smaller than 1. ### augmentation_args.fixed_pitch_shifting.targets @@ -108,8 +111,9 @@ Targets (in semitones) of fixed pitch shifting augmentation. visibilityacoustic scopepreprocessing customizabilitynot recommended -typetuple +typelist[float] default[-5.0, 5.0] +constraintsMust not contain duplicate values. ### augmentation_args.random_pitch_shifting @@ -117,7 +121,7 @@ Targets (in semitones) of fixed pitch shifting augmentation. Arguments for random pitch shifting augmentation. - +
typedict
typedict[str, Any]
### augmentation_args.random_pitch_shifting.enabled @@ -129,20 +133,21 @@ Whether to apply random pitch shifting augmentation. scopepreprocessing customizabilityrecommended typebool -defaulttrue -constraintsMust be false if augmentation_args.fixed_pitch_shifting.enabled is set to true. +defaultfalse +constraintsMust be false if augmentation_args.fixed_pitch_shifting.enabled is set to true. Enabling it requires use_key_shift_embed to be true. ### augmentation_args.random_pitch_shifting.range -Range of the random pitch shifting ( in semitones). +Range of the random pitch shifting (in semitones). Besides being the augmentation sampling range, this value also calibrates the `gender` parameter at inference and ONNX export time: positive gender values are scaled by `max`, negative ones by the absolute value of `min`, and the resulting key shift of a *dynamic* (curve) gender value is clipped to this range. At Python inference time, a *static* scalar gender value is scaled the same way but **not** clipped, so values with absolute magnitude larger than 1 can produce key shifts outside this range; at ONNX export time, however, a static (frozen) gender value **is** clipped to this range, and exported graphs also clip the gender input to [-1, 1] before scaling, so the key shift always stays within this range. Do not modify it after preprocessing or training, otherwise inference behavior becomes inconsistent with the training data. An error is raised at inference or export time if [use_key_shift_embed](#use_key_shift_embed) is `true` while this key is missing from the configuration. - + - + +
visibilityacoustic
scopepreprocessing
scopepreprocessing, inference
customizabilitynot recommended
typetuple
typelist[float]
default[-5.0, 5.0]
constraintsMust satisfy min < 0 < max.
### augmentation_args.random_pitch_shifting.scale @@ -162,7 +167,7 @@ Scale ratio of the random pitch shifting augmentation. Arguments for random time stretching augmentation. - +
typedict
typedict[str, Any]
### augmentation_args.random_time_stretching.enabled @@ -174,19 +179,21 @@ Whether to apply random time stretching augmentation. scopepreprocessing customizabilityrecommended typebool -defaulttrue +defaultfalse +constraintsEnabling it requires use_speed_embed to be true. ### augmentation_args.random_time_stretching.range -Range of random time stretching factors. +Range of random time stretching factors. Besides being the augmentation sampling range, this value is also read at inference and ONNX export time as the clipping bounds of the `velocity` parameter curve before it is embedded. Do not modify it after preprocessing or training, otherwise inference behavior becomes inconsistent with the training data. At ONNX export time an error is raised if [use_speed_embed](#use_speed_embed) is `true` while this key is missing from the configuration; at inference time the key is only read when the input data actually provides a `velocity` parameter curve — if no velocity curve is given, speed silently defaults to 1.0 and the key is not accessed at all (unlike the pitch shifting range, which is read unconditionally at inference). - + - + +
visibilityacoustic
scopepreprocessing
scopepreprocessing, inference
customizabilitynot recommended
typetuple
typelist[float]
default[0.5, 2]
constraintsMust satisfy 0 < min < 1 < max.
### augmentation_args.random_time_stretching.scale @@ -203,45 +210,45 @@ Scale ratio of random time stretching augmentation. ### backbone_args -Keyword arguments for the backbone of main decoder module. +Keyword arguments for the backbone of the main decoder module. - - - +
visibilityacoustic, variance
scopenn
typedict
typedict[str, Any]
Available arguments for each backbone type are listed below. **WaveNet** (`backbone_type: wavenet`) -| argument name | type | default | description | -|:----------------------|:----:|:-------:|:--------------------------------------------------------------------------------------------------------------| -| num_layers | int | 20 | Number of residual block layers, or depth of the network | -| num_channels | int | 512 | Number of channels, or width of the network | -| dilation_cycle_length | int | 4 | Length k of the cycle $2^0, 2^1, \ldots, 2^k$ of convolution dilation factors through WaveNet residual blocks | +| argument name | type | default | description | +| :-- | :-: | :-: | :-- | +| num_layers | int | 20 | Number of residual block layers, or depth of the network | +| num_channels | int | 512 | Number of channels, or width of the network | +| dilation_cycle_length | int | 4 | Length k of the cycle $2^0, 2^1, \ldots, 2^{k-1}$ of convolution dilation factors through WaveNet residual blocks | **LYNXNet** (`backbone_type: lynxnet`) -| argument name | type | default | description | -|:--------------|:-----:|:-------:|:--------------------------------------------------------------------------------| -| num_layers | int | 6 | Number of LYNXNet blocks, or depth of the network | -| num_channels | int | 1024 | Number of channels, or width of the network | -| kernel_size | int | 31 | Kernel size of the depthwise convolution layers | -| dropout_rate | float | 0.0 | Dropout rate applied in each LYNXNet block | -| strong_cond | bool | false | Whether to use strong conditioning, which injects condition before the GLU gate | +| argument name | type | default | description | +| :-- | :-: | :-: | :-- | +| num_layers | int | 6 | Number of LYNXNet blocks, or depth of the network | +| num_channels | int | 1024 | Number of channels, or width of the network | +| expansion_factor | int | 2 | Channel expansion factor within each conv module | +| kernel_size | int | 31 | Kernel size of the depthwise convolution layers | +| activation | str | `PReLU` | Type of activation function. Choose from `PReLU`, `SiLU`, `ReLU`. | +| dropout_rate | float | 0.0 | Dropout rate applied in each LYNXNet block | +| strong_cond | bool | true | Whether to use strong conditioning, which injects condition before the residual split of each block | **LYNXNet2** (`backbone_type: lynxnet2`) -| argument name | type | default | description | -|:----------------------|:-----:|:-------:|:-------------------------------------------------------------------------------------------------| -| num_layers | int | 6 | Number of LYNXNet2 blocks, or depth of the network | -| num_channels | int | 1024 | Number of channels, or width of the network | -| kernel_size | int | 31 | Kernel size of the depthwise convolution layers | -| dropout_rate | float | 0.0 | Dropout rate applied in each LYNXNet2 block | -| use_conditioner_cache | bool | true | Whether to use Conv1d-based conditioner projection (compatible with conditioner caching) | -| glu_type | str | atanglu | Type of gated linear unit activation. Choose from `'swiglu'` for SwiGLU, `'atanglu'` for ATanGLU | -| expansion_factor | int | 1 | Channel expansion factor within each gated block (not commonly overridden) | +| argument name | type | default | description | +| :-- | :-: | :-: | :-- | +| num_layers | int | 6 | Number of LYNXNet2 blocks, or depth of the network | +| num_channels | int | 1024 | Number of channels, or width of the network | +| kernel_size | int | 31 | Kernel size of the depthwise convolution layers | +| dropout_rate | float | 0.0 | Dropout rate applied in each LYNXNet2 block | +| use_conditioner_cache | bool | true | Whether to use Conv1d-based conditioner projection (compatible with conditioner caching) | +| glu_type | str | `atanglu` | Type of gated linear unit activation. Choose from `swiglu` for SwiGLU, `atanglu` for ATanGLU, `softsign_glu` for SoftSignGLU | +| expansion_factor | int | 1 | Channel expansion factor within each gated block (not commonly overridden) | ### backbone_type @@ -258,11 +265,10 @@ Backbone type of the main decoder/predictor module. ### base_config -Path(s) of other config files that the current config is based on and will override. +Path(s) to other configuration files on which the current configuration is based; values in the current configuration override them. - - +
scopeothers
typeUnion[str, list]
typestr | list[str]
### binarization_args @@ -270,19 +276,19 @@ Path(s) of other config files that the current config is based on and will overr Arguments for binarizers. - +
typedict
typedict[str, Any]
### binarization_args.num_workers -Number of worker subprocesses when running binarizers. More workers can speed up the preprocessing but will consume more memory. 0 means the main processing doing everything. +Number of worker subprocesses when running binarizers. More workers can speed up the preprocessing but will consume more memory. 0 means the main process does everything. - + - +
visibilityall
visibilityacoustic, variance
scopepreprocessing
customizabilityrecommended
typeint
default1
default0
### binarization_args.prefer_ds @@ -294,19 +300,7 @@ Whether to prefer loading attributes and parameters from DS files. scopepreprocessing customizabilityrecommended typebool -defaultFalse - - -### binarization_args.shuffle - -Whether binarized dataset will be shuffled or not. - - - - - - - +
visibilityall
scopepreprocessing
customizabilitynormal
typebool
defaulttrue
defaultfalse
### binarizer_cls @@ -314,10 +308,12 @@ Whether binarized dataset will be shuffled or not. Binarizer class name. - + - + + +
visibilityall
visibilityacoustic, variance
scopepreprocessing
customizabilityreserved
typestr
typestr | None
defaultnull
constraintsThe base configuration may leave this as `null`; the preprocessing entry point requires a non-null importable class name.
### binary_data_dir @@ -325,19 +321,21 @@ Binarizer class name. Path to the binarized dataset. - + - + + +
visibilityall
visibilityacoustic, variance
scopepreprocessing, training
customizabilityrequired
typestr
typestr | None
defaultnull
constraintsThe base configuration may leave this as `null`; a non-null path must be supplied before preprocessing or training.
### breathiness_db_max -Maximum breathiness value in dB used for normalization to [-1, 1]. +Maximum breathiness value in dB used for normalization to [-1, 1]. Note that with [diffusion_type](#diffusion_type) `'ddpm'`, this value is latched into persistent buffers in checkpoints: modifying it for an existing experiment does not raise errors, but is silently overridden by the checkpoint on loading, so it only takes effect when training from scratch; with Rectified Flow it is always read from the current configuration. - + @@ -345,11 +343,11 @@ Maximum breathiness value in dB used for normalization to [-1, 1]. ### breathiness_db_min -Minimum breathiness value in dB used for normalization to [-1, 1]. +Minimum breathiness value in dB used for normalization to [-1, 1]. Note that with [diffusion_type](#diffusion_type) `'ddpm'`, this value is latched into persistent buffers in checkpoints: modifying it for an existing experiment does not raise errors, but is silently overridden by the checkpoint on loading, so it only takes effect when training from scratch; with Rectified Flow it is always read from the current configuration.
visibilityvariance
scopeinference
scopetraining, inference
customizabilityrecommended
typefloat
default-20.0
- - + + @@ -357,7 +355,7 @@ Minimum breathiness value in dB used for normalization to [-1, 1]. ### breathiness_smooth_width -Length of sinusoidal smoothing convolution kernel (in seconds) on extracted breathiness curve. +Length of sinusoidal smoothing convolution kernel (in seconds) on the extracted breathiness curve.
visibilityacoustic, variance
scopeinference
visibilityvariance
scopetraining, inference
customizabilityrecommended
typefloat
default-96.0
@@ -372,10 +370,10 @@ Length of sinusoidal smoothing convolution kernel (in seconds) on extracted brea The value at which to clip gradients. Equivalent to `gradient_clip_val` in `lightning.pytorch.Trainer`.
visibilityacoustic, variance
- + - +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynot recommended
typefloat
typefloat | None
default1
@@ -384,7 +382,7 @@ The value at which to clip gradients. Equivalent to `gradient_clip_val` in `ligh Number of batches loaded in advance by each `torch.utils.data.DataLoader` worker. - + @@ -393,10 +391,10 @@ Number of batches loaded in advance by each `torch.utils.data.DataLoader` worker ### dataset_size_key -The key that indexes the binarized metadata to be used as the `sizes` when batching by size +The key that indexes the binarized metadata to be used as `sizes` when batching by size.
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynormal
typeint
- + @@ -408,28 +406,27 @@ The key that indexes the binarized metadata to be used as the `sizes` when batch List of dataset configs for preprocessing.
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynot recommended
typestr
- - - +
visibilityacoustic, variance
scopepreprocessing
typeList[dict]
typelist[dict[str, Any]]
### datasets[].language -Language context of this dataset. Must be a key of [dictionaries](#dictionaries). +Language context of this dataset. +
visibilityacoustic, variance
scopepreprocessing
customizabilityrequired
typestr
constraintsMust be a key of dictionaries.
### datasets[].raw_data_dir -Path to this dataset including wave files, transcriptions, etc. +Path to this dataset including audio files, transcriptions, etc. - + @@ -437,7 +434,7 @@ Path to this dataset including wave files, transcriptions, etc. ### datasets[].speaker -The name of speaker of this dataset. Speaker names are mapped to speaker indexes and stored into spk_map.json when preprocessing. +The name of the speaker of this dataset. Speaker names are mapped to speaker indexes and stored in spk_map.json when preprocessing.
visibilityall
visibilityacoustic, variance
scopepreprocessing
customizabilityrequired
typestr
@@ -448,38 +445,40 @@ The name of speaker of this dataset. Speaker names are mapped to speaker indexes ### datasets[].spk_id -The speaker ID assigned to this dataset. Will be automatically assigned if not given. IDs can be duplicate or discontinuous to merge multiple datasets to one speaker. +The speaker ID assigned to this dataset. Will be automatically assigned if not given. IDs can be duplicated or discontinuous to merge multiple datasets into one speaker.
visibilityacoustic, variance
- + +
visibilityacoustic, variance
scopepreprocessing
customizabilitynormal
typeint
typeint | None
constraintsMust be smaller than num_spk. The same speaker name must always map to the same ID.
### datasets[].test_prefixes List of data item names or name prefixes in this dataset for the validation set. For each string `s` in the list: -- If `s` equals to an actual item name, add that item to validation set. -- If `s` does not equal to any item names, add all items whose names start with `s` to validation set. +- If `s` equals an actual item name, add that item to the validation set. +- If `s` does not equal any item name, add all items whose names start with `s` to the validation set. - + - +
visibilityall
visibilityacoustic, variance
scopepreprocessing
customizabilityrequired
typelist
typelist[str]
### dictionaries -Map of language names and their corresponding dictionary file paths. The phonemes in these dictionaries will be combined as the final phoneme set and have their phoneme IDs. Training data must fully cover all phoneme IDs. +Map of language names and their corresponding dictionary file paths. The phonemes in these dictionaries will be combined into the final phoneme set and assigned phoneme IDs. Note that the phoneme set built from these dictionaries directly determines the vocabulary size of the token embedding when models are constructed or loaded (in training, inference and ONNX export), and defines how inference inputs are converted to phoneme IDs. The standard format is a mapping; `null` is accepted only for legacy single-dictionary configurations, which must provide the legacy `dictionary` path. - + - + +
visibilityacoustic, variance
scopepreprocessing
scopenn, preprocessing, inference
customizabilityrequired
typeDict[str, str]
typedict[str, str] | None
constraintsEvery phoneme ID in the final phoneme set must occur in at least one data item, including validation items.
default{}
@@ -487,10 +486,10 @@ Map of language names and their corresponding dictionary file paths. The phoneme DDPM sampling acceleration method. The following methods are currently available: -- DDIM: the DDIM method from [Denoising Diffusion Implicit Models](https://arxiv.org/abs/2010.02502) -- PNDM: the PLMS method from [Pseudo Numerical Methods for Diffusion Models on Manifolds](https://arxiv.org/abs/2202.09778) -- DPM-Solver++ adapted from [DPM-Solver: A Fast ODE Solver for Diffusion Probabilistic Model Sampling in Around 10 Steps](https://github.com/LuChengTHU/dpm-solver) -- UniPC adapted from [UniPC: A Unified Predictor-Corrector Framework for Fast Sampling of Diffusion Models](https://github.com/wl-zhao/UniPC) +- DDIM: the DDIM method from [Denoising Diffusion Implicit Models](https://arxiv.org/abs/2010.02502). +- PNDM: the PLMS method from [Pseudo Numerical Methods for Diffusion Models on Manifolds](https://arxiv.org/abs/2202.09778). +- DPM-Solver++ adapted from [DPM-Solver: A Fast ODE Solver for Diffusion Probabilistic Model Sampling in Around 10 Steps](https://github.com/LuChengTHU/dpm-solver). +- UniPC adapted from [UniPC: A Unified Predictor-Corrector Framework for Fast Sampling of Diffusion Models](https://github.com/wl-zhao/UniPC). @@ -511,19 +510,21 @@ DDPM sampling speed-up ratio. 1 means no speeding up. - +
visibilityacoustic, variance
customizabilitynormal
typeint
default10
constraintsMust be a factor of K_step.
constraintsMust be a factor of K_step_infer.
### diffusion_type -The type of ODE-based generative model algorithm. The following models are currently available: +The generative modeling algorithm used by the main decoder/predictor module. The following algorithms are currently available: - Denoising Diffusion Probabilistic Models (DDPM) from [Denoising Diffusion Probabilistic Models](https://arxiv.org/abs/2006.11239) - Rectified Flow from [Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow](https://arxiv.org/abs/2209.03003) +Modifying it switches the algorithm family used by training loss computation and by inference sampling, and results in failure when loading or resuming from checkpoints, because DDPM and Rectified Flow modules keep different saved states. + - + @@ -532,11 +533,11 @@ The type of ODE-based generative model algorithm. The following models are curre ### dropout -Dropout rate in some FastSpeech2 modules. +Dropout rate in some FastSpeech2 modules. Modifying it does not change any parameter or saved state, so it does not prevent checkpoint loading; dropout is inactive in evaluation, so modifications only silently affect training behavior.
visibilityacoustic, variance
scopenn
scopenn, training, inference
customizabilitynormal
typestr
defaultreflow
- + @@ -544,14 +545,15 @@ Dropout rate in some FastSpeech2 modules. ### ds_workers -Number of workers of `torch.utils.data.DataLoader`. +Number of workers for `torch.utils.data.DataLoader`.
visibilityacoustic, variance
scopenn
scopetraining
customizabilitynot recommended
typefloat
default0.1
- + +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynormal
typeint
default4
constraintsMust be at least 1. The data loaders are always constructed with a non-null prefetch factor and persistent_workers=True; setting this to 0 makes torch.utils.data.DataLoader raise a ValueError at the very beginning of training or validation.
### dur_prediction_args @@ -559,7 +561,7 @@ Number of workers of `torch.utils.data.DataLoader`. Arguments for phoneme duration prediction. - +
typedict
typedict[str, Any]
### dur_prediction_args.arch @@ -569,7 +571,7 @@ Architecture of duration predictor. `'fs2'` uses the original FastSpeech2 durati - + @@ -577,11 +579,11 @@ Architecture of duration predictor. `'fs2'` uses the original FastSpeech2 durati ### dur_prediction_args.dropout -Dropout rate in duration predictor. +Dropout rate in duration predictor. Like [dropout](#dropout), modifying it does not change any parameter or saved state, so it does not prevent checkpoint loading and only silently affects training behavior.
visibilityvariance
scopenn
customizabilityreserved
customizabilitynormal
typestr
defaultresnet
constraintsChoose from 'fs2', 'resnet'.
- + @@ -613,7 +615,7 @@ Kernel size of convolution layers of duration predictor. ### dur_prediction_args.lambda_pdur_loss -Coefficient of single phone duration loss when calculating joint duration loss. +Coefficient of single-phoneme duration loss when calculating joint duration loss.
visibilityvariance
scopenn
scopetraining
customizabilitynot recommended
typefloat
default0.1
@@ -657,7 +659,7 @@ with the offset value $d$.
visibilityvariance
- + @@ -690,7 +692,7 @@ Number of duration predictor layers. ### enc_ffn_kernel_size -Size of TransformerFFNLayer convolution kernel size in FastSpeech2 encoder. +Size of TransformerFFNLayer convolution kernel in FastSpeech2 encoder.
visibilityvariance
scopetraining
scopetraining, inference
customizabilitynot recommended
typefloat
default1.0
@@ -714,11 +716,11 @@ Number of FastSpeech2 encoder layers. ### energy_db_max -Maximum energy value in dB used for normalization to [-1, 1]. +Maximum energy value in dB used for normalization to [-1, 1]. Note that with [diffusion_type](#diffusion_type) `'ddpm'`, this value is latched into persistent buffers in checkpoints: modifying it for an existing experiment does not raise errors, but is silently overridden by the checkpoint on loading, so it only takes effect when training from scratch; with Rectified Flow it is always read from the current configuration.
visibilityacoustic, variance
- + @@ -726,11 +728,11 @@ Maximum energy value in dB used for normalization to [-1, 1]. ### energy_db_min -Minimum energy value in dB used for normalization to [-1, 1]. +Minimum energy value in dB used for normalization to [-1, 1]. Note that with [diffusion_type](#diffusion_type) `'ddpm'`, this value is latched into persistent buffers in checkpoints: modifying it for an existing experiment does not raise errors, but is silently overridden by the checkpoint on loading, so it only takes effect when training from scratch; with Rectified Flow it is always read from the current configuration.
visibilityvariance
scopeinference
scopetraining, inference
customizabilityrecommended
typefloat
default-12.0
- + @@ -738,7 +740,7 @@ Minimum energy value in dB used for normalization to [-1, 1]. ### energy_smooth_width -Length of sinusoidal smoothing convolution kernel (in seconds) on extracted energy curve. +Length of sinusoidal smoothing convolution kernel (in seconds) on the extracted energy curve.
visibilityvariance
scopeinference
scopetraining, inference
customizabilityrecommended
typefloat
default-96.0
@@ -750,37 +752,37 @@ Length of sinusoidal smoothing convolution kernel (in seconds) on extracted ener ### extra_phonemes -Extra phonemes to be added to the phoneme set. This list can be used to define custom global phoneme tags besides `AP` and `SP`, or to contain phonemes that are not present in any of the dictionaries. +Extra phonemes to be added to the phoneme set. This list can be used to define custom global phoneme tags besides `AP` and `SP`, or to contain phonemes that are not present in any of the dictionaries. Like [dictionaries](#dictionaries), this list directly determines the vocabulary size of the token embedding when models are constructed or loaded.
visibilityacoustic, variance
- + - +
visibilityacoustic, variance
scopepreprocessing
scopenn, preprocessing, inference
customizabilitynormal
typelist
typelist[str] | None
default[]
### f0_max -Maximum base frequency (F0) in Hz for pitch extraction. +Maximum fundamental frequency (F0) in Hz for pitch extraction. - +
visibilityacoustic, variance
scopepreprocessing
customizabilitynormal
typeint
typefloat
default1100
### f0_min -Minimum base frequency (F0) in Hz for pitch extraction. +Minimum fundamental frequency (F0) in Hz for pitch extraction. - +
visibilityacoustic, variance
scopepreprocessing
customizabilitynormal
typeint
typefloat
default65
@@ -791,6 +793,10 @@ Activation function of TransformerFFNLayer in FastSpeech2 encoder: - `torch.nn.ReLU` if 'relu' - `torch.nn.GELU` if 'gelu' - `torch.nn.SiLU` if 'swish' +- `SwiGLU` if 'swiglu' +- `ATanGLU` if 'atanglu' + +The last two are gated linear unit activations (the filter size of the first convolution is internally doubled to compensate for the halved output of the GLU). Switching between a GLU-family activation and a non-GLU one changes parameter shapes and prevents checkpoint loading; switching within the non-GLU family (`relu`, `gelu`, `swish`) keeps shapes unchanged and does not prevent checkpoint loading, but silently changes the behavior of an already trained model. @@ -798,16 +804,16 @@ Activation function of TransformerFFNLayer in FastSpeech2 encoder: - +
visibilityacoustic, variance
customizabilitynot recommended
typestr
defaultgelu
constraintsChoose from 'relu', 'gelu', 'swish'.
constraintsChoose from 'relu', 'gelu', 'swish', 'swiglu', 'atanglu'.
### fft_size -Fast Fourier Transforms parameter for mel extraction. +Fast Fourier Transform parameter for mel extraction. - + @@ -818,11 +824,11 @@ Fast Fourier Transforms parameter for mel extraction. Whether to finetune from a pretrained model.
visibilityacoustic, variance
scopepreprocessing
scopepreprocessing, inference
customizabilityreserved
typeint
default2048
- + - +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynormal
typebool
defaultFalse
defaultfalse
### finetune_ckpt_path @@ -830,10 +836,10 @@ Whether to finetune from a pretrained model. Path to the pretrained model for finetuning. - + - +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynormal
typestr
typestr | None
defaultnull
@@ -842,33 +848,34 @@ Path to the pretrained model for finetuning. Prefixes of parameter key names in the state dict of the pretrained model that need to be dropped before finetuning. - + - + +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynormal
typelist
typelist[str] | None
default[]
### finetune_strict_shapes -Whether to raise error if the tensor shapes of any parameter of the pretrained model and the target model mismatch. If set to `False`, parameters with mismatching shapes will be skipped. +Whether to raise an error if the tensor shapes of any parameter of the pretrained model and the target model mismatch. If set to `false`, parameters with mismatching shapes will be skipped. - + - +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynormal
typebool
defaultTrue
defaulttrue
### fmax -Maximum frequency of mel extraction. +Maximum frequency of mel extraction. `null` uses the Nyquist frequency (`audio_sample_rate / 2`). - + - +
visibilityacoustic
scopepreprocessing
scopepreprocessing, inference
customizabilityreserved
typeint
typefloat | None
default16000
@@ -878,22 +885,22 @@ Minimum frequency of mel extraction. - + - +
visibilityacoustic
scopepreprocessing
scopepreprocessing, inference
customizabilityreserved
typeint
typefloat
default40
### freezing_enabled -Whether enabling parameter freezing during training. +Whether to enable parameter freezing during training. - + - +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynormal
typebool
defaultFalse
defaultfalse
### frozen_params @@ -901,20 +908,20 @@ Whether enabling parameter freezing during training. Parameter name prefixes to freeze during training. - + - +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynormal
typelist
typelist[str]
default[]
### glide_embed_scale -The scale factor to be multiplied on the glide embedding values for melody encoder. +The scale factor by which the glide embedding values are multiplied for melody encoder. - + @@ -926,10 +933,11 @@ Type names of glide notes.
visibilityvariance
scopenn
scopetraining, inference
customizabilitynot recommended
typefloat
default11.313708498984760
- + - - + + +
visibilityvariance
scopepreprocessing
scopenn, preprocessing, inference
customizabilitynormal
typelist
default[up, down]
typelist[str]
default['up', 'down']
constraintsType name none is reserved (index 0 in the glide embedding, whose size is len(glide_types) + 1) and must not appear in this list.
### hidden_size @@ -949,11 +957,11 @@ Dimension of hidden layers of FastSpeech2, token and parameter embeddings, and d Harmonic-noise separation algorithm type. - + - +
visibilityall
visibilityacoustic, variance
scopepreprocessing
customizabilitynormal
typestr
defaultworld
defaultvr
constraintsChoose from 'world', 'vr'.
@@ -962,10 +970,11 @@ Harmonic-noise separation algorithm type. Checkpoint or model path of NN-based harmonic-noise separator. - + +
visibilityall
visibilityacoustic, variance
scopepreprocessing
customizabilitynormal
typestr
defaultcheckpoints/vr/model.pt
### hop_size @@ -974,7 +983,7 @@ Hop size or step length (in number of waveform samples) of mel and feature extra - + @@ -1030,19 +1039,20 @@ Coefficient of variance loss (all variance parameters other than pitch, like ene ### K_step -Maximum number of DDPM steps used by shallow diffusion. +Maximum number of DDPM steps used by shallow diffusion. Only takes effect when [diffusion_type](#diffusion_type) is `'ddpm'` and [use_shallow_diffusion](#use_shallow_diffusion) is set to `true`; with Rectified Flow the shallow starting point is controlled by [T_start](#t_start) instead, and this key is ignored.
visibilityacoustic, variance
scopepreprocessing
scopepreprocessing, inference
customizabilityreserved
typeint
default512
- + +
visibilityacoustic
scopetraining
scopetraining, inference
customizabilityrecommended
typeint
default400
constraintsMust not be larger than timesteps.
### K_step_infer -Number of DDPM steps used during shallow diffusion inference. Normally set as same as [K_step](#K_step). +Number of DDPM steps used during shallow diffusion inference. Normally set to the same value as [K_step](#k_step). Only takes effect when [diffusion_type](#diffusion_type) is `'ddpm'` and [use_shallow_diffusion](#use_shallow_diffusion) is set to `true`; with Rectified Flow the shallow starting point is controlled by [T_start_infer](#t_start_infer) instead, and this key is ignored. @@ -1050,15 +1060,15 @@ Number of DDPM steps used during shallow diffusion inference. Normally set as sa - +
visibilityacoustic
customizabilityrecommended
typeint
default400
constraintsShould be no larger than K_step.
constraintsShould be no larger than K_step. Values larger than K_step are silently clamped to K_step instead of raising errors.
### log_interval -Controls how often to log within training steps. Equivalent to `log_every_n_steps` in `lightning.pytorch.Trainer`. +Controls how often training metrics are logged to TensorBoard, measured in global training steps. - + @@ -1070,7 +1080,7 @@ Controls how often to log within training steps. Equivalent to `log_every_n_step Arguments of learning rate scheduler. Keys will be used as keyword arguments of the `__init__()` method of [lr_scheduler_args.scheduler_cls](#lr_scheduler_argsscheduler_cls).
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynormal
typeint
- +
typedict
typedict[str, Any]
### lr_scheduler_args.scheduler_cls @@ -1078,7 +1088,7 @@ Arguments of learning rate scheduler. Keys will be used as keyword arguments of Learning rate scheduler class name. - + @@ -1087,13 +1097,14 @@ Learning rate scheduler class name. ### main_loss_log_norm -Whether to use log-normalized weight for the main loss. This is similar to the method in the Stable Diffusion 3 paper [Scaling Rectified Flow Transformers for High-Resolution Image Synthesis](https://arxiv.org/abs/2403.03206). +Whether to use log-normalized weight for the main loss. This is similar to the method in the Stable Diffusion 3 paper [Scaling Rectified Flow Transformers for High-Resolution Image Synthesis](https://arxiv.org/abs/2403.03206). Only takes effect when [diffusion_type](#diffusion_type) is `'reflow'`; ignored with DDPM.
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynot recommended
typestr
+
visibilityacoustic, variance
scopetraining
customizabilitynormal
typebool
defaultfalse
### main_loss_type @@ -1118,7 +1129,7 @@ Maximum number of data frames in each training batch. Used to dynamically contro scopetraining customizabilityrecommended typeint -default80000 +default50000 ### max_batch_size @@ -1126,20 +1137,20 @@ Maximum number of data frames in each training batch. Used to dynamically contro The maximum training batch size. - + - +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilityrecommended
typeint
default48
default64
### max_beta -Max beta of the DDPM noise schedule. +Max beta of the DDPM noise schedule. Only takes effect when [diffusion_type](#diffusion_type) is `'ddpm'` and [schedule_type](#schedule_type) is `'linear'`; ignored with Rectified Flow and with the cosine schedule. The noise schedule derived from this value is saved as persistent buffers in checkpoints: modifying it for an existing experiment does not raise errors, but the value is silently overridden by the buffers stored in the checkpoint, so it only takes effect when training from scratch. - + @@ -1150,7 +1161,7 @@ Max beta of the DDPM noise schedule. Stop training after this number of steps. Equivalent to `max_steps` in `lightning.pytorch.Trainer`.
visibilityacoustic, variance
scopenn, inference
scopetraining, inference
customizabilitynormal
typefloat
default0.02
- + @@ -1174,7 +1185,7 @@ Maximum number of data frames in each validation batch. The maximum validation batch size.
visibilityall
visibilityacoustic, variance
scopetraining
customizabilityrecommended
typeint
- + @@ -1183,21 +1194,22 @@ The maximum validation batch size. ### mel_base -The logarithmic base of mel spectrogram calculation. +The logarithmic base of the mel-spectrogram calculation. The legacy value `10` (integer or string `'10'`) and the natural-log value `'e'` are accepted by vocoder compatibility paths. New dataset preprocessing and NSF-HiFiGAN export require `'e'`. -**WARNING: Since v2.4.0 release, this value is no longer configurable for preprocessing new datasets.** +**WARNING: Since the v2.4.0 release, this value is no longer configurable for preprocessing new datasets.**
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynormal
typeint
- + - + +
visibilityacoustic
scopepreprocessing
scopepreprocessing, inference
customizabilityreserved
typestr
typestr | int
defaulte
constraintsUse `'e'` for current preprocessing and export; legacy vocoder paths may also accept `'10'` or `10`.
### mel_vmax -Maximum mel spectrogram heatmap value for TensorBoard plotting. +Maximum mel-spectrogram heatmap value for TensorBoard plotting. @@ -1209,7 +1221,7 @@ Maximum mel spectrogram heatmap value for TensorBoard plotting. ### mel_vmin -Minimum mel spectrogram heatmap value for TensorBoard plotting. +Minimum mel-spectrogram heatmap value for TensorBoard plotting.
visibilityacoustic
@@ -1221,21 +1233,21 @@ Minimum mel spectrogram heatmap value for TensorBoard plotting. ### melody_encoder_args -Arguments for melody encoder. Available sub-keys: `hidden_size`, `enc_layers`, `enc_ffn_kernel_size`, `ffn_act`, `dropout`, `num_heads`, `use_pos_embed`, `rel_pos`. If either of the parameter does not exist in this configuration key, it inherits from the linguistic encoder. +Arguments for melody encoder. Available sub-keys: `hidden_size`, `enc_layers`, `enc_ffn_kernel_size`, `ffn_act`, `dropout`, `num_heads`, `use_pos_embed`, `rel_pos`, `use_rope`. If any parameter does not exist in this configuration key, it inherits from the linguistic encoder. The scope implications of each sub-key follow the root-level keys of the same names.
visibilityacoustic
- +
typedict
typedict[str, Any]
### merged_phoneme_groups -Phoneme groups to merge. Each group is a phoneme name list. The merged phonemes share the same ID and thus the same phoneme embedding. +Phoneme groups to merge. Each group is a phoneme name list. The merged phonemes share the same ID and thus the same phoneme embedding. Like [dictionaries](#dictionaries), these groups directly determine the vocabulary size of the token embedding when models are constructed or loaded. - - - + + +
visibilityacoustic, variance
scopepreprocessing
customizabilityrequired
typelist
scopenn, preprocessing, inference
customizabilitynormal
typelist[list[str]] | None
default[]
@@ -1245,7 +1257,7 @@ Length of sinusoidal smoothing convolution kernel (in seconds) on the step funct - + @@ -1257,18 +1269,19 @@ List of 0-based encoder layer indices where Mixed LayerNorm is applied. Only tak
visibilityvariance
scopepreprocessing
scopepreprocessing, inference
customizabilitynormal
typefloat
default0.06
- + - + +
visibilityacoustic
scopenn
scopenn, inference
customizabilitynormal
typeList[int]
typelist[int]
default[0, 2]
constraintsEvery element should be in the range [0, enc_layers).
### nccl_p2p -Whether to enable P2P when using NCCL as the backend. Turn it to `false` if the training process is stuck upon beginning. +Whether to enable P2P when using NCCL as the backend. Set it to `false` if the training process is stuck upon beginning. - + @@ -1280,23 +1293,24 @@ Whether to enable P2P when using NCCL as the backend. Turn it to `false` if the Number of newest checkpoints kept during training.
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynormal
typebool
- + - +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynormal
typeint
default5
default8
### num_heads -The number of attention heads of `torch.nn.MultiheadAttention` in FastSpeech2 encoder. +The number of attention heads of the in-house `MultiheadSelfAttentionWithRoPE` (formerly `torch.nn.MultiheadAttention`, which has been deprecated due to ONNX export issues) in FastSpeech2 encoder. This does not change parameter shapes (the Q/K/V and output projections have the same shapes regardless of the number of heads); modifying it does not prevent checkpoint loading, but silently changes the behavior of an already trained model. - + +
visibilityacoustic, variance
scopenn
scopetraining, inference
customizabilitynot recommended
typeint
default2
constraintshidden_size must be divisible by num_heads. When both use_pos_embed and use_rope are true, hidden_size must be divisible by 2 × num_heads.
### num_lang @@ -1308,6 +1322,8 @@ Number of languages. This value is used to allocate language embeddings in the l scopenn customizabilityrequired typeint +default1 +constraintsMust be at least the number of entries in dictionaries. ### num_sanity_val_steps @@ -1315,10 +1331,10 @@ Number of languages. This value is used to allocate language embeddings in the l Number of sanity validation steps at the beginning. - + - - + +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilityreserved
typeint
customizabilitynormal
typeint | None
default1
@@ -1336,7 +1352,7 @@ Maximum number of speakers in multi-speaker models. ### num_valid_plots -Number of validation plots in each validation. Plots will be chosen from the start of the validation set. +Number of validation plots for each validation run. Plots will be chosen from the start of the validation set. @@ -1348,23 +1364,23 @@ Number of validation plots in each validation. Plots will be chosen from the sta ### optimizer_args -Arguments of optimizer. Keys will be used as keyword arguments of the `__init__()` method of [optimizer_args.optimizer_cls](#optimizer_argsoptimizer_cls). +Arguments of optimizer. Keys will be used as keyword arguments of the `__init__()` method of [optimizer_args.optimizer_cls](#optimizer_argsoptimizer_cls).
visibilityacoustic, variance
- +
typedict
typedict[str, Any]
### optimizer_args.optimizer_cls Optimizer class name. The following optimizers are currently recommended: -- `torch.optim.AdamW` — Standard AdamW optimizer. Use with `adamw_args` for the weight decay setting. -- `modules.optimizer.muon.Muon_AdamW` — Chained optimizer that applies Muon (MomentUm Orthogonalized by Newton-schulz) to internal weight matrices (e.g. linear layers) and AdamW to other parameters (e.g. biases, embeddings). Configure via `muon_args` and `adamw_args` sub-keys under [optimizer_args](#optimizer_args). +- `torch.optim.AdamW` — Standard AdamW optimizer. Set `weight_decay` and other arguments (`lr`, `betas`, `eps`, ...) as top-level keys of [optimizer_args](#optimizer_args). +- `modules.optimizer.muon.Muon_AdamW` — Chained optimizer that applies Muon (MomentUm Orthogonalized by Newton-Schulz) to internal weight matrices (e.g. linear layers) and AdamW to other parameters (e.g. biases, embeddings). Per-optimizer arguments are configured via the `muon_args` and `adamw_args` sub-keys under [optimizer_args](#optimizer_args), while the top-level `lr` and `weight_decay` serve as the shared defaults of both sub-optimizers. Note that an `lr` set in either sub-key takes no effect in practice: at every `optimizer.step()` the top-level `lr` is copied into all parameter groups of the sub-optimizers, so that the learning rate scheduler keeps applying. - + - +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilityreserved
customizabilitynormal
typestr
defaultmodules.optimizer.muon.Muon_AdamW
@@ -1374,7 +1390,7 @@ Optimizer class name. The following optimizers are currently recommended: Pitch extraction algorithm type. - + @@ -1387,31 +1403,34 @@ Pitch extraction algorithm type. Checkpoint or model path of NN-based pitch extractor.
visibilityall
visibilityacoustic, variance
scopepreprocessing
customizabilitynormal
typestr
- + +
visibilityall
visibilityacoustic, variance
scopepreprocessing
customizabilitynormal
typestr
defaultcheckpoints/rmvpe/model.pt
### permanent_ckpt_interval -The interval (in number of training steps) of permanent checkpoints. Permanent checkpoints will not be removed even if they are not the newest ones. +The interval (in number of training steps) of permanent checkpoints. Permanent checkpoints will not be removed even if they are not the newest ones. Permanent checkpoints are enabled only when this value is larger than 9 and [permanent_ckpt_start](#permanent_ckpt_start) is larger than 0; `null` or `false` is normalized to 0 and silently disables them. - + - + +
visibilityall
visibilityacoustic, variance
scopetraining
typeint
customizabilitynormal
typeint | bool | None
default10000
### permanent_ckpt_start -Checkpoints will be marked as permanent every [permanent_ckpt_interval](#permanent_ckpt_interval) training steps after this number of training steps. +Checkpoints are only saved at validation checks, i.e. every [val_check_interval](#val_check_interval) global steps (the interval passed to the trainer is multiplied by [accumulate_grad_batches](#accumulate_grad_batches), so proportionally more micro-batches run between validation checks when gradient accumulation is enabled). A saved checkpoint is kept as permanent if its step count is no less than this value and the difference is divisible by [permanent_ckpt_interval](#permanent_ckpt_interval). Milestone steps that do not coincide with a saved checkpoint are skipped, so the effective cadence of permanent checkpoints is the least common multiple of the two intervals. Permanent checkpoints are enabled only when this value is larger than 0 and [permanent_ckpt_interval](#permanent_ckpt_interval) is larger than 9; `null` or `false` is normalized to 0 and silently disables them. - + - + +
visibilityall
visibilityacoustic, variance
scopetraining
typeint
customizabilitynormal
typeint | bool | None
default60000
@@ -1420,24 +1439,28 @@ Checkpoints will be marked as permanent every [permanent_ckpt_interval](#permane Arguments for pitch prediction. - +
typedict
typedict[str, Any]
### pitch_prediction_args.backbone_args -Equivalent to [backbone_args](#backbone_args) but only for the pitch predictor model. If not set, use the root backbone type. +Equivalent to [backbone_args](#backbone_args) but only for the pitch predictor model. - +
visibilityvariance
typedict[str, Any]
### pitch_prediction_args.backbone_type -Equivalent to [backbone_type](#backbone_type) but only for the pitch predictor model. +Equivalent to [backbone_type](#backbone_type) but only for the pitch predictor model. If not set, use the root backbone type. + + + +
visibilityvariance
scopenn
customizabilitynormal
typestr
defaultlynxnet2
constraintsChoose from 'wavenet', 'lynxnet', 'lynxnet2'.
### pitch_prediction_args.pitd_clip_max @@ -1446,7 +1469,8 @@ Maximum clipping value (in semitones) of pitch delta between actual pitch and ba - + +
visibilityvariance
scopeinference
scopetraining, inference
customizabilitynormal
typefloat
default12.0
@@ -1457,18 +1481,19 @@ Minimum clipping value (in semitones) of pitch delta between actual pitch and ba - + +
visibilityvariance
scopeinference
scopetraining, inference
customizabilitynormal
typefloat
default-12.0
### pitch_prediction_args.pitd_norm_max -Maximum pitch delta value in semitones used for normalization to [-1, 1]. +Maximum pitch delta value in semitones used for normalization to [-1, 1]. Note that with [diffusion_type](#diffusion_type) `'ddpm'`, this value is latched into persistent buffers in checkpoints: modifying it for an existing experiment does not raise errors, but is silently overridden by the checkpoint on loading, so it only takes effect when training from scratch; with Rectified Flow it is always read from the current configuration. - + @@ -1476,11 +1501,11 @@ Maximum pitch delta value in semitones used for normalization to [-1, 1]. ### pitch_prediction_args.pitd_norm_min -Minimum pitch delta value in semitones used for normalization to [-1, 1]. +Minimum pitch delta value in semitones used for normalization to [-1, 1]. Note that with [diffusion_type](#diffusion_type) `'ddpm'`, this value is latched into persistent buffers in checkpoints: modifying it for an existing experiment does not raise errors, but is silently overridden by the checkpoint on loading, so it only takes effect when training from scratch; with Rectified Flow it is always read from the current configuration.
visibilityvariance
scopeinference
scopetraining, inference
customizabilityrecommended
typefloat
default8.0
- + @@ -1503,7 +1528,7 @@ Number of repeating bins in the pitch predictor. Type of Lightning trainer hardware accelerator.
visibilityvariance
scopeinference
scopetraining, inference
customizabilityrecommended
typefloat
default-8.0
- + @@ -1513,15 +1538,15 @@ Type of Lightning trainer hardware accelerator. ### pl_trainer_devices -To determine on which device(s) model should be trained. +Determines which device(s) the model should be trained on. -'auto' will utilize all visible devices defined with the `CUDA_VISIBLE_DEVICES` environment variable, or utilize all available devices if that variable is not set. Otherwise, it behaves like `CUDA_VISIBLE_DEVICES` which can filter out visible devices. +`'auto'` will utilize all visible devices defined with the `CUDA_VISIBLE_DEVICES` environment variable, or utilize all available devices if that variable is not set. Otherwise, it behaves like `CUDA_VISIBLE_DEVICES` which can filter out visible devices. Lightning also accepts a positive device count as an integer or a list of device indices.
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynot recommended
typestr
- + - +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynot recommended
typestr
typestr | int | list[int]
defaultauto
@@ -1530,12 +1555,12 @@ To determine on which device(s) model should be trained. The computation precision of training. - + - + - +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilitynormal
typestr
typestr | int | None
default16-mixed
constraintsChoose from '32-true', 'bf16-mixed', '16-mixed'. See more possible values at Trainer — PyTorch Lightning 2.X.X documentation.
constraintsLightning accepts integer precisions `16`, `32`, `64` and string forms such as `'32-true'`, `'bf16-mixed'` and `'16-mixed'`; `null` is passed through to Lightning and falls back to `'32-true'`. See the Trainer — PyTorch Lightning 2.X.X documentation for the version-specific list.
### pl_trainer_num_nodes @@ -1543,7 +1568,7 @@ The computation precision of training. Number of nodes in the training cluster of Lightning trainer. - + @@ -1555,7 +1580,7 @@ Number of nodes in the training cluster of Lightning trainer. Arguments of Lightning Strategy. Values will be used as keyword arguments when constructing the Strategy object.
visibilityall
visibilityacoustic, variance
scopetraining
customizabilityreserved
typeint
- +
typedict
typedict[str, Any]
### pl_trainer_strategy.name @@ -1563,7 +1588,7 @@ Arguments of Lightning Strategy. Values will be used as keyword arguments when c Strategy name for the Lightning trainer. - + @@ -1644,23 +1669,23 @@ Whether to enable voicing prediction. ### rel_pos -Whether to use relative positional encoding in FastSpeech2 module. +Whether to use relative positional encoding in FastSpeech2 module. Only consulted when [use_rope](#use_rope) is `false`: with `rel_pos: false` the encoder uses `SinusoidalPositionalEmbedding`, which owns a persistent buffer saved in checkpoints, so toggling this option changes the set of saved keys and results in failure when loading or resuming from checkpoints.
visibilityall
visibilityacoustic, variance
scopetraining
customizabilityreserved
typestr
- +
visibilityacoustic, variance
scopenn
customizabilitynot recommended
typeboolean
typebool
defaulttrue
### rope_interleaved -Whether to use the interleaved (alternating) layout for RoPE (Rotary Positional Encoding) in the encoder self-attention. When set to `false`, the non-interleaved (contiguous half-real-half-imaginary) layout is used instead. +Whether to use the interleaved (alternating) layout for RoPE (Rotary Positional Encoding) in the encoder self-attention. When set to `false`, the non-interleaved (contiguous half-real-half-imaginary) layout is used instead. This option only changes the layout of the frequency buffers, which are recomputed at initialization; modifying it does not change parameter shapes or prevent checkpoint loading, but silently changes the behavior of an already trained model. - + @@ -1668,13 +1693,15 @@ Whether to use the interleaved (alternating) layout for RoPE (Rotary Positional ### sampler_frame_count_grid -The batch sampler applies an algorithm called _sorting by similar length_ when collecting batches. Data samples are first grouped by their approximate lengths before they get shuffled within each group. Assume this value is set to $L_{grid}$, the approximate length of a data sample with length $L_{real}$ can be calculated through the following expression: +The batch sampler applies an algorithm called _sorting by similar length_ when collecting batches. Data samples are first shuffled, and then stably sorted by their approximate lengths, so that samples of similar lengths are grouped together while the order within each group stays random. Assuming this value is set to $L_{grid}$, the approximate length of a data sample with length $L_{real}$ can be calculated through the following expression: $$ -L_{approx} = \lfloor\frac{L_{real}}{L_{grid}}\rfloor\cdot L_{grid} +L_{approx} = \max\left(\mathrm{round}\left(\frac{L_{real}}{L_{grid}}\right)\cdot L_{grid},\; L_{grid}\right) $$ -Training performance on some datasets may be very sensitive to this value. Change it to 1 (completely sorted by length without shuffling) to get the best performance in theory. +where $\mathrm{round}$ is the nearest-integer rounding (round half to even), and the result is clamped to a minimum of $L_{grid}$. + +Training performance on some datasets may be very sensitive to this value. Change it to 1 (approximate length becomes the exact length, so batches are perfectly sorted by length) to get the best performance in theory.
visibilityacoustic, variance
scopenn
scopetraining, inference
customizabilitynot recommended
typebool
defaultfalse
@@ -1688,10 +1715,10 @@ Training performance on some datasets may be very sensitive to this value. Chang The algorithm to solve the ODE of Rectified Flow. The following methods are currently available: -- Euler: The Euler method. -- Runge-Kutta (order 2): The 2nd-order Runge-Kutta method. -- Runge-Kutta (order 4): The 4th-order Runge-Kutta method. -- Runge-Kutta (order 5): The 5th-order Runge-Kutta method. +- Euler: the Euler method. +- Runge-Kutta (order 2): the 2nd-order Runge-Kutta method. +- Runge-Kutta (order 4): the 4th-order Runge-Kutta method. +- Runge-Kutta (order 5): the 5th-order Runge-Kutta method.
visibilityacoustic, variance
@@ -1704,7 +1731,7 @@ The algorithm to solve the ODE of Rectified Flow. The following methods are curr ### sampling_steps -The total sampling steps to solve the ODE of Rectified Flow. Note that this value may not equal to NFE (Number of Function Evaluations) because some methods may require more than one function evaluation per step. +The total number of sampling steps to solve the Rectified Flow ODE. Note that this value may not be equal to NFE (Number of Function Evaluations) because some methods may require more than one function evaluation per step.
visibilityacoustic, variance
@@ -1716,11 +1743,11 @@ The total sampling steps to solve the ODE of Rectified Flow. Note that this valu ### schedule_type -The DDPM schedule type. +The DDPM schedule type. Only takes effect when [diffusion_type](#diffusion_type) is `'ddpm'`; ignored with Rectified Flow. Like [max_beta](#max_beta), the derived noise schedule is saved as persistent buffers in checkpoints, so modifying this value for an existing experiment is silently overridden on checkpoint loading and only takes effect when training from scratch.
visibilityacoustic, variance
- + @@ -1732,7 +1759,7 @@ The DDPM schedule type. Arguments for shallow diffusion.
visibilityacoustic, variance
scopenn
scopetraining, inference
customizabilitynot recommended
typestr
defaultlinear
- +
typedict
typedict[str, Any]
### shallow_diffusion_args.aux_decoder_arch @@ -1753,9 +1780,7 @@ Architecture type of the auxiliary decoder. Keyword arguments for dynamically constructing the auxiliary decoder. - - - +
visibilityacoustic
scopenn
typedict
typedict[str, Any]
### shallow_diffusion_args.aux_decoder_grad @@ -1772,7 +1797,7 @@ Scale factor of the gradients from the auxiliary decoder to the encoder. ### shallow_diffusion_args.train_aux_decoder -Whether to forward and backward the auxiliary decoder during training. If set to `false`, the auxiliary decoder hangs in the memory and does not get any updates. +Whether to run the auxiliary decoder in both the forward and backward passes during training. If set to `false`, the auxiliary decoder remains in memory and does not get any updates. @@ -1784,7 +1809,7 @@ Whether to forward and backward the auxiliary decoder during training. If set to ### shallow_diffusion_args.train_diffusion -Whether to forward and backward the diffusion (main) decoder during training. If set to `false`, the diffusion decoder hangs in the memory and does not get any updates. +Whether to run the diffusion (main) decoder in both the forward and backward passes during training. If set to `false`, the diffusion decoder remains in memory and does not get any updates.
visibilityacoustic
@@ -1796,11 +1821,11 @@ Whether to forward and backward the diffusion (main) decoder during training. If ### shallow_diffusion_args.val_gt_start -Whether to use the ground truth as `x_start` in the shallow diffusion validation process. If set to `true`, gaussian noise is added to the ground truth before shallow diffusion is performed; otherwise the noise is added to the output of the auxiliary decoder. This option is useful when the auxiliary decoder has not been trained yet. +Whether to use the ground truth as `x_start` in the shallow diffusion validation process. If set to `true`, Gaussian noise is added to the ground truth before shallow diffusion is performed; otherwise the noise is added to the output of the auxiliary decoder. This option is useful when the auxiliary decoder has not been trained yet. It only takes effect in validation runs during training, where a ground truth mel-spectrogram is available; pure inference (where none is given) is unaffected.
visibilityacoustic
- + @@ -1820,43 +1845,46 @@ Whether to apply the _sorting by similar length_ algorithm described in [sampler ### spec_min -Minimum mel spectrogram value used for normalization to [-1, 1]. Different mel bins can have different minimum values. +Minimum mel-spectrogram value used for normalization to [-1, 1]. Different mel bins can have different minimum values. Note that with `diffusion_type: ddpm` these values are stored as persistent buffers in checkpoints: changing the list length causes checkpoint loading to fail, while changed values are silently overridden by the checkpoint on loading; with Rectified Flow they are always read from the current configuration.
visibilityacoustic
scopetraining
scopetraining, inference
customizabilitynormal
typebool
defaultfalse
- + - + +
visibilityacoustic
scopeinference
scopenn, training, inference
customizabilitynot recommended
typeList[float]
typelist[float]
default[-12]
constraintsMust contain either one value or audio_num_mel_bins values.
### spec_max -Maximum mel spectrogram value used for normalization to [-1, 1]. Different mel bins can have different maximum values. +Maximum mel-spectrogram value used for normalization to [-1, 1]. Different mel bins can have different maximum values. For buffer persistence behavior in checkpoints, see the note in [spec_min](#spec_min). - + - + +
visibilityacoustic
scopeinference
scopenn, training, inference
customizabilitynot recommended
typeList[float]
typelist[float]
default[0.0]
constraintsMust contain either one value or audio_num_mel_bins values.
### T_start -The starting value of time $t$ in the Rectified Flow ODE which applies on $t \in (T_{start}, 1)$. +The starting value of time $t$ in the Rectified Flow ODE which applies for $t \in (T_{start}, 1)$. Only takes effect when [use_shallow_diffusion](#use_shallow_diffusion) is set to `true`; otherwise it is forced to 0. The [0, 1] range constraint is asserted only when shallow diffusion is enabled. - + +
visibilityacoustic
scopetraining
scopetraining, inference
customizabilityrecommended
typefloat
default0.4
constraintsMust be in the range [0, 1].
### T_start_infer -The starting value of time $t$ in the ODE during shallow Rectified Flow inference. Normally set as same as [T_start](#T_start). +The starting value of time $t$ in the ODE during shallow Rectified Flow inference. Normally set to the same value as [T_start](#t_start); when this key is not set, [T_start](#t_start) is used as the fallback. Only takes effect when [use_shallow_diffusion](#use_shallow_diffusion) is set to `true`; ignored otherwise. @@ -1864,7 +1892,7 @@ The starting value of time $t$ in the ODE during shallow Rectified Flow inferenc - +
visibilityacoustic
customizabilityrecommended
typefloat
default0.4
constraintsShould be no less than T_start.
constraintsShould be no less than T_start. This is not asserted: smaller values silently sample from time steps outside the trained range. Values greater than or equal to 1 are silently treated as 1, i.e., the shallow diffusion source is returned without any actual sampling; values no greater than 0 are silently treated as 0, i.e., full sampling from pure noise.
### task_cls @@ -1872,15 +1900,17 @@ The starting value of time $t$ in the ODE during shallow Rectified Flow inferenc Task trainer class name. - + - + + +
visibilityall
visibilityacoustic, variance
scopetraining
customizabilityreserved
typestr
typestr | None
defaultnull
constraintsThe base configuration may leave this as `null`; the training entry point requires a non-null importable class name.
### tension_logit_max -Maximum tension logit value used for normalization to [-1, 1]. Logit is the reverse function of Sigmoid: +Maximum tension logit value used for normalization to [-1, 1]. Note that with [diffusion_type](#diffusion_type) `'ddpm'`, this value is latched into persistent buffers in checkpoints: modifying it for an existing experiment does not raise errors, but is silently overridden by the checkpoint on loading, so it only takes effect when training from scratch; with Rectified Flow it is always read from the current configuration. Logits are calculated using the inverse of Sigmoid function: $$ f(x) = \ln\frac{x}{1-x} @@ -1888,7 +1918,7 @@ $$ - + @@ -1896,7 +1926,7 @@ $$ ### tension_logit_min -Minimum tension logit value used for normalization to [-1, 1]. Logit is the reverse function of Sigmoid: +Minimum tension logit value used for normalization to [-1, 1]. Note that with [diffusion_type](#diffusion_type) `'ddpm'`, this value is latched into persistent buffers in checkpoints: modifying it for an existing experiment does not raise errors, but is silently overridden by the checkpoint on loading, so it only takes effect when training from scratch; with Rectified Flow it is always read from the current configuration. Logits are calculated using the inverse of Sigmoid function: $$ f(x) = \ln\frac{x}{1-x} @@ -1904,7 +1934,7 @@ $$
visibilityvariance
scopeinference
scopetraining, inference
customizabilityrecommended
typefloat
default10.0
- + @@ -1912,7 +1942,7 @@ $$ ### tension_smooth_width -Length of sinusoidal smoothing convolution kernel (in seconds) on extracted tension curve. +Length of sinusoidal smoothing convolution kernel (in seconds) on the extracted tension curve.
visibilityvariance
scopeinference
scopetraining, inference
customizabilityrecommended
typefloat
default-10.0
@@ -1924,11 +1954,11 @@ Length of sinusoidal smoothing convolution kernel (in seconds) on extracted tens ### time_scale_factor -The scale factor that will be multiplied on the time $t$ of Rectified Flow before embedding into the model. +The scale factor that applied to time $t$ of Rectified Flow before embedding into the model. It is read in both the training loss computation and the inference ODE solver, and is baked into exported ONNX graphs; modifying it does not change parameter shapes or prevent checkpoint loading, but silently changes the behavior of an already trained model. Only takes effect when [diffusion_type](#diffusion_type) is `'reflow'`; with DDPM the time scaling is internally fixed to [timesteps](#timesteps) and this key is ignored.
visibilityacoustic, variance
- + @@ -1936,11 +1966,11 @@ The scale factor that will be multiplied on the time $t$ of Rectified Flow befor ### timesteps -Total number of DDPM steps. +Total number of DDPM steps. Only takes effect when [diffusion_type](#diffusion_type) is `'ddpm'`; ignored with Rectified Flow, whose sampling grid is controlled by [sampling_steps](#sampling_steps) and [T_start_infer](#t_start_infer) instead.
visibilityacoustic, variance
scopenn
scopetraining, inference
customizabilitynot recommended
typefloat
default1000
- + @@ -1954,7 +1984,7 @@ Whether to accept and embed breathiness values into the model. - +
visibilityacoustic, variance
scopenn
scopenn, training, inference
customizabilitynot recommended
typeint
default1000
visibilityacoustic
scopenn, preprocessing, inference
customizabilityrecommended
typeboolean
typebool
defaultfalse
@@ -1966,21 +1996,20 @@ Whether to accept and embed energy values into the model. visibilityacoustic scopenn, preprocessing, inference customizabilityrecommended -typeboolean +typebool defaultfalse ### use_glide_embed -Whether to accept and embed glide types in melody encoder. +Whether to accept and embed glide types in the melody encoder. This option only takes effect when [use_melody_encoder](#use_melody_encoder) is enabled. - + -
visibilityvariance
scopenn, preprocessing, inference
customizabilityrecommended
typeboolean
typebool
defaultfalse
constraintsOnly take affects when melody encoder is enabled.
### use_key_shift_embed @@ -1991,18 +2020,18 @@ Whether to embed key shifting values introduced by random pitch shifting augment visibilityacoustic scopenn, preprocessing, inference customizabilityrecommended -typeboolean +typebool defaultfalse -constraintsMust be true if random pitch shifting is enabled. +constraintsMust be true if random pitch shifting is enabled. ### use_lang_id -Whether to embed the language ID from a multilingual dataset. This option only takes effect for those cross-lingual phonemes in the merged groups. +Whether to embed the language ID from a multilingual dataset. This option only takes effect for those cross-lingual phonemes in the merged groups. Language IDs are always extracted and stored by binarizers regardless of this value, so enabling it after preprocessing does not require re-running binarizers. - + @@ -2010,14 +2039,14 @@ Whether to embed the language ID from a multilingual dataset. This option only t ### use_melody_encoder -Whether to enable melody encoder for the pitch predictor. +Whether to enable the melody encoder for the pitch predictor. This option only takes effect when [predict_pitch](#predict_pitch) is true; otherwise the melody encoder is not built regardless of this value.
visibilityacoustic, variance
scopenn, preprocessing, inference
scopenn, inference
customizabilityrecommended
typebool
defaultfalse
- + - - + +
visibilityvariance
scopenn
scopenn, inference
customizabilityrecommended
typeboolean
defaultfalse
typebool
defaulttrue
### use_mix_ln @@ -2026,7 +2055,7 @@ Whether to use Mixed LayerNorm with speaker-conditioned mixup in the acoustic en - + @@ -2034,25 +2063,25 @@ Whether to use Mixed LayerNorm with speaker-conditioned mixup in the acoustic en ### use_pos_embed -Whether to use SinusoidalPositionalEmbedding in FastSpeech2 encoder. +Whether to enable positional encoding in FastSpeech2 encoder. When [use_rope](#use_rope) is `false`, this key controls the additive input embedding (`SinusoidalPositionalEmbedding` when `rel_pos` is `false`, or `RelPositionalEncoding` when `rel_pos` is `true`). When `use_rope` is `true`, no additive embedding is created, but RoPE is only created if this key is also `true` — disabling it removes RoPE as well and leaves the encoder with no positional encoding at all. The additive embedding module itself is created based on [use_rope](#use_rope) and [rel_pos](#rel_pos) alone, regardless of this key, so toggling it never changes parameter shapes or the set of saved keys and never prevents checkpoint loading; it only selects whether the positional encoding is actually applied at run time (and, when `use_rope` is `true`, whether RoPE is created and applied in attention), which changes the behavior of both training and inference. Since an already trained model expects its trained positional encoding scheme, modifying it silently produces inconsistent or wrong outputs.
visibilityacoustic
scopenn
scopenn, inference
customizabilitynormal
typebool
defaultfalse
- + - +
visibilityacoustic, variance
scopenn
scopetraining, inference
customizabilitynot recommended
typeboolean
typebool
defaulttrue
### use_rope -Whether to use RoPE (Rotary Positional Encoding) in FastSpeech2 encoder. +Whether to use RoPE (Rotary Positional Encoding) in FastSpeech2 encoder. RoPE is only created when [use_pos_embed](#use_pos_embed) is also `true`; otherwise the encoder gets no positional encoding. When enabled, no positional embedding is added to the encoder input, so [rel_pos](#rel_pos) has no effect. RoPE itself keeps no parameters, and its frequency buffers are recomputed at initialization and never saved in checkpoints; however, enabling RoPE removes and disabling RoPE creates the input positional embedding module. When [rel_pos](#rel_pos) is `true` that module (`RelPositionalEncoding`) owns no parameters or persistent buffers, so toggling this option does not prevent checkpoint loading but silently changes the behavior of an already trained model. When `rel_pos` is `false` that module is a `SinusoidalPositionalEmbedding`, which owns a persistent buffer saved in checkpoints, so toggling this option then changes the set of saved keys and results in failure when loading or resuming from checkpoints. - + - +
visibilityacoustic, variance
scopenn
scopenn, training, inference
customizabilitynot recommended
typeboolean
typebool
defaulttrue
@@ -2062,10 +2091,10 @@ Whether to use shallow diffusion. - + - - + +
visibilityacoustic
scopenn, inference
scopenn, training, inference
customizabilityrecommended
typeboolean
defaultfalse
typebool
defaulttrue
### use_speed_embed @@ -2075,18 +2104,19 @@ Whether to embed speed values introduced by random time stretching augmentation. - + + - +
visibilityacoustic
scopenn, preprocessing, inference
typeboolean
customizabilityrecommended
typebool
defaultfalse
constraintsMust be true if random time stretching is enabled.
constraintsMust be true if random time stretching is enabled.
### use_spk_id -Whether to embed the speaker ID from a multi-speaker dataset. +Whether to embed the speaker ID from a multi-speaker dataset. Speaker IDs are always extracted and stored by binarizers regardless of this value, so enabling it after preprocessing does not require re-running binarizers. - + @@ -2094,14 +2124,14 @@ Whether to embed the speaker ID from a multi-speaker dataset. ### use_stretch_embed -Whether to accept and embed phoneme-level time stretching ratios into the acoustic encoder. The stretch ratio is computed by the `StretchRegulator` module, which measures how much each mel frame is stretched or compressed relative to its corresponding phoneme's average duration. When random time stretching augmentation is enabled, this embedding helps the model condition on the actual stretch applied during data augmentation. +Whether to embed the per-frame relative position within phonemes into the encoder. The value is computed by the `StretchRegulator` module: for each mel frame, its zero-based position within its phoneme is divided by that phoneme's duration, forming a normalized ramp from 0 to 1.
visibilityacoustic, variance
scopenn, preprocessing, inference
scopenn, inference
customizabilityrecommended
typebool
defaultfalse
- + - +
visibilityacoustic, variance
scopenn, preprocessing, inference
scopenn, inference
customizabilitynot recommended
typebool
defaulttrue for acoustic, false for variance
defaulttrue
### use_tension_embed @@ -2112,17 +2142,17 @@ Whether to accept and embed tension values into the model. visibilityacoustic scopenn, preprocessing, inference customizabilityrecommended -typeboolean +typebool defaultfalse ### use_variance_scaling -Whether to apply log-domain scaling to duration and MIDI embeddings to compress their dynamic range. When enabled, phoneme duration values are embedded in log space via `log(1 + dur)`, and MIDI note numbers are normalized by 1/128. This scaling helps the model handle the wide range of duration and MIDI values more stably during training and inference. +Whether to normalize variance-related inputs to compress their dynamic range before embedding. When enabled: phoneme durations are embedded in log space via `log(1 + dur)` in the acoustic task, and in the variance task only when [predict_dur](#predict_dur) is `false` — in the word mode of the variance task (`predict_dur: true`), word durations are embedded linearly without log scaling; note durations in the melody encoder are embedded via `log(1 + dur)`; MIDI note numbers are divided by 128; pitch is divided by 12; in the pitch prediction branch, the division differs by mode: when the melody encoder is disabled, base pitch is divided by 128 before embedding, but when the melody encoder is enabled (see [use_melody_encoder](#use_melody_encoder), which defaults to `true`), base pitch is not embedded at all and delta pitch (pitch minus base pitch) divided by 12 is embedded instead; energy, breathiness and voicing are divided by 96; tension is multiplied by 0.1; key shift is divided by 12. This scaling helps the model handle the wide range of these values more stably during training and inference. It only selects the scaling factors applied inside the model graph and does not change parameter shapes, so modifying it does not prevent checkpoint loading, but silently changes the behavior of an already trained model. - + @@ -2136,16 +2166,16 @@ Whether to accept and embed voicing values into the model. - +
visibilityacoustic, variance
scopenn, inference
scopetraining, inference
customizabilitynot recommended
typebool
defaulttrue
visibilityacoustic
scopenn, preprocessing, inference
customizabilityrecommended
typeboolean
typebool
defaultfalse
### val_check_interval -Interval (in number of training steps) between validation checks. +Interval (in number of optimizer updates, i.e. global steps) between validation checks. The value actually passed to the trainer is multiplied by [accumulate_grad_batches](#accumulate_grad_batches), so when gradient accumulation is larger than 1, proportionally more micro-batches run between validation checks. - + @@ -2166,10 +2196,10 @@ Whether to load and use the vocoder to generate audio during validation. Validat ### variances_prediction_args -Arguments for prediction of variance parameters other than pitch, like energy, breathiness, etc. +Arguments for predicting variance parameters other than pitch, such as energy, breathiness, etc.
visibilityall
visibilityacoustic, variance
scopetraining
customizabilityrecommended
typeint
- +
typedict
typedict[str, Any]
### variances_prediction_args.backbone_args @@ -2177,7 +2207,7 @@ Arguments for prediction of variance parameters other than pitch, like energy, b Equivalent to [backbone_args](#backbone_args) but only for the multi-variance predictor. - +
visibilityvariance
typedict[str, Any]
### variances_prediction_args.backbone_type @@ -2186,12 +2216,16 @@ Equivalent to [backbone_type](#backbone_type) but only for the multi-variance pr + + + +
visibilityvariance
scopenn
customizabilitynormal
typestr
defaultlynxnet2
constraintsChoose from 'wavenet', 'lynxnet', 'lynxnet2'.
### variances_prediction_args.total_repeat_bins -Total number of repeating bins in the multi-variance predictor. Repeating bins are distributed evenly to each variance parameter. +Total number of repeating bins in the multi-variance predictor. Repeating bins are distributed evenly among the variance parameters. @@ -2199,15 +2233,16 @@ Total number of repeating bins in the multi-variance predictor. Repeating bins a +
visibilityvariance
customizabilityrecommended
typeint
default72
constraintsMust be divisible by the number of predicted variance parameters.
### vocoder -The vocoder class name. +Vocoder class name. - + @@ -2215,35 +2250,35 @@ The vocoder class name. ### vocoder_ckpt -Path of the vocoder model. +Checkpoint or model path of NN-based vocoder.
visibilityacoustic
scopepreprocessing, training, inference
scopetraining, inference
customizabilitynormal
typestr
defaultNsfHifiGAN
- + - +
visibilityacoustic
scopepreprocessing, training, inference
scopetraining, inference
customizabilitynormal
typestr
defaultcheckpoints/nsf_hifigan/model
defaultcheckpoints/pc_nsf_hifigan_44.1k_hop512_128bin_2025.02/model.ckpt
### voicing_db_max -Maximum voicing value in dB used for normalization to [-1, 1]. +Maximum voicing value in dB used for normalization to [-1, 1]. Note that with [diffusion_type](#diffusion_type) `'ddpm'`, this value is latched into persistent buffers in checkpoints: modifying it for an existing experiment does not raise errors, but is silently overridden by the checkpoint on loading, so it only takes effect when training from scratch; with Rectified Flow it is always read from the current configuration. - + - +
visibilityvariance
scopeinference
scopetraining, inference
customizabilityrecommended
typefloat
default-20.0
default-12.0
### voicing_db_min -Minimum voicing value in dB used for normalization to [-1, 1]. +Minimum voicing value in dB used for normalization to [-1, 1]. Note that with [diffusion_type](#diffusion_type) `'ddpm'`, this value is latched into persistent buffers in checkpoints: modifying it for an existing experiment does not raise errors, but is silently overridden by the checkpoint on loading, so it only takes effect when training from scratch; with Rectified Flow it is always read from the current configuration. - - + + @@ -2251,7 +2286,7 @@ Minimum voicing value in dB used for normalization to [-1, 1]. ### voicing_smooth_width -Length of sinusoidal smoothing convolution kernel (in seconds) on extracted voicing curve. +Length of sinusoidal smoothing convolution kernel (in seconds) on the extracted voicing curve.
visibilityacoustic, variance
scopeinference
visibilityvariance
scopetraining, inference
customizabilityrecommended
typefloat
default-96.0
@@ -2267,7 +2302,7 @@ Window size for mel or feature extraction.
visibilityacoustic, variance
- + diff --git a/docs/GettingStarted.md b/docs/GettingStarted.md index 1f7ef8f94..24538f801 100644 --- a/docs/GettingStarted.md +++ b/docs/GettingStarted.md @@ -6,7 +6,7 @@ DiffSinger requires Python 3.10 or later. We strongly recommend you create a virtual environment via Conda, venv or uv before installing dependencies. -1. Install The latest PyTorch following the [official instructions](https://pytorch.org/get-started/locally/) according to your OS and hardware. We recommend using the latest stable release that is >= 2.4.0. +1. Install the latest PyTorch following the [official instructions](https://pytorch.org/get-started/locally/) according to your OS and hardware. We recommend using a stable release >= 2.4.0. 2. Install other dependencies via the following command: @@ -20,13 +20,14 @@ Before you proceed, it is necessary to understand some fundamental concepts in t ## Configuration -Every model needs a configuration file to run preprocessing, training, inference and deployment. Templates of configurations files are in [configs/templates](../configs/templates). Please **copy** the templates to your own data directory before you edit them. +Every model needs a configuration file to run preprocessing, training, inference and deployment. Templates of configuration files are in [configs/templates](../configs/templates). Please **copy** the templates to your own data directory before you edit them. Before you continue, it is highly recommended to read through [Best Practices](BestPractices.md), which is a more detailed tutorial on how to configure your experiments. For more details about configurable parameters, see [Configuration Schemas](ConfigurationSchemas.md). -> Tips: to see which parameters are required or recommended to be edited, you can search by _customizability_ in the configuration schemas. +> [!TIP] +> To see which parameters are required or recommended to be edited, you can search by _customizability_ in the configuration schemas. ## Preprocessing @@ -38,7 +39,7 @@ Assume that you have a configuration file called `my_config.yaml`. Run: python scripts/binarize.py --config my_config.yaml ``` -Preprocessing can be accelerated through multiprocessing. See [binarization_args.num_workers](ConfigurationSchemas.md#binarization_args.num_workers) for more explanations. +Preprocessing can be accelerated through multiprocessing. See [binarization_args.num_workers](ConfigurationSchemas.md#binarization_argsnum_workers) for more explanations. ## Training @@ -54,15 +55,14 @@ For more suggestions related to training performance, see [performance tuning](B ### TensorBoard -Run the following command to start the TensorBoard: +Run the following command to start TensorBoard: ```bash tensorboard --logdir checkpoints/ ``` -> NOTICE -> -> If you are training a model with multiple GPUs (DDP), please add `--reload_multifile=true` option when launching TensorBoard, otherwise it may not update properly. +> [!NOTE] +> If you are training a model with multiple GPUs (DDP), please add the `--reload_multifile=true` option when launching TensorBoard, otherwise it may not update properly. ## Inference @@ -136,7 +136,7 @@ To export an NSF-HiFiGAN vocoder checkpoint, run: python scripts/export.py nsf-hifigan --config CONFIG --ckpt CKPT ``` -where `CONFIG` is a configuration file that has configured the same mel parameters as the vocoder (can be configs/acoustic.yaml for most cases) and `CKPT` is the path of the checkpoint to be exported. +where `CONFIG` is a configuration file that configures the same mel parameters as the vocoder (can be `configs/acoustic.yaml` for most cases) and `CKPT` is the path of the checkpoint to be exported. `--ckpt` is optional; if it is omitted, the checkpoint path is read from `vocoder_ckpt` in `CONFIG`. For more configurable options, run @@ -148,5 +148,5 @@ python scripts/export.py nsf-hifigan --help There are other useful CLI tools in the [scripts/](../scripts) directory not mentioned above: -- drop_spk.py - delete speaker embeddings from checkpoints (for data security reasons when distributing models) -- vocoder.py - bypass the acoustic model and only run the vocoder on given mel-spectrograms +- `drop_spk.py` - drop speaker embeddings from a checkpoint and save as a new one, refilling the dropped embeddings with zeros (by default) or other values (for data security reasons when distributing models) +- `vocode.py` - bypass the acoustic model and only run the vocoder on given mel-spectrograms diff --git a/preprocessing/variance_binarizer.py b/preprocessing/variance_binarizer.py index 5cc14f3c8..a20160ca6 100644 --- a/preprocessing/variance_binarizer.py +++ b/preprocessing/variance_binarizer.py @@ -157,7 +157,7 @@ def require(attr, optional=False): if self.phoneme_dictionary.is_cross_lingual(p if '/' in p else f'{lang}/{p}') else 0 ) - for p in utterance_label['ph_seq'].split() + for p in require('ph_seq').split() ], 'ph_seq': self.phoneme_dictionary.encode(require('ph_seq'), lang=lang), 'ph_dur': [float(x) for x in require('ph_dur').split()], From 1c18c59ab0e371ee96b967bd83a7c3e160dda2f3 Mon Sep 17 00:00:00 2001 From: yxlllc <33565655+yxlllc@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:27:40 +0800 Subject: [PATCH 12/23] probe and cap max_frames for the batch sampler (#327) --- basics/base_task.py | 3 ++- utils/training_utils.py | 34 +++++++++++++++++++++++++++++++--- 2 files changed, 33 insertions(+), 4 deletions(-) diff --git a/basics/base_task.py b/basics/base_task.py index 656893d96..f4cde5e3e 100644 --- a/basics/base_task.py +++ b/basics/base_task.py @@ -337,7 +337,8 @@ def train_dataloader(self): size_reversed=True, required_batch_count_multiple=hparams['accumulate_grad_batches'], shuffle_sample=True, - shuffle_batch=True + shuffle_batch=True, + probe_and_cap_max_frames=True ) return torch.utils.data.DataLoader( self.train_dataset, diff --git a/utils/training_utils.py b/utils/training_utils.py index e906f7721..909b1f457 100644 --- a/utils/training_utils.py +++ b/utils/training_utils.py @@ -78,7 +78,8 @@ def __init__(self, dataset, max_batch_frames, max_batch_size, sub_indices=None, num_replicas=None, rank=None, required_batch_count_multiple=1, batch_by_size=True, sort_by_similar_size=True, size_reversed=False, shuffle_sample=False, shuffle_batch=False, - disallow_empty_batch=True, pad_batch_assignment=True, seed=0, drop_last=False) -> None: + disallow_empty_batch=True, pad_batch_assignment=True, seed=0, drop_last=False, + probe_and_cap_max_frames=False) -> None: if rank >= num_replicas or rank < 0: raise ValueError( f"Invalid rank {rank}, rank should be in the interval [0, {num_replicas - 1}]") @@ -98,6 +99,11 @@ def __init__(self, dataset, max_batch_frames, max_batch_size, sub_indices=None, self.pad_batch_assignment = pad_batch_assignment self.seed = seed self.drop_last = drop_last + # When enabled, the first epoch is a memory probe: its batches are + # served in strictly decreasing padded-frames order, and the measured + # maximum becomes the batching cap for all later epochs. + self.probe_and_cap_max_frames = probe_and_cap_max_frames + self.measured_max_batch_frames = None self.epoch = 0 self.batches = None self.formed = None @@ -105,7 +111,7 @@ def __init__(self, dataset, max_batch_frames, max_batch_size, sub_indices=None, def __form_batches(self): if self.formed == self.epoch + self.seed: return - rng = np.random.default_rng() + rng = np.random.default_rng(self.epoch + self.seed) # Create indices if self.shuffle_sample: if self.sub_indices is not None: @@ -127,11 +133,33 @@ def __form_batches(self): # Batching if self.batch_by_size: + # After the probe epoch, cap by the measured maximum so later + # groupings can never exceed the proven memory footprint. + max_frames_cap = ( + self.measured_max_batch_frames + if self.probe_and_cap_max_frames and self.measured_max_batch_frames is not None + else self.max_batch_frames + ) batches = utils.batch_by_size( indices, self.dataset.num_frames, - max_batch_frames=self.max_batch_frames, + max_batch_frames=max_frames_cap, max_batch_size=self.max_batch_size ) + if self.probe_and_cap_max_frames and self.measured_max_batch_frames is None: + # Probe epoch: sort batches by padded frames (largest first), + # then record the measured maximum for later epochs. + padded_frames = [ + len(b) * max((self.dataset.num_frames(i) for i in b), default=0) + for b in batches + ] + self.measured_max_batch_frames = ( + max(padded_frames) if padded_frames else self.max_batch_frames + ) + batches = [batches[i] for i in sorted( + range(len(batches)), key=lambda i: -padded_frames[i])] + rank_zero_info( + 'DsBatchSampler: auto-capped batch frames = %d (config max %d).', + self.measured_max_batch_frames, self.max_batch_frames) else: batches = [indices[i:i + self.max_batch_size] for i in range(0, len(indices), self.max_batch_size)] if len(batches) < self.num_replicas and self.disallow_empty_batch: From 336cf01b57f2ad44c6b37a79cf33993043291759 Mon Sep 17 00:00:00 2001 From: yxlllc <33565655+yxlllc@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:32:45 +0800 Subject: [PATCH 13/23] enable mixed-precision for layernorm (#328) --- modules/aux_decoder/convnext.py | 4 ++-- modules/backbones/lynxnet.py | 5 +++-- modules/backbones/lynxnet2.py | 5 +++-- modules/commons/common_layers.py | 16 ++++++++++++++++ 4 files changed, 24 insertions(+), 6 deletions(-) diff --git a/modules/aux_decoder/convnext.py b/modules/aux_decoder/convnext.py index ad3fa1e2f..e1a22e103 100644 --- a/modules/aux_decoder/convnext.py +++ b/modules/aux_decoder/convnext.py @@ -3,7 +3,7 @@ import torch import torch.nn as nn -from modules.commons.common_layers import AdamWConv1d +from modules.commons.common_layers import AdamWConv1d, MixedPrecisionLayerNorm class ConvNeXtBlock(nn.Module): @@ -26,7 +26,7 @@ def __init__( super().__init__() self.dwconv = nn.Conv1d(dim, dim, kernel_size=7, padding=3, groups=dim) # depthwise conv - self.norm = nn.LayerNorm(dim, eps=1e-6) + self.norm = MixedPrecisionLayerNorm(dim, eps=1e-6) self.pwconv1 = nn.Linear(dim, intermediate_dim) # pointwise/1x1 convs, implemented with linear layers self.act = nn.GELU() self.pwconv2 = nn.Linear(intermediate_dim, dim) diff --git a/modules/backbones/lynxnet.py b/modules/backbones/lynxnet.py index 9529d1efe..85405744a 100644 --- a/modules/backbones/lynxnet.py +++ b/modules/backbones/lynxnet.py @@ -7,6 +7,7 @@ from modules.commons.common_layers import SinusoidalPosEmb, SwiGLU, Transpose, AdamWConv1d from modules.commons.common_layers import KaimingNormalConv1d as Conv1d +from modules.commons.common_layers import MixedPrecisionLayerNorm as LayerNorm from utils.hparams import hparams @@ -34,7 +35,7 @@ def __init__(self, dim, expansion_factor, kernel_size=31, activation='PReLU', dr else: _dropout = nn.Identity() self.net = nn.Sequential( - nn.LayerNorm(dim), + LayerNorm(dim), Transpose((1, 2)), nn.Conv1d(dim, inner_dim * 2, 1), SwiGLU(dim=1), @@ -104,7 +105,7 @@ def __init__(self, in_dims, n_feats, *, num_layers=6, num_channels=512, expansio for _ in range(num_layers) ] ) - self.norm = nn.LayerNorm(num_channels) + self.norm = LayerNorm(num_channels) self.output_projection = AdamWConv1d(num_channels, in_dims * n_feats, kernel_size=1) self.strong_cond = strong_cond nn.init.zeros_(self.output_projection.weight) diff --git a/modules/backbones/lynxnet2.py b/modules/backbones/lynxnet2.py index e2c717462..4fae179db 100644 --- a/modules/backbones/lynxnet2.py +++ b/modules/backbones/lynxnet2.py @@ -5,6 +5,7 @@ from modules.commons.common_layers import ( SinusoidalPosEmb, SwiGLU, ATanGLU, SoftSignGLU, Transpose, AdamWLinear ) +from modules.commons.common_layers import MixedPrecisionLayerNorm as LayerNorm from utils.hparams import hparams @@ -25,7 +26,7 @@ def __init__(self, dim, expansion_factor, kernel_size=31, dropout=0., glu_type=' else: _dropout = nn.Identity() self.net = nn.Sequential( - nn.LayerNorm(dim), + LayerNorm(dim), Transpose((1, 2)), nn.Conv1d(dim, dim, kernel_size=kernel_size, padding=kernel_size // 2, groups=dim), Transpose((1, 2)), @@ -75,7 +76,7 @@ def __init__(self, in_dims, n_feats, *, num_layers=6, num_channels=512, expansio for _ in range(num_layers) ] ) - self.norm = nn.LayerNorm(num_channels) + self.norm = LayerNorm(num_channels) self.output_projection = AdamWLinear(num_channels, in_dims * n_feats) nn.init.kaiming_normal_(self.input_projection.weight) nn.init.kaiming_normal_(self.conditioner_projection.weight) diff --git a/modules/commons/common_layers.py b/modules/commons/common_layers.py index 10852e20e..467f6e4ec 100644 --- a/modules/commons/common_layers.py +++ b/modules/commons/common_layers.py @@ -287,6 +287,22 @@ def forward( return mixed_gammas * x + mixed_betas +class MixedPrecisionLayerNorm(nn.LayerNorm): + """LayerNorm that keeps fp16/bf16 activations under AMP autocast""" + + def forward(self, x: torch.Tensor) -> torch.Tensor: + with torch.autocast(device_type=x.device.type, enabled=False): + weight = self.weight + bias = self.bias + if weight is not None and weight.dtype != x.dtype: + weight = weight.to(x.dtype) + if bias is not None and bias.dtype != x.dtype: + bias = bias.to(x.dtype) + return F.layer_norm( + x, self.normalized_shape, weight, bias, self.eps + ) + + class TransformerFFNLayer(nn.Module): def __init__(self, hidden_size, filter_size, kernel_size=1, dropout=0., act='gelu'): super().__init__() From 8935f61780d4483fa36a673d27c33eb36de75d03 Mon Sep 17 00:00:00 2001 From: Kakaru <97896816+KakaruHayate@users.noreply.github.com> Date: Tue, 15 Sep 2026 11:53:04 +0800 Subject: [PATCH 14/23] Rolling back the training precision of the last layernorm in LynxNet (#331) --- modules/backbones/lynxnet.py | 2 +- modules/backbones/lynxnet2.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/modules/backbones/lynxnet.py b/modules/backbones/lynxnet.py index 85405744a..ea7efbce2 100644 --- a/modules/backbones/lynxnet.py +++ b/modules/backbones/lynxnet.py @@ -105,7 +105,7 @@ def __init__(self, in_dims, n_feats, *, num_layers=6, num_channels=512, expansio for _ in range(num_layers) ] ) - self.norm = LayerNorm(num_channels) + self.norm = nn.LayerNorm(num_channels) self.output_projection = AdamWConv1d(num_channels, in_dims * n_feats, kernel_size=1) self.strong_cond = strong_cond nn.init.zeros_(self.output_projection.weight) diff --git a/modules/backbones/lynxnet2.py b/modules/backbones/lynxnet2.py index 4fae179db..613cb7ff0 100644 --- a/modules/backbones/lynxnet2.py +++ b/modules/backbones/lynxnet2.py @@ -76,7 +76,7 @@ def __init__(self, in_dims, n_feats, *, num_layers=6, num_channels=512, expansio for _ in range(num_layers) ] ) - self.norm = LayerNorm(num_channels) + self.norm = nn.LayerNorm(num_channels) self.output_projection = AdamWLinear(num_channels, in_dims * n_feats) nn.init.kaiming_normal_(self.input_projection.weight) nn.init.kaiming_normal_(self.conditioner_projection.weight) From 86518e87b1da59f9ce317e56d182ba4624e7a1a4 Mon Sep 17 00:00:00 2001 From: yxlllc <33565655+yxlllc@users.noreply.github.com> Date: Tue, 15 Sep 2026 12:33:48 +0800 Subject: [PATCH 15/23] Support dual-timestep reflow (#323) * support dual-timestep reflow * fix: preserve timestep rank in sinusoidal embedding (#316) --------- Co-authored-by: Kakaru <97896816+KakaruHayate@users.noreply.github.com> --- configs/acoustic.yaml | 1 + configs/templates/config_acoustic.yaml | 1 + configs/templates/config_variance.yaml | 1 + configs/variance.yaml | 1 + modules/backbones/lynxnet.py | 16 +++++++++++++--- modules/backbones/lynxnet2.py | 15 +++++++++++++-- modules/backbones/wavenet.py | 22 +++++++++++++++++----- modules/commons/common_layers.py | 2 +- modules/core/reflow.py | 25 ++++++++++++++++++------- modules/losses/reflow_loss.py | 4 ++-- 10 files changed, 68 insertions(+), 20 deletions(-) diff --git a/configs/acoustic.yaml b/configs/acoustic.yaml index 6d0ff747b..e49eb1142 100644 --- a/configs/acoustic.yaml +++ b/configs/acoustic.yaml @@ -63,6 +63,7 @@ use_key_shift_embed: false use_speed_embed: false diffusion_type: reflow +use_dual_timestep: true time_scale_factor: 1000 timesteps: 1000 max_beta: 0.02 diff --git a/configs/templates/config_acoustic.yaml b/configs/templates/config_acoustic.yaml index 1040dca47..68dabb3a7 100644 --- a/configs/templates/config_acoustic.yaml +++ b/configs/templates/config_acoustic.yaml @@ -72,6 +72,7 @@ augmentation_args: # diffusion and shallow diffusion diffusion_type: reflow +use_dual_timestep: true enc_ffn_kernel_size: 3 use_rope: true rope_interleaved: false diff --git a/configs/templates/config_variance.yaml b/configs/templates/config_variance.yaml index d3acd51a6..7c5af0ea2 100644 --- a/configs/templates/config_variance.yaml +++ b/configs/templates/config_variance.yaml @@ -91,6 +91,7 @@ glide_types: [up, down] glide_embed_scale: 11.313708498984760 # sqrt(128) diffusion_type: reflow +use_dual_timestep: true pitch_prediction_args: pitd_norm_min: -8.0 diff --git a/configs/variance.yaml b/configs/variance.yaml index bac0f8154..2547e47a6 100644 --- a/configs/variance.yaml +++ b/configs/variance.yaml @@ -109,6 +109,7 @@ lambda_pitch_loss: 1.0 lambda_var_loss: 1.0 diffusion_type: reflow # ddpm +use_dual_timestep: true time_scale_factor: 1000 schedule_type: 'linear' K_step: 1000 diff --git a/modules/backbones/lynxnet.py b/modules/backbones/lynxnet.py index ea7efbce2..d0e40e306 100644 --- a/modules/backbones/lynxnet.py +++ b/modules/backbones/lynxnet.py @@ -2,6 +2,7 @@ # https://github.com/CNChTu/Diffusion-SVC/blob/v2.0_dev/diffusion/naive_v2/model_conformer_naive.py # https://github.com/CNChTu/Diffusion-SVC/blob/v2.0_dev/diffusion/naive_v2/naive_v2_diff.py +import torch import torch.nn as nn import torch.nn.functional as F @@ -110,7 +111,7 @@ def __init__(self, in_dims, n_feats, *, num_layers=6, num_channels=512, expansio self.strong_cond = strong_cond nn.init.zeros_(self.output_projection.weight) - def forward(self, spec, diffusion_step, cond): + def forward(self, spec, diffusion_step, cond, diffusion_step_2=None, mask=None): """ :param spec: [B, F, M, T] :param diffusion_step: [B, 1] @@ -127,10 +128,19 @@ def forward(self, spec, diffusion_step, cond): if not self.strong_cond: x = F.gelu(x) - diffusion_step = self.diffusion_embedding(diffusion_step).unsqueeze(-1) + if mask is not None: + step = torch.cat((diffusion_step, diffusion_step_2), dim=0) + step = self.diffusion_embedding(step) + step, step_2 = torch.split(step, x.shape[0], dim=0) #[B, 1, C] + mask = mask.to(x).unsqueeze(-1) # [B, T, 1] + step = step + (step_2 - step) * mask + else: + step = self.diffusion_embedding(diffusion_step) + if step.dim() == 2: + step = step.unsqueeze(1) for layer in self.residual_layers: - x = layer(x, cond, diffusion_step, front_cond_inject=self.strong_cond) + x = layer(x, cond, step.transpose(1, 2), front_cond_inject=self.strong_cond) # post-norm x = self.norm(x.transpose(1, 2)).transpose(1, 2) diff --git a/modules/backbones/lynxnet2.py b/modules/backbones/lynxnet2.py index 613cb7ff0..6d87eebeb 100644 --- a/modules/backbones/lynxnet2.py +++ b/modules/backbones/lynxnet2.py @@ -82,7 +82,7 @@ def __init__(self, in_dims, n_feats, *, num_layers=6, num_channels=512, expansio nn.init.kaiming_normal_(self.conditioner_projection.weight) nn.init.zeros_(self.output_projection.weight) - def forward(self, spec, diffusion_step, cond): + def forward(self, spec, diffusion_step, cond, diffusion_step_2=None, mask=None): """ :param spec: [B, F, M, T] :param diffusion_step: [B, 1] @@ -100,7 +100,18 @@ def forward(self, spec, diffusion_step, cond): x = x + self.conditioner_projection(cond).transpose(1, 2) else: x = x + self.conditioner_projection(cond.transpose(1, 2)) - x = x + self.diffusion_embedding(diffusion_step).unsqueeze(1) + + if mask is not None: + step = torch.cat((diffusion_step, diffusion_step_2), dim=0) + step = self.diffusion_embedding(step) + step, step_2 = torch.split(step, x.shape[0], dim=0) #[B, 1, C] + mask = mask.to(x).unsqueeze(-1) # [B, T, 1] + x = x + step + (step_2 - step) * mask + else: + step = self.diffusion_embedding(diffusion_step) + if step.dim() == 2: + step = step.unsqueeze(1) + x = x + step for layer in self.residual_layers: x = layer(x) diff --git a/modules/backbones/wavenet.py b/modules/backbones/wavenet.py index 77ccc6430..b31f341ad 100644 --- a/modules/backbones/wavenet.py +++ b/modules/backbones/wavenet.py @@ -26,7 +26,7 @@ def __init__(self, encoder_hidden, residual_channels, dilation): self.output_projection = nn.Conv1d(residual_channels, 2 * residual_channels, 1) def forward(self, x, conditioner, diffusion_step): - diffusion_step = self.diffusion_projection(diffusion_step).unsqueeze(-1) + diffusion_step = self.diffusion_projection(diffusion_step).transpose(1, 2) conditioner = self.conditioner_projection(conditioner) y = x + diffusion_step @@ -67,7 +67,7 @@ def __init__(self, in_dims, n_feats, *, num_layers=20, num_channels=256, dilatio self.output_projection = AdamWConv1d(num_channels, in_dims * n_feats, 1) nn.init.zeros_(self.output_projection.weight) - def forward(self, spec, diffusion_step, cond): + def forward(self, spec, diffusion_step, cond, diffusion_step_2=None, mask=None): """ :param spec: [B, F, M, T] :param diffusion_step: [B, 1] @@ -84,11 +84,23 @@ def forward(self, spec, diffusion_step, cond): x = self.input_projection(x) # [B, C, T] x = F.relu(x) - diffusion_step = self.diffusion_embedding(diffusion_step) - diffusion_step = self.mlp(diffusion_step) + + if mask is not None: + step = torch.cat((diffusion_step, diffusion_step_2), dim=0) + step = self.diffusion_embedding(step) + step = self.mlp(step) + step, step_2 = torch.split(step, x.shape[0], dim=0) #[B, 1, C] + mask = mask.to(x).unsqueeze(-1) # [B, T, 1] + step = step + (step_2 - step) * mask + else: + step = self.diffusion_embedding(diffusion_step) + step = self.mlp(step) + if step.dim() == 2: + step = step.unsqueeze(1) + skip = [] for layer in self.residual_layers: - x, skip_connection = layer(x, cond, diffusion_step) + x, skip_connection = layer(x, cond, step) skip.append(skip_connection) x = torch.sum(torch.stack(skip), dim=0) / sqrt(len(self.residual_layers)) diff --git a/modules/commons/common_layers.py b/modules/commons/common_layers.py index 467f6e4ec..4ad555804 100644 --- a/modules/commons/common_layers.py +++ b/modules/commons/common_layers.py @@ -479,6 +479,6 @@ def forward(self, x): half_dim = self.dim // 2 emb = math.log(10000) / (half_dim - 1) emb = torch.exp(torch.arange(half_dim, device=device) * -emb) - emb = x.unsqueeze(-1) * emb.unsqueeze(0) + emb = x.unsqueeze(-1) * emb emb = torch.cat((emb.sin(), emb.cos()), dim=-1) return emb diff --git a/modules/core/reflow.py b/modules/core/reflow.py index c2f9f1ef7..2a7bd8441 100644 --- a/modules/core/reflow.py +++ b/modules/core/reflow.py @@ -18,6 +18,7 @@ def __init__(self, out_dims, num_feats=1, t_start=0., time_scale_factor=1000, self.velocity_fn: nn.Module = build_backbone(out_dims, num_feats, backbone_type, backbone_args) self.out_dims = out_dims self.num_feats = num_feats + self.use_dual_timestep = hparams.get('use_dual_timestep', False) self.use_shallow_diffusion = hparams.get('use_shallow_diffusion', False) if self.use_shallow_diffusion: assert 0. <= t_start <= 1., 'T_start should be in [0, 1].' @@ -33,24 +34,34 @@ def __init__(self, out_dims, num_feats=1, t_start=0., time_scale_factor=1000, self.register_buffer('spec_min', spec_min, persistent=False) self.register_buffer('spec_max', spec_max, persistent=False) - def p_losses(self, x_end, t, cond): + def p_losses(self, x_end, t1, cond, t2=None, mask=None): + t = t1 if mask is None else t1 + (t2 - t1) * mask x_start = torch.randn_like(x_end) - x_t = x_start + t[:, None, None, None] * (x_end - x_start) - v_pred = self.velocity_fn(x_t, t * self.time_scale_factor, cond) + x_t = x_start + t[:, None, None,:] * (x_end - x_start) + s1 = t1 * self.time_scale_factor + s2 = None if t2 is None else t2 * self.time_scale_factor + v_pred = self.velocity_fn(x_t, s1, cond, s2, mask) - return v_pred, x_end - x_start + return v_pred, x_end - x_start, t def forward(self, condition, gt_spec=None, src_spec=None, infer=True): cond = condition.transpose(1, 2) - b, device = condition.shape[0], condition.device + b, _, n_frames = cond.shape + device = condition.device if not infer: # gt_spec: [B, T, M] or [B, F, T, M] spec = self.norm_spec(gt_spec).transpose(-2, -1) # [B, M, T] or [B, F, M, T] if self.num_feats == 1: spec = spec[:, None, :, :] # [B, F=1, M, T] - t = self.t_start + (1.0 - self.t_start) * torch.rand((b,), device=device) - v_pred, v_gt = self.p_losses(spec, t, cond=cond) + t1 = self.t_start + (1.0 - self.t_start) * torch.rand((b, 1), device=device) + if self.use_dual_timestep: + t2 = self.t_start + (1.0 - self.t_start) * torch.rand((b, 1), device=device) + mask = (torch.rand(b, n_frames, device=device) < 0.25).float() + else: + t2 = None + mask = None + v_pred, v_gt, t = self.p_losses(spec, t1, cond=cond, t2=t2, mask=mask) return v_pred, v_gt, t else: # src_spec: [B, T, M] or [B, F, T, M] diff --git a/modules/losses/reflow_loss.py b/modules/losses/reflow_loss.py index 4917dce2e..640062552 100644 --- a/modules/losses/reflow_loss.py +++ b/modules/losses/reflow_loss.py @@ -31,7 +31,7 @@ def get_weights(t): weights = 0.398942 / t / (1 - t) * torch.exp( -0.5 * torch.log(t / (1 - t)) ** 2 ) + eps - return weights[:, None, None, None] + return weights[:, None, None, :] def _forward(self, v_pred, v_gt, t=None): if self.log_norm: @@ -43,7 +43,7 @@ def forward(self, v_pred: Tensor, v_gt: Tensor, t: Tensor, non_padding: Tensor = """ :param v_pred: [B, 1, M, T] :param v_gt: [B, 1, M, T] - :param t: [B,] + :param t: [B, 1] or [B, T] :param non_padding: [B, T, M] """ v_pred, v_gt = self._mask_non_padding(v_pred, v_gt, non_padding) From d942b6304a736c0e339a3c89b6524f47ae2de77a Mon Sep 17 00:00:00 2001 From: yxlllc <33565655+yxlllc@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:02:08 +0800 Subject: [PATCH 16/23] fix energy calculation (#332) --- utils/binarizer_utils.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/utils/binarizer_utils.py b/utils/binarizer_utils.py index 940d14a5d..de11dd03d 100644 --- a/utils/binarizer_utils.py +++ b/utils/binarizer_utils.py @@ -94,7 +94,7 @@ def get_energy_librosa(waveform, length, *, hop_size, win_size, domain='db'): energy = np.pad(energy, (0, length - len(energy))) energy = energy[: length] if domain == 'db': - energy = librosa.amplitude_to_db(energy) + energy = librosa.amplitude_to_db(energy, top_db=None) elif domain == 'amplitude': pass else: @@ -202,7 +202,7 @@ def get_tension_base_harmonic( tension = np.clip(tension, a_min=0, a_max=1) elif domain == 'db': tension = np.clip(tension, a_min=1e-5, a_max=1) - tension = librosa.amplitude_to_db(tension) + tension = librosa.amplitude_to_db(tension, top_db=None) elif domain == 'logit': tension = np.clip(tension, a_min=1e-4, a_max=1 - 1e-4) tension = np.log(tension / (1 - tension)) From bf78948f5b4c829b41b7cc987474757ebe5f6b59 Mon Sep 17 00:00:00 2001 From: yxlllc <33565655+yxlllc@users.noreply.github.com> Date: Fri, 18 Sep 2026 16:15:01 +0800 Subject: [PATCH 17/23] add missed configuration items (#334) * fix energy calculation * arrange configuration items alphabetically * add missed configuration items --- docs/ConfigurationSchemas.md | 139 ++++++++++++++++++++++------------- 1 file changed, 88 insertions(+), 51 deletions(-) diff --git a/docs/ConfigurationSchemas.md b/docs/ConfigurationSchemas.md index 449e9e3c0..dd8a7d446 100644 --- a/docs/ConfigurationSchemas.md +++ b/docs/ConfigurationSchemas.md @@ -819,28 +819,28 @@ Fast Fourier Transform parameter for mel extraction.
visibilityacoustic, variance
scopepreprocessing
scopepreprocessing, inference
customizabilityreserved
typeint
default2048
default2048
-### finetune_enabled +### finetune_ckpt_path -Whether to finetune from a pretrained model. +Path to the pretrained model for finetuning. - - + +
visibilityacoustic, variance
scopetraining
customizabilitynormal
typebool
defaultfalse
typestr | None
defaultnull
-### finetune_ckpt_path +### finetune_enabled -Path to the pretrained model for finetuning. +Whether to finetune from a pretrained model. - - + +
visibilityacoustic, variance
scopetraining
customizabilitynormal
typestr | None
defaultnull
typebool
defaultfalse
### finetune_ignored_params @@ -989,6 +989,32 @@ Hop size or step length (in number of waveform samples) of mel and feature extra default512 +### K_step + +Maximum number of DDPM steps used by shallow diffusion. Only takes effect when [diffusion_type](#diffusion_type) is `'ddpm'` and [use_shallow_diffusion](#use_shallow_diffusion) is set to `true`; with Rectified Flow the shallow starting point is controlled by [T_start](#t_start) instead, and this key is ignored. + + + + + + + + +
visibilityacoustic
scopetraining, inference
customizabilityrecommended
typeint
default400
constraintsMust not be larger than timesteps.
+ +### K_step_infer + +Number of DDPM steps used during shallow diffusion inference. Normally set to the same value as [K_step](#k_step). Only takes effect when [diffusion_type](#diffusion_type) is `'ddpm'` and [use_shallow_diffusion](#use_shallow_diffusion) is set to `true`; with Rectified Flow the shallow starting point is controlled by [T_start_infer](#t_start_infer) instead, and this key is ignored. + + + + + + + + +
visibilityacoustic
scopeinference
customizabilityrecommended
typeint
default400
constraintsShould be no larger than K_step. Values larger than K_step are silently clamped to K_step instead of raising errors.
+ ### lambda_aux_mel_loss Coefficient of aux mel loss when calculating total loss of acoustic model with shallow diffusion. @@ -1037,32 +1063,6 @@ Coefficient of variance loss (all variance parameters other than pitch, like ene default1.0 -### K_step - -Maximum number of DDPM steps used by shallow diffusion. Only takes effect when [diffusion_type](#diffusion_type) is `'ddpm'` and [use_shallow_diffusion](#use_shallow_diffusion) is set to `true`; with Rectified Flow the shallow starting point is controlled by [T_start](#t_start) instead, and this key is ignored. - - - - - - - - -
visibilityacoustic
scopetraining, inference
customizabilityrecommended
typeint
default400
constraintsMust not be larger than timesteps.
- -### K_step_infer - -Number of DDPM steps used during shallow diffusion inference. Normally set to the same value as [K_step](#k_step). Only takes effect when [diffusion_type](#diffusion_type) is `'ddpm'` and [use_shallow_diffusion](#use_shallow_diffusion) is set to `true`; with Rectified Flow the shallow starting point is controlled by [T_start_infer](#t_start_infer) instead, and this key is ignored. - - - - - - - - -
visibilityacoustic
scopeinference
customizabilityrecommended
typeint
default400
constraintsShould be no larger than K_step. Values larger than K_step are silently clamped to K_step instead of raising errors.
- ### log_interval Controls how often training metrics are logged to TensorBoard, measured in global training steps. @@ -1550,29 +1550,29 @@ Determines which device(s) the model should be trained on. defaultauto -### pl_trainer_precision +### pl_trainer_num_nodes -The computation precision of training. +Number of nodes in the training cluster of Lightning trainer. - - - - + + +
visibilityacoustic, variance
scopetraining
customizabilitynormal
typestr | int | None
default16-mixed
constraintsLightning accepts integer precisions `16`, `32`, `64` and string forms such as `'32-true'`, `'bf16-mixed'` and `'16-mixed'`; `null` is passed through to Lightning and falls back to `'32-true'`. See the Trainer — PyTorch Lightning 2.X.X documentation for the version-specific list.
customizabilityreserved
typeint
default1
-### pl_trainer_num_nodes +### pl_trainer_precision -Number of nodes in the training cluster of Lightning trainer. +The computation precision of training. - - - + + + +
visibilityacoustic, variance
scopetraining
customizabilityreserved
typeint
default1
customizabilitynormal
typestr | int | None
default16-mixed
constraintsLightning accepts integer precisions `16`, `32`, `64` and string forms such as `'32-true'`, `'bf16-mixed'` and `'16-mixed'`; `null` is passed through to Lightning and falls back to `'32-true'`. See the Trainer — PyTorch Lightning 2.X.X documentation for the version-specific list.
### pl_trainer_strategy @@ -1691,6 +1691,18 @@ Whether to use the interleaved (alternating) layout for RoPE (Rotary Positional defaultfalse +### rope_theta + +Base used to compute the RoPE (Rotary Positional Encoding) frequencies in encoder self-attention. For attention head dimension $d$, the frequency of dimension pair $i$ is $\theta^{-2i/d}$, where $\theta$ is this value. The frequency buffers are recomputed at initialization and are not saved in checkpoints, so modifying this value does not change parameter shapes or prevent checkpoint loading, but silently changes the behavior of an already trained model. + + + + + + + +
visibilityacoustic, variance
scopetraining, inference
customizabilitynot recommended
typefloat
default10000
+ ### sampler_frame_count_grid The batch sampler applies an algorithm called _sorting by similar length_ when collecting batches. Data samples are first shuffled, and then stably sorted by their approximate lengths, so that samples of similar lengths are grouped together while the order within each group stays random. Assuming this value is set to $L_{grid}$, the approximate length of a data sample with length $L_{real}$ can be calculated through the following expression: @@ -1843,29 +1855,29 @@ Whether to apply the _sorting by similar length_ algorithm described in [sampler defaulttrue -### spec_min +### spec_max -Minimum mel-spectrogram value used for normalization to [-1, 1]. Different mel bins can have different minimum values. Note that with `diffusion_type: ddpm` these values are stored as persistent buffers in checkpoints: changing the list length causes checkpoint loading to fail, while changed values are silently overridden by the checkpoint on loading; with Rectified Flow they are always read from the current configuration. +Maximum mel-spectrogram value used for normalization to [-1, 1]. Different mel bins can have different maximum values. For buffer persistence behavior in checkpoints, see the note in [spec_min](#spec_min). - +
visibilityacoustic
scopenn, training, inference
customizabilitynot recommended
typelist[float]
default[-12]
default[0.0]
constraintsMust contain either one value or audio_num_mel_bins values.
-### spec_max +### spec_min -Maximum mel-spectrogram value used for normalization to [-1, 1]. Different mel bins can have different maximum values. For buffer persistence behavior in checkpoints, see the note in [spec_min](#spec_min). +Minimum mel-spectrogram value used for normalization to [-1, 1]. Different mel bins can have different minimum values. Note that with `diffusion_type: ddpm` these values are stored as persistent buffers in checkpoints: changing the list length causes checkpoint loading to fail, while changed values are silently overridden by the checkpoint on loading; with Rectified Flow they are always read from the current configuration. - +
visibilityacoustic
scopenn, training, inference
customizabilitynot recommended
typelist[float]
default[0.0]
default[-12]
constraintsMust contain either one value or audio_num_mel_bins values.
@@ -1988,6 +2000,18 @@ Whether to accept and embed breathiness values into the model. defaultfalse +### use_dual_timestep + +Whether to sample two independent timesteps per sample when training Rectified Flow. Only takes effect when [diffusion_type](#diffusion_type) is `'reflow'`; ignored with DDPM. + + + + + + + +
visibilityacoustic, variance
scopetraining
customizabilitynormal
typebool
defaulttrue
+ ### use_energy_embed Whether to accept and embed energy values into the model. @@ -2000,6 +2024,19 @@ Whether to accept and embed energy values into the model. defaultfalse +### use_fused_kernels + +Whether to use Triton-fused Linear + SoftSignGLU operations during training in LYNXNet2 backbones, reducing kernel launches and intermediate memory traffic. + + + + + + + + +
visibilityacoustic, variance
scopetraining
customizabilityrecommended
typebool
defaultfalse
constraintsFusion requires a LYNXNet2 backbone with glu_type: softsign_glu, Triton, and a supported CUDA device and activation dtype (float16 or bfloat16).
+ ### use_glide_embed Whether to accept and embed glide types in the melody encoder. This option only takes effect when [use_melody_encoder](#use_melody_encoder) is enabled. From 8333dd615eef8a04dab6c5e0401215f16b6461b1 Mon Sep 17 00:00:00 2001 From: yxlllc <33565655+yxlllc@users.noreply.github.com> Date: Sun, 20 Sep 2026 12:25:55 +0800 Subject: [PATCH 18/23] Share manager across multiprocessing queues (#335) * fix energy calculation * arrange configuration items alphabetically * add missed configuration items * Share manager across multiprocessing queues --- utils/multiprocess_utils.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/utils/multiprocess_utils.py b/utils/multiprocess_utils.py index 236d94a16..ed554b50f 100644 --- a/utils/multiprocess_utils.py +++ b/utils/multiprocess_utils.py @@ -30,7 +30,8 @@ def chunked_multiprocess_run(map_func, args, num_workers, q_max_size=1000): if num_jobs < num_workers: num_workers = num_jobs - queues = [Manager().Queue(maxsize=q_max_size // num_workers) for _ in range(num_workers)] + manager = Manager() + queues = [manager.Queue(maxsize=q_max_size // num_workers) for _ in range(num_workers)] if platform.system().lower() != 'windows': process_creation_func = get_context('spawn').Process else: From 52e025626f1865c74ba44800a33f704c46128196 Mon Sep 17 00:00:00 2001 From: KakaruHayate Date: Sun, 23 Aug 2026 14:11:19 +0800 Subject: [PATCH 19/23] feat(mixln): add config switch to disable speaker-shuffle during training Mixed_LayerNorm mixes affine (beta/gamma) params across shuffled speakers in the batch at training time. Add a mixln_shuffle_speakers flag (default true = original behaviour) that keeps the conditional-affine network structure but skips the speaker shuffle when set to false, so each speaker's params stay consistent within a step (less inter-speaker leakage). The flag is threaded constructor-to-constructor (no new hparams imports in layer modules, per repo convention) and synced into both acoustic configs. --- configs/acoustic.yaml | 2 ++ configs/templates/config_acoustic.yaml | 2 ++ modules/commons/common_layers.py | 15 +++++++++++---- modules/fastspeech/acoustic_encoder.py | 3 ++- modules/fastspeech/tts_modules.py | 11 +++++++---- 5 files changed, 24 insertions(+), 9 deletions(-) diff --git a/configs/acoustic.yaml b/configs/acoustic.yaml index 206585a51..353f6b145 100644 --- a/configs/acoustic.yaml +++ b/configs/acoustic.yaml @@ -58,6 +58,8 @@ use_spk_id: false num_spk: 1 use_mix_ln: false mix_ln_layer: [0, 2] +# If true, Mixed_LayerNorm shuffles/mixes speakers within a batch during training. +mixln_shuffle_speakers: false use_energy_embed: false use_breathiness_embed: false use_voicing_embed: false diff --git a/configs/templates/config_acoustic.yaml b/configs/templates/config_acoustic.yaml index 55a7e0d61..692d68d32 100644 --- a/configs/templates/config_acoustic.yaml +++ b/configs/templates/config_acoustic.yaml @@ -45,6 +45,8 @@ num_spk: 1 use_mix_ln: false mix_ln_layer: [0, 2] +# If true, Mixed_LayerNorm shuffles/mixes speakers within a batch during training. +mixln_shuffle_speakers: false # NOTICE: before enabling variance embeddings, please read the docs at # https://github.com/openvpi/DiffSinger/tree/main/docs/BestPractices.md#choosing-variance-parameters diff --git a/modules/commons/common_layers.py b/modules/commons/common_layers.py index 4ad555804..920718774 100644 --- a/modules/commons/common_layers.py +++ b/modules/commons/common_layers.py @@ -245,11 +245,16 @@ def __init__( condition_channels: int, beta_distribution_concentration: float = 0.2, eps: float = 1e-5, - bias: bool = True + bias: bool = True, + *, + shuffle_speakers: bool = False ): super().__init__() self.channels = channels self.eps = eps + # If false, skip the speaker-shuffle mixture while keeping the + # conditional-affine structure (default). + self.shuffle_speakers = shuffle_speakers self.beta_distribution = torch.distributions.Beta( beta_distribution_concentration, @@ -275,6 +280,8 @@ def forward( if not self.training or x.size(0) == 1: return gammas * x + betas + if not self.shuffle_speakers: + return gammas * x + betas shuffle_indices = torch.randperm(x.size(0), device=x.device) shuffled_betas = betas[shuffle_indices] @@ -412,7 +419,7 @@ def forward(self, x, key_padding_mask=None): class EncSALayer(nn.Module): def __init__(self, c, num_heads, dropout, attention_dropout=0.1, relu_dropout=0.1, kernel_size=9, act='gelu', rotary_embed=None, - layer_idx=None, mix_ln_layer=None + layer_idx=None, mix_ln_layer=None, mixln_shuffle_speakers=False ): super().__init__() self.dropout = dropout @@ -422,7 +429,7 @@ def __init__(self, c, num_heads, dropout, attention_dropout=0.1, and layer_idx in mix_ln_layer ) if self.use_mix_ln: - self.layer_norm1 = Mixed_LayerNorm(c, c) + self.layer_norm1 = Mixed_LayerNorm(c, c, shuffle_speakers=mixln_shuffle_speakers) else: self.layer_norm1 = LayerNorm(c) # Always use the in-house manual attention. With rotary_embed=None this @@ -435,7 +442,7 @@ def __init__(self, c, num_heads, dropout, attention_dropout=0.1, c, num_heads, dropout=attention_dropout, bias=False, rotary_embed=rotary_embed ) if self.use_mix_ln: - self.layer_norm2 = Mixed_LayerNorm(c, c) + self.layer_norm2 = Mixed_LayerNorm(c, c, shuffle_speakers=mixln_shuffle_speakers) else: self.layer_norm2 = LayerNorm(c) self.ffn = TransformerFFNLayer( diff --git a/modules/fastspeech/acoustic_encoder.py b/modules/fastspeech/acoustic_encoder.py index 70a186d61..083b3c708 100644 --- a/modules/fastspeech/acoustic_encoder.py +++ b/modules/fastspeech/acoustic_encoder.py @@ -59,7 +59,8 @@ def __init__(self, vocab_size): use_pos_embed=hparams['use_pos_embed'], rel_pos=hparams.get('rel_pos', False), use_rope=hparams.get('use_rope', False), rope_interleaved=hparams.get('rope_interleaved', True), rope_theta=hparams.get('rope_theta', 10000), - mix_ln_layer=self.mix_ln_layer + mix_ln_layer=self.mix_ln_layer, + mixln_shuffle_speakers=hparams.get('mixln_shuffle_speakers', False), ) self.pitch_embed = AdamWLinear(1, hparams['hidden_size']) diff --git a/modules/fastspeech/tts_modules.py b/modules/fastspeech/tts_modules.py index 2b1549956..fd52349e2 100644 --- a/modules/fastspeech/tts_modules.py +++ b/modules/fastspeech/tts_modules.py @@ -13,14 +13,15 @@ class TransformerEncoderLayer(nn.Module): def __init__(self, hidden_size, dropout, kernel_size=None, act='gelu', num_heads=2, rotary_embed=None, - layer_idx=None, mix_ln_layer=None): + layer_idx=None, mix_ln_layer=None, mixln_shuffle_speakers=False): super().__init__() self.op = EncSALayer( hidden_size, num_heads, dropout=dropout, attention_dropout=0.0, relu_dropout=dropout, kernel_size=kernel_size, act=act, rotary_embed=rotary_embed, - layer_idx=layer_idx, mix_ln_layer=mix_ln_layer + layer_idx=layer_idx, mix_ln_layer=mix_ln_layer, + mixln_shuffle_speakers=mixln_shuffle_speakers ) def forward(self, x, **kwargs): @@ -373,7 +374,8 @@ def __init__( self, hidden_size, num_layers, ffn_kernel_size=9, ffn_act='gelu', dropout=None, num_heads=2, use_pos_embed=True, rel_pos=True, - use_rope=False, rope_interleaved=True, rope_theta=10000, mix_ln_layer=None + use_rope=False, rope_interleaved=True, rope_theta=10000, mix_ln_layer=None, + mixln_shuffle_speakers=False ): super().__init__() self.num_layers = num_layers @@ -396,7 +398,8 @@ def __init__( self.hidden_size, self.dropout, kernel_size=ffn_kernel_size, act=ffn_act, num_heads=num_heads, rotary_embed=rotary_embed, - layer_idx=i, mix_ln_layer=mix_ln_layer + layer_idx=i, mix_ln_layer=mix_ln_layer, + mixln_shuffle_speakers=mixln_shuffle_speakers ) for i in range(self.num_layers) ]) From be6a9f186a58f1209fde9ddaa8bda9355d652990 Mon Sep 17 00:00:00 2001 From: KakaruHayate Date: Sun, 23 Aug 2026 14:13:21 +0800 Subject: [PATCH 20/23] chore(configs): drop redundant comment on mixln_shuffle_speakers --- configs/acoustic.yaml | 1 - configs/templates/config_acoustic.yaml | 1 - 2 files changed, 2 deletions(-) diff --git a/configs/acoustic.yaml b/configs/acoustic.yaml index 353f6b145..9bf8330af 100644 --- a/configs/acoustic.yaml +++ b/configs/acoustic.yaml @@ -58,7 +58,6 @@ use_spk_id: false num_spk: 1 use_mix_ln: false mix_ln_layer: [0, 2] -# If true, Mixed_LayerNorm shuffles/mixes speakers within a batch during training. mixln_shuffle_speakers: false use_energy_embed: false use_breathiness_embed: false diff --git a/configs/templates/config_acoustic.yaml b/configs/templates/config_acoustic.yaml index 692d68d32..fab631fff 100644 --- a/configs/templates/config_acoustic.yaml +++ b/configs/templates/config_acoustic.yaml @@ -45,7 +45,6 @@ num_spk: 1 use_mix_ln: false mix_ln_layer: [0, 2] -# If true, Mixed_LayerNorm shuffles/mixes speakers within a batch during training. mixln_shuffle_speakers: false # NOTICE: before enabling variance embeddings, please read the docs at From d9f233706470b60b30342f9fa24d16cf688735d4 Mon Sep 17 00:00:00 2001 From: KakaruHayate Date: Sun, 23 Aug 2026 14:14:19 +0800 Subject: [PATCH 21/23] chore(common_layers): drop redundant shuffle_speakers comment --- modules/commons/common_layers.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/modules/commons/common_layers.py b/modules/commons/common_layers.py index 920718774..c31d9d42b 100644 --- a/modules/commons/common_layers.py +++ b/modules/commons/common_layers.py @@ -252,8 +252,6 @@ def __init__( super().__init__() self.channels = channels self.eps = eps - # If false, skip the speaker-shuffle mixture while keeping the - # conditional-affine structure (default). self.shuffle_speakers = shuffle_speakers self.beta_distribution = torch.distributions.Beta( From 89fc557969757683dad16e3ef1fdcf4ed22330e1 Mon Sep 17 00:00:00 2001 From: KakaruHayate Date: Mon, 21 Sep 2026 15:27:34 +0800 Subject: [PATCH 22/23] fix: address CodeRabbit review findings from upstream sync - variance_encoder: MelodyEncoder honors melody_encoder_args.rope_theta override (fallback chain: enc_hparams -> hparams -> 10000) - kernels/integration: patch_diffusion_module degrades to eager with a warning when Triton is unavailable instead of raising RuntimeError, so use_fused_kernels=true never crashes task construction - acoustic_binarizer: cap k_mutate at len(aug_list) before building aug_types/aug_items so zip() no longer silently drops type-2 entries - acoustic_task: read resolved self.model.backbone_args instead of hparams['backbone_args'] (KeyError on legacy-only configs; may also disagree with the backbone actually constructed) --- modules/fastspeech/variance_encoder.py | 2 +- modules/kernels/integration.py | 12 ++++++++++++ preprocessing/acoustic_binarizer.py | 7 ++++++- training/acoustic_task.py | 2 +- 4 files changed, 20 insertions(+), 3 deletions(-) diff --git a/modules/fastspeech/variance_encoder.py b/modules/fastspeech/variance_encoder.py index 6e0aa79a2..a58f5d235 100644 --- a/modules/fastspeech/variance_encoder.py +++ b/modules/fastspeech/variance_encoder.py @@ -130,7 +130,7 @@ def get_hparam(key): dropout=get_hparam('dropout'), num_heads=get_hparam('num_heads'), use_pos_embed=get_hparam('use_pos_embed'), rel_pos=get_hparam('rel_pos'), use_rope=get_hparam('use_rope'), rope_interleaved=hparams.get('rope_interleaved', True), - rope_theta=hparams.get('rope_theta', 10000) + rope_theta=enc_hparams.get('rope_theta', hparams.get('rope_theta', 10000)) ) self.out_proj = Linear(hidden_size, hparams['hidden_size']) diff --git a/modules/kernels/integration.py b/modules/kernels/integration.py index a30abf445..d541afb5d 100644 --- a/modules/kernels/integration.py +++ b/modules/kernels/integration.py @@ -177,9 +177,21 @@ def patch_diffusion_module(diffusion, glu_type='softsign_glu'): GaussianDiffusion / PitchDiffusion / MultiVarianceDiffusion → .denoise_fn RectifiedFlow / PitchRectifiedFlow / MultiVarianceRectifiedFlow → .velocity_fn + Degrades gracefully to eager kernels when Triton is unavailable + (returns 0 with a warning instead of raising), so task construction + with use_fused_kernels=true never fails on Triton-less platforms. + Returns: Number of blocks patched. """ + if not is_triton_available(): + warnings.warn( + 'Fused kernels require a working Triton installation; ' + 'running eager. Install Triton for this platform or set ' + 'use_fused_kernels=false.', + stacklevel=2, + ) + return 0 return ( _try_patch(diffusion, 'denoise_fn', glu_type) + _try_patch(diffusion, 'velocity_fn', glu_type) diff --git a/preprocessing/acoustic_binarizer.py b/preprocessing/acoustic_binarizer.py index ba6fc4f4a..e8226e3f0 100644 --- a/preprocessing/acoustic_binarizer.py +++ b/preprocessing/acoustic_binarizer.py @@ -347,9 +347,14 @@ def arrange_data_augmentation(self, data_iterator): k_from_raw = int(scale / (1 + total_scale) * len(all_item_names)) k_from_aug = int(total_scale * scale / (1 + total_scale) * len(all_item_names)) k_mutate = int(total_scale * scale / (1 + scale) * len(all_item_names)) + # Cap k_mutate at len(aug_list): random.sample cannot return more + # distinct items than exist, and an uncapped count would leave + # aug_types/aug_items mismatched so zip() silently drops the extra + # type-2 entries. + k_mutate = min(k_mutate, len(aug_list)) aug_types = [0] * k_from_raw + [1] * k_from_aug + [2] * k_mutate aug_items = random.choices(all_item_names, k=k_from_raw) + \ - random.choices(aug_list, k=k_from_aug) + random.sample(aug_list, k=min(k_mutate, len(aug_list))) + random.choices(aug_list, k=k_from_aug) + random.sample(aug_list, k=k_mutate) for aug_type, aug_item in zip(aug_types, aug_items): # Uniform distribution in log domain diff --git a/training/acoustic_task.py b/training/acoustic_task.py index c4acc51c7..4c139d6bf 100644 --- a/training/acoustic_task.py +++ b/training/acoustic_task.py @@ -150,7 +150,7 @@ def __init__(self): # NOTE: LYNXNet2 defaults to swiglu when glu_type is unset self._fused_kernels_patched = patch_diffusion_module( self.model.diffusion, - glu_type=hparams['backbone_args'].get('glu_type', 'swiglu'), + glu_type=(self.model.backbone_args or {}).get('glu_type', 'swiglu'), ) rank_zero_info('Fused kernels: patched %d LYNXNet2 blocks', self._fused_kernels_patched) except ImportError as e: From 815a98c45405f25dbdd8b79440589eb5e5d3c167 Mon Sep 17 00:00:00 2001 From: KakaruHayate Date: Mon, 21 Sep 2026 18:32:04 +0800 Subject: [PATCH 23/23] docs: document config options missing from ConfigurationSchemas Audit of configs/templates/*.yaml against docs/ConfigurationSchemas.md found 18 undocumented keys, now added in alphabetical order with the standard entry format (visibility / scope / customizability / type / default / constraints): - mixln_shuffle_speakers (synced from openvpi PR #326): controls the cross-batch speaker shuffle of Mixed_LayerNorm affine params; default false = each speaker's params stay its own, true = original Mix-LN Beta-mix behavior - voicing_domain / voicing_mu (fork mulaw voicing): 'db' | 'amplitude' | 'mulaw' plus the mu-law parameter; non-default domains are exported into dsconfig for inference frontends - use_mouth_opening_embed / mouth_opening_estimator_ckpt / mouth_opening_smooth_width (fork SHMC mouth-opening conditioning) - use_shift_mouth_opening_embed + shift_mouth_opening_args.* (6 keys, fork SHMC shift self-distillation: teacher ckpt, alpha sigma, replacement prob, opec bounds, teacher AMP) - predict_mouth_opening / mouth_opening_min / mouth_opening_max (variance-side mouth opening with normalization range) - use_acoustic_retake (fork note-level condition-level inpainting) Template coverage is now complete: 0 undocumented keys in all three templates (was 17), documented keys 184 -> 202. --- docs/ConfigurationSchemas.md | 217 +++++++++++++++++++++++++++++++++++ 1 file changed, 217 insertions(+) diff --git a/docs/ConfigurationSchemas.md b/docs/ConfigurationSchemas.md index dd8a7d446..22f00d8e7 100644 --- a/docs/ConfigurationSchemas.md +++ b/docs/ConfigurationSchemas.md @@ -1276,6 +1276,66 @@ List of 0-based encoder layer indices where Mixed LayerNorm is applied. Only tak constraintsEvery element should be in the range [0, enc_layers). +### mixln_shuffle_speakers + +Whether to shuffle speaker embeddings across the batch when mixing the affine (beta/gamma) parameters of `Mixed_LayerNorm` during training. Only takes effect when [use_mix_ln](#use_mix_ln) is enabled. When `false` (default), each speaker's affine parameters are applied to its own samples only, so speaker conditioning stays consistent within a step. When `true`, the affine parameters are shuffled across the batch and mixed with weights sampled from a Beta distribution (the original Mix-LN behavior), which regularizes speaker conditioning at the cost of inter-speaker leakage within a training step. Has no effect in inference. + + + + + + + +
visibilityacoustic
scopenn, training
customizabilitynormal
typebool
defaultfalse
+ +### mouth_opening_estimator_ckpt + +Path to the checkpoint of the mouth-opening curve estimator (R3MOE). Binarizers use it to extract ground-truth mouth-opening curves whenever mouth opening is enabled — see [use_mouth_opening_embed](#use_mouth_opening_embed), [use_shift_mouth_opening_embed](#use_shift_mouth_opening_embed) and [predict_mouth_opening](#predict_mouth_opening). Required if any of them is enabled. + + + + + + + +
visibilityacoustic, variance
scopepreprocessing
customizabilityrecommended
typestr
defaultcheckpoints/r3moe/0508_s2k_noise_aug_0.15/ema_model_4.pt
+ +### mouth_opening_max + +Maximum mouth-opening value used for normalization to [-1, 1] when [predict_mouth_opening](#predict_mouth_opening) is enabled. Also serves as the upper bound of the mouth-opening range used by shift mouth-opening conditioning (see [shift_mouth_opening_args.opec_max](#shift_mouth_opening_argsopec_max)). Note that with [diffusion_type](#diffusion_type) `'ddpm'`, this value is latched into persistent buffers in checkpoints: modifying it for an existing experiment does not raise errors, but is silently overridden by the checkpoint on loading, so it only takes effect when training from scratch; with Rectified Flow it is always read from the current configuration. + + + + + + + +
visibilityvariance
scopetraining, inference
customizabilityrecommended
typefloat
default1.0
+ +### mouth_opening_min + +Minimum mouth-opening value used for normalization to [-1, 1] when [predict_mouth_opening](#predict_mouth_opening) is enabled. Also serves as the lower bound of the mouth-opening range used by shift mouth-opening conditioning (see [shift_mouth_opening_args.opec_min](#shift_mouth_opening_argsopec_min)). Note that with [diffusion_type](#diffusion_type) `'ddpm'`, this value is latched into persistent buffers in checkpoints: modifying it for an existing experiment does not raise errors, but is silently overridden by the checkpoint on loading, so it only takes effect when training from scratch; with Rectified Flow it is always read from the current configuration. + + + + + + + +
visibilityvariance
scopetraining, inference
customizabilityrecommended
typefloat
default0.0
+ +### mouth_opening_smooth_width + +Length of the sinusoidal smoothing convolution kernel (in seconds) applied to the extracted mouth-opening curve. Required whenever binarizers extract mouth-opening ground truth (see [mouth_opening_estimator_ckpt](#mouth_opening_estimator_ckpt)). + + + + + + + +
visibilityacoustic, variance
scopepreprocessing
customizabilitynormal
typefloat
default0.06
+ ### nccl_p2p Whether to enable P2P when using NCCL as the backend. Set it to `false` if the training process is stuck upon beginning. @@ -1631,6 +1691,18 @@ Whether to enable energy prediction. defaultfalse +### predict_mouth_opening + +Whether to enable mouth-opening prediction. When enabled, binarizers extract ground-truth mouth-opening curves with the estimator at [mouth_opening_estimator_ckpt](#mouth_opening_estimator_ckpt) (smoothed with [mouth_opening_smooth_width](#mouth_opening_smooth_width)) and normalize them with [mouth_opening_min](#mouth_opening_min) / [mouth_opening_max](#mouth_opening_max). Mouth opening counts as one of the variance parameters, so variances_prediction_args.total_repeat_bins must be divisible by the resulting number of predicted variances. + + + + + + + +
visibilityvariance
scopenn, preprocessing, training, inference
customizabilityrecommended
typebool
defaultfalse
+ ### predict_pitch Whether to enable pitch prediction. @@ -1843,6 +1915,88 @@ Whether to use the ground truth as `x_start` in the shallow diffusion validation defaultfalse +### shift_mouth_opening_args + +Arguments for shift mouth-opening conditioning (SHMC). Only takes effect when [use_shift_mouth_opening_embed](#use_shift_mouth_opening_embed) is enabled. + + + +
typedict[str, Any]
+ +### shift_mouth_opening_args.alpha_sigma + +Standard deviation of the zero-mean Gaussian (truncated to [-1, 1]) used to sample the per-sample shift strength alpha. A non-negative alpha shifts the mouth-opening curve toward the open end, a negative alpha toward the closed end; non-replaced samples always see alpha = 0. + + + + + + + +
visibilityacoustic
scopetraining
customizabilitynormal
typefloat
default0.5
+ +### shift_mouth_opening_args.opec_max + +Upper bound of the mouth-opening curve values used when computing the shifted curve. The shifted curve is clamped to [opec_min, opec_max]. + + + + + + + +
visibilityacoustic
scopetraining
customizabilitynormal
typefloat
default0.8
+ +### shift_mouth_opening_args.opec_min + +Lower bound of the mouth-opening curve values used when computing the shifted curve. The shifted curve is clamped to [opec_min, opec_max]. + + + + + + + +
visibilityacoustic
scopetraining
customizabilitynormal
typefloat
default0.06
+ +### shift_mouth_opening_args.replacement_prob + +Probability of each sample in a batch being replaced by the teacher's output (conditioned on the shifted mouth-opening curve) instead of using its own ground-truth mel as the training target. + + + + + + + + +
visibilityacoustic
scopetraining
customizabilitynormal
typefloat
default0.25
constraintsShould be in the range [0, 1].
+ +### shift_mouth_opening_args.teacher_ckpt_path + +Path to the teacher checkpoint used for shift self-distillation. The teacher must be an acoustic model trained with [use_mouth_opening_embed](#use_mouth_opening_embed) enabled, and its `config.yaml` must sit next to the checkpoint file. At startup the student validates the teacher's mel features, data space, dictionaries and phoneme settings against its own, and the teacher runs frozen (no gradients). + + + + + + + + +
visibilityacoustic
scopetraining
customizabilityrecommended
typestr
default''
constraintsMust point to an existing checkpoint file when use_shift_mouth_opening_embed is enabled.
+ +### shift_mouth_opening_args.teacher_use_amp + +Whether to run the teacher forward pass under float16 autocast on CUDA devices. Has no effect on CPU or when the input mel is not on CUDA. + + + + + + + +
visibilityacoustic
scopetraining
customizabilitynormal
typebool
defaulttrue
+ ### sort_by_len Whether to apply the _sorting by similar length_ algorithm described in [sampler_frame_count_grid](#sampler_frame_count_grid). Turning off this option may slow down training because sorting by length can better utilize the computing resources. @@ -1988,6 +2142,18 @@ Total number of DDPM steps. Only takes effect when [diffusion_type](#diffusion_t default1000 +### use_acoustic_retake + +Whether to enable note-level acoustic retake (condition-level / soft inpainting) during training. When enabled, a fresh continuous retake mask is sampled at each training step: in keep regions the ground-truth mel is fed back as a condition (normalized with [spec_min](#spec_min) / [spec_max](#spec_max) and projected into hidden space) so the model learns to reproduce it, while in retake regions the model conditions only on the encoder output. The option is baked into the exported ONNX graph, so it cannot be toggled at inference time. + + + + + + + +
visibilityacoustic
scopenn, training
customizabilitynormal
typebool
defaultfalse
+ ### use_breathiness_embed Whether to accept and embed breathiness values into the model. @@ -2098,6 +2264,19 @@ Whether to use Mixed LayerNorm with speaker-conditioned mixup in the acoustic en defaultfalse +### use_mouth_opening_embed + +Whether to accept and embed mouth-opening values into the acoustic model. When enabled, binarizers extract ground-truth mouth-opening curves with the estimator at [mouth_opening_estimator_ckpt](#mouth_opening_estimator_ckpt) (smoothed with [mouth_opening_smooth_width](#mouth_opening_smooth_width)), and the model takes `mouth_opening` as an additional variance input at inference. Mutually exclusive with [use_shift_mouth_opening_embed](#use_shift_mouth_opening_embed). + + + + + + + + +
visibilityacoustic
scopenn, preprocessing, training, inference
customizabilitynormal
typebool
defaultfalse
constraintsMutually exclusive with use_shift_mouth_opening_embed.
+ ### use_pos_embed Whether to enable positional encoding in FastSpeech2 encoder. When [use_rope](#use_rope) is `false`, this key controls the additive input embedding (`SinusoidalPositionalEmbedding` when `rel_pos` is `false`, or `RelPositionalEncoding` when `rel_pos` is `true`). When `use_rope` is `true`, no additive embedding is created, but RoPE is only created if this key is also `true` — disabling it removes RoPE as well and leaves the encoder with no positional encoding at all. The additive embedding module itself is created based on [use_rope](#use_rope) and [rel_pos](#rel_pos) alone, regardless of this key, so toggling it never changes parameter shapes or the set of saved keys and never prevents checkpoint loading; it only selects whether the positional encoding is actually applied at run time (and, when `use_rope` is `true`, whether RoPE is created and applied in attention), which changes the behavior of both training and inference. Since an already trained model expects its trained positional encoding scheme, modifying it silently produces inconsistent or wrong outputs. @@ -2134,6 +2313,19 @@ Whether to use shallow diffusion. defaulttrue +### use_shift_mouth_opening_embed + +Whether to enable shift mouth-opening conditioning (SHMC) in the acoustic model. Instead of feeding the mouth-opening curve directly, training shifts the curve by a sampled strength alpha and distills from a teacher (see [shift_mouth_opening_args](#shift_mouth_opening_args)): with probability `replacement_prob` a sample's target mel is replaced by the teacher's output conditioned on the shifted curve, while non-replaced samples see alpha = 0 so the model does not learn to associate non-zero alpha with ground truth. Mutually exclusive with [use_mouth_opening_embed](#use_mouth_opening_embed). + + + + + + + + +
visibilityacoustic
scopenn, preprocessing, training, inference
customizabilitynormal
typebool
defaultfalse
constraintsMutually exclusive with use_mouth_opening_embed; requires shift_mouth_opening_args.teacher_ckpt_path.
+ ### use_speed_embed Whether to embed speed values introduced by random time stretching augmentation. @@ -2321,6 +2513,31 @@ Minimum voicing value in dB used for normalization to [-1, 1]. Note that with [d default-96.0 +### voicing_domain + +Domain in which voicing values are represented. `'db'` (default) keeps the logarithmic dB representation bounded by [voicing_db_min](#voicing_db_min) / [voicing_db_max](#voicing_db_max). `'mulaw'` applies mu-law compression (see [voicing_mu](#voicing_mu)) and maps the result back into the dB-like range for API compatibility, in which case the upper normalization bound becomes 0 instead of [voicing_db_max](#voicing_db_max). `'amplitude'` uses the raw linear amplitude. Non-default domains are written into the exported `dsconfig` (`voicing_domain` and `voicing_mu`) so inference frontends can enable the corresponding domain conversion automatically. + + + + + + + + +
visibilityacoustic, variance
scopepreprocessing, training, inference
customizabilitynormal
typestr
default'db'
constraintsChoose from 'db', 'amplitude', 'mulaw'.
+ +### voicing_mu + +Mu parameter of the mu-law compression applied to voicing values. Only takes effect when [voicing_domain](#voicing_domain) is `'mulaw'`. Written into the exported `dsconfig` together with `voicing_domain` so inference frontends can invert the compression. + + + + + + + +
visibilityacoustic, variance
scopepreprocessing, training, inference
customizabilitynormal
typefloat
default255.0
+ ### voicing_smooth_width Length of sinusoidal smoothing convolution kernel (in seconds) on the extracted voicing curve.