diff --git a/README.md b/README.md index 09e8461..0d39d6a 100644 --- a/README.md +++ b/README.md @@ -122,6 +122,12 @@ bash example/infer.sh This script relies on metadata generated from the preprocessing pipeline, including vocal separation and transcription. Users should follow the steps in [preprocess](preprocess/README.md) to prepare the necessary metadata before running the demo with their own data. +**⚠️ Important Note** +The metadata produced by the automatic preprocessing pipeline may not perfectly align the singing audio with the corresponding lyrics and musical notes. For best synthesis quality, we strongly recommend manually correcting the alignment using the 🎼 [Midi-Editor](https://huggingface.co/spaces/Soul-AILab/SoulX-Singer-Midi-Editor). + +How to use the Midi-Editor: +- [Eiditing Metadata with Midi-Editor](preprocess/README.md#L104-L105) + ## 🚧 Roadmap diff --git a/soulxsinger/models/soulxsinger.py b/soulxsinger/models/soulxsinger.py index 989c056..18aff6c 100644 --- a/soulxsinger/models/soulxsinger.py +++ b/soulxsinger/models/soulxsinger.py @@ -140,7 +140,7 @@ class SoulXSinger(nn.Module): print("Warning: pitch_shift is True but note_pitch or f0 is None. Set f0_shift to 0.") f0_shift = 0 else: - f0_shift = 0 + f0_shift = pitch_shift if gt_f0 is None or pt_f0 is None: gt_f0, pt_f0 = torch.zeros_like(gt_mel2note).float().to(gt_mel2note.device), torch.zeros_like(pt_mel2note).float().to(pt_mel2note.device)